{
  "slug": "credit-assignment",
  "focus": "literature",
  "title": "Credit assignment in agent reinforcement learning",
  "query": "\"credit assignment\"",
  "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/credit-assignment.html",
  "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/credit-assignment.json",
  "paper_count": 3,
  "matching_block_count": 24,
  "all_matching_blocks_included": true,
  "corpus": {
    "scope": {
      "declared_source_count": 13,
      "indexed_source_count": 8,
      "metadata_only_source_count": 5,
      "comprehensive": false,
      "query_language": "en",
      "topics": [
        "agent_reinforcement_learning",
        "credit_assignment",
        "cli_agents",
        "software_benchmarks"
      ]
    },
    "source_catalog": [
      {
        "arxiv": "2405.15793",
        "arxiv_version": "2405.15793v1",
        "paper_url": "https://arxiv.org/abs/2405.15793v1",
        "html_url": "https://arxiv.org/html/2405.15793v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering",
        "authors": [
          "Yang, John",
          "Jimenez, Carlos E.",
          "Wettig, Alexander",
          "Lieret, Kilian",
          "Yao, Shunyu",
          "Narasimhan, Karthik",
          "Press, Ofir"
        ],
        "citation_date": "2024/05/06",
        "evidence_count": 91
      },
      {
        "arxiv": "2310.06770",
        "arxiv_version": "2310.06770v1",
        "paper_url": "https://arxiv.org/abs/2310.06770v1",
        "html_url": "https://arxiv.org/html/2310.06770v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "SWE-bench: Can Language Models Resolve Real-World GitHub Issues?",
        "authors": [
          "Jimenez, Carlos E.",
          "Yang, John",
          "Wettig, Alexander",
          "Yao, Shunyu",
          "Pei, Kexin",
          "Press, Ofir",
          "Narasimhan, Karthik"
        ],
        "citation_date": "2023/10/10",
        "evidence_count": 135
      },
      {
        "arxiv": "2601.11868",
        "arxiv_version": "2601.11868v1",
        "paper_url": "https://arxiv.org/abs/2601.11868v1",
        "html_url": "https://arxiv.org/html/2601.11868v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces",
        "authors": [
          "Merrill, Mike A.",
          "Shaw, Alexander G.",
          "Carlini, Nicholas",
          "Li, Boxuan",
          "Raj, Harsh",
          "Bercovich, Ivan",
          "Shi, Lin",
          "Shin, Jeong Yeon",
          "Walshe, Thomas",
          "Buchanan, E. Kelly",
          "Shen, Junhong",
          "Ye, Guanghao",
          "Lin, Haowei",
          "Poulos, Jason",
          "Wang, Maoyu",
          "Nezhurina, Marianna",
          "Jitsev, Jenia",
          "Lu, Di",
          "Mastromichalakis, Orfeas Menis",
          "Xu, Zhiwei",
          "Chen, Zizhao",
          "Liu, Yue",
          "Zhang, Robert",
          "Chen, Leon Liangyu",
          "Kashyap, Anurag",
          "Uslu, Jan-Lucas",
          "Li, Jeffrey",
          "Wu, Jianbo",
          "Yan, Minghao",
          "Bian, Song",
          "Sharma, Vedang",
          "Sun, Ke",
          "Dillmann, Steven",
          "Anand, Akshay",
          "Lanpouthakoun, Andrew",
          "Koopah, Bardia",
          "Hu, Changran",
          "Guha, Etash",
          "Dreiman, Gabriel H. S.",
          "Zhu, Jiacheng",
          "Krauth, Karl",
          "Zhong, Li",
          "Muennighoff, Niklas",
          "Amanfu, Robert",
          "Tan, Shangyin",
          "Pimpalgaonkar, Shreyas",
          "Aggarwal, Tushar",
          "Lin, Xiangning",
          "Lan, Xin",
          "Zhao, Xuandong",
          "Liang, Yiqing",
          "Wang, Yuanli",
          "Wang, Zilong",
          "Zhou, Changzhi",
          "Heineman, David",
          "Liu, Hange",
          "Trivedi, Harsh",
          "Yang, John",
          "Lin, Junhong",
          "Shetty, Manish",
          "Yang, Michael",
          "Omi, Nabil",
          "Raoof, Negin",
          "Li, Shanda",
          "Zhuo, Terry Yue",
          "Lin, Wuwei",
          "Dai, Yiwei",
          "Wang, Yuxin",
          "Chai, Wenhao",
          "Zhou, Shang",
          "Wahdany, Dariush",
          "She, Ziyu",
          "Hu, Jiaming",
          "Dong, Zhikang",
          "Zhu, Yuxuan",
          "Cui, Sasha",
          "Saiyed, Ahson",
          "Kolbeinsson, Arinbjörn",
          "Hu, Jesse",
          "Rytting, Christopher Michael",
          "Marten, Ryan",
          "Wang, Yixin",
          "Dimakis, Alex",
          "Konwinski, Andy",
          "Schmidt, Ludwig"
        ],
        "citation_date": "2026/01/17",
        "evidence_count": 160
      },
      {
        "arxiv": "2607.22724",
        "arxiv_version": "2607.22724v1",
        "paper_url": "https://arxiv.org/abs/2607.22724v1",
        "html_url": "https://arxiv.org/html/2607.22724v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "Progress-conditioned Group Policy Optimization for Long-Horizon Agentic Tasks",
        "authors": [
          "Yang, Kaibing",
          "Cai, Guangfeng",
          "Yang, Shengtian",
          "He, Shuo",
          "Li, Yu",
          "Liu, Mengyi",
          "Chen, Pengwei",
          "Xu, Jun",
          "Feng, Lei"
        ],
        "citation_date": "2026/07/22",
        "evidence_count": 0
      },
      {
        "arxiv": "2505.11821",
        "arxiv_version": "2505.11821v1",
        "paper_url": "https://arxiv.org/abs/2505.11821v1",
        "html_url": "https://arxiv.org/html/2505.11821v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
        "authors": [
          "Zeng, Siliang",
          "Wei, Quan",
          "Brown, William",
          "Frunza, Oana",
          "Nevmyvaka, Yuriy",
          "Hong, Mingyi"
        ],
        "citation_date": "2025/05/17",
        "evidence_count": 51
      },
      {
        "arxiv": "2402.03300",
        "arxiv_version": "2402.03300v1",
        "paper_url": "https://arxiv.org/abs/2402.03300v1",
        "html_url": "https://arxiv.org/html/2402.03300v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models",
        "authors": [
          "Shao, Zhihong",
          "Wang, Peiyi",
          "Zhu, Qihao",
          "Xu, Runxin",
          "Song, Junxiao",
          "Zhang, Mingchuan",
          "Li, Y. K.",
          "Wu, Y.",
          "Guo, Daya"
        ],
        "citation_date": "2024/02/05",
        "evidence_count": 0
      },
      {
        "arxiv": "2602.22817",
        "arxiv_version": "2602.22817v1",
        "paper_url": "https://arxiv.org/abs/2602.22817v1",
        "html_url": "https://arxiv.org/html/2602.22817v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Hierarchy-of-Groups Policy Optimization for Long-Horizon Agentic Tasks",
        "authors": [
          "He, Shuo",
          "Feng, Lang",
          "Wei, Qi",
          "Cheng, Xin",
          "Feng, Lei",
          "An, Bo"
        ],
        "citation_date": "2026/02/26",
        "evidence_count": 64
      },
      {
        "arxiv": "2407.16741",
        "arxiv_version": "2407.16741v1",
        "paper_url": "https://arxiv.org/abs/2407.16741v1",
        "html_url": "https://arxiv.org/html/2407.16741v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "OpenDevin: An Open Platform for AI Software Developers as Generalist Agents",
        "authors": [
          "Wang, Xingyao",
          "Li, Boxuan",
          "Song, Yufan",
          "Xu, Frank F.",
          "Tang, Xiangru",
          "Zhuge, Mingchen",
          "Pan, Jiayi",
          "Song, Yueqi",
          "Li, Bowen",
          "Singh, Jaskirat",
          "Tran, Hoang H.",
          "Li, Fuqiang",
          "Ma, Ren",
          "Zheng, Mingzhang",
          "Qian, Bill",
          "Shao, Yanjun",
          "Muennighoff, Niklas",
          "Zhang, Yizhe",
          "Hui, Binyuan",
          "Lin, Junyang",
          "Brennan, Robert",
          "Peng, Hao",
          "Ji, Heng",
          "Neubig, Graham"
        ],
        "citation_date": "2024/07/23",
        "evidence_count": 107
      },
      {
        "arxiv": "2502.18449",
        "arxiv_version": "2502.18449v1",
        "paper_url": "https://arxiv.org/abs/2502.18449v1",
        "html_url": "https://arxiv.org/html/2502.18449v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "SWE-RL : Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution",
        "authors": [
          "Wei, Yuxiang",
          "Duchenne, Olivier",
          "Copet, Jade",
          "Carbonneaux, Quentin",
          "Zhang, Lingming",
          "Fried, Daniel",
          "Synnaeve, Gabriel",
          "Singh, Rishabh",
          "Wang, Sida I."
        ],
        "citation_date": "2025/02/25",
        "evidence_count": 0
      },
      {
        "arxiv": "2605.08013",
        "arxiv_version": "2605.08013v1",
        "paper_url": "https://arxiv.org/abs/2605.08013v1",
        "html_url": "https://arxiv.org/html/2605.08013v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
        "authors": [
          "Su, Haoyang",
          "Wen, Ying"
        ],
        "citation_date": "2026/05/08",
        "evidence_count": 74
      },
      {
        "arxiv": "2503.09516",
        "arxiv_version": "2503.09516v1",
        "paper_url": "https://arxiv.org/abs/2503.09516v1",
        "html_url": "https://arxiv.org/html/2503.09516v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning",
        "authors": [
          "Jin, Bowen",
          "Zeng, Hansi",
          "Yue, Zhenrui",
          "Wang, Dong",
          "Zamani, Hamed",
          "Han, Jiawei"
        ],
        "citation_date": "2025/03/12",
        "evidence_count": 0
      },
      {
        "arxiv": "2504.20073",
        "arxiv_version": "2504.20073v1",
        "paper_url": "https://arxiv.org/abs/2504.20073v1",
        "html_url": "https://arxiv.org/html/2504.20073v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning",
        "authors": [
          "Wang, Zihan",
          "Wang, Kangrui",
          "Wang, Qineng",
          "Zhang, Pingyue",
          "Li, Linjie",
          "Yang, Zhengyuan",
          "Yu, Kefan",
          "Nguyen, Minh Nhat",
          "Liu, Licheng",
          "Gottlieb, Eli",
          "Lam, Monica",
          "Lu, Yiping",
          "Cho, Kyunghyun",
          "Wu, Jiajun",
          "Fei-Fei, Li",
          "Wang, Lijuan",
          "Choi, Yejin",
          "Li, Manling"
        ],
        "citation_date": "2025/04/24",
        "evidence_count": 115
      },
      {
        "arxiv": "2505.10978",
        "arxiv_version": "2505.10978v1",
        "paper_url": "https://arxiv.org/abs/2505.10978v1",
        "html_url": "https://arxiv.org/html/2505.10978v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "Group-in-Group Policy Optimization for LLM Agent Training",
        "authors": [
          "Feng, Lang",
          "Xue, Zhenghai",
          "Liu, Tingcong",
          "An, Bo"
        ],
        "citation_date": "2025/05/16",
        "evidence_count": 0
      }
    ],
    "service_links": {
      "Topics": "https://hoyant-su-agentic-rl.hf.space/topics/index.html",
      "Comparisons": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/index.html",
      "Method filters": "https://hoyant-su-agentic-rl.hf.space/methods/facets",
      "Method catalog": "https://hoyant-su-agentic-rl.hf.space/topics/methods.json",
      "MCP": "https://hoyant-su-agentic-rl.hf.space/gradio_api/mcp/",
      "Tool schema": "https://hoyant-su-agentic-rl.hf.space/gradio_api/mcp/schema",
      "Skill index": "https://hoyant-su-agentic-rl.hf.space/.well-known/agent-skills/index.json"
    }
  },
  "papers": [
    {
      "source": {
        "arxiv": "2505.11821",
        "arxiv_version": "2505.11821v1",
        "paper_url": "https://arxiv.org/abs/2505.11821v1",
        "html_url": "https://arxiv.org/html/2505.11821v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
        "authors": [
          "Zeng, Siliang",
          "Wei, Quan",
          "Brown, William",
          "Frunza, Oana",
          "Nevmyvaka, Yuriy",
          "Hong, Mingyi"
        ],
        "citation_date": "2025/05/17",
        "evidence_count": 51
      },
      "matching_block_count": 16,
      "representative_block": {
        "evidence_id": "2505.11821v1:S1.p5",
        "arxiv": "2505.11821",
        "arxiv_version": "2505.11821v1",
        "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
        "authors": [
          "Zeng, Siliang",
          "Wei, Quan",
          "Brown, William",
          "Frunza, Oana",
          "Nevmyvaka, Yuriy",
          "Hong, Mingyi"
        ],
        "citation_date": "2025/05/17",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "source_url": "https://arxiv.org/html/2505.11821v1#S1.p5",
        "paper_url": "https://arxiv.org/abs/2505.11821v1",
        "anchor": "S1.p5",
        "section": "1 Introduction",
        "section_url": "https://arxiv.org/html/2505.11821v1#S1",
        "block_classes": [
          "ltx_para"
        ],
        "text": "Inspired by recent work on credit assignment ( Pignatelli et al., 2023 ) for pure text reasoning tasks ( Shao et al., 2024 ; Cui et al., 2025 ; Cheng et al., 2025 ) , in this paper, we introduce a fine-grained turn-level credit assignment strategy for multi-turn LLM agent training. Compared with textual reasoning tasks like mathematical problem solving, multi-turn agent interactive tasks present a more intuitive setting to highlight the importance of fine-grained credit assignment. The key contributions are as follows: • We propose modeling multi-turn long-horizon reasoning tasks in LLM agents as Markov Decision Processes (MDPs), which naturally capture the sequential decision-making structure of such problems. To train multi-turn LLM agents effectively within the MDP framework, we present a fine-grained turn-level advantage estimation strategy using both outcome and turn-level rewards. In this work, we instantiate our approach within the GRPO algorithm. Notably, our strategy is general and can be compatible with a wide range of RL methods. • To highlight the importance of credit assignment mechanisms in multi-turn reasoning, we construct an agent that performs question answering using a Wikipedia search tool. The agent operates in multiple steps: reasoning, search, and answer summarization. It learns to leverage the Wikipedia search engine to retrieve relevant information in support of its final answer through RL training. Figure 1 illustrates the multi-turn agent workflow and compares baselines of trajectory-level advantage estimation with our proposed GRPO-based variant. • Experimental results on multi-turn reasoning and search tasks show that compared with baselines using trajectory-level advantage estimation, our MDP formulation and fine-grained turn-level credit assignment significantly improve the multi-turn reasoning performance of LLM agents in complex decision-making tasks. In particular, our method achieves 100% success in tool invocation and 50% accuracy in exact answer matching, significantly outperforming baselines, which fail to invoke tools and achieve only 20–30% exact match accuracy. Additionally, we find that our method promotes more stable and consistent tool use during training, whereas baselines with coarse-grained trajectory-level credit assignment often forget to call tools and exhibit higher variance. These findings further highlight the critical role of precise credit assignment in effective multi-turn agent training.",
        "equations": [],
        "tables": [],
        "links": [
          {
            "text": "Pignatelli et al., 2023",
            "url": "https://arxiv.org/html/2505.11821v1#bib.bib26"
          },
          {
            "text": "Shao et al., 2024",
            "url": "https://arxiv.org/html/2505.11821v1#bib.bib32"
          },
          {
            "text": "Cui et al., 2025",
            "url": "https://arxiv.org/html/2505.11821v1#bib.bib9"
          },
          {
            "text": "Cheng et al., 2025",
            "url": "https://arxiv.org/html/2505.11821v1#bib.bib8"
          },
          {
            "text": "1",
            "url": "https://arxiv.org/html/2505.11821v1#S0.F1"
          }
        ],
        "untranscribed_graphics": 0,
        "bm25_score": -5.254984706141114
      },
      "matching_blocks": [
        {
          "evidence_id": "2505.11821v1:S1.p4",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S1.p4",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S1.p4",
          "section": "1 Introduction",
          "section_url": "https://arxiv.org/html/2505.11821v1#S1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "While recent studies ( Li et al., 2025 ; Qian et al., 2025 ; Wang et al., 2025a ; Labs, 2025 ; Wang et al., 2025b ; Zhang et al., 2025 ; Singh et al., 2025 ) incorporate turn-level rewards like tool execution, they still treat agent tasks as bandit problems and estimate advantages at the trajectory level by merging outcome and turn-level rewards, which lacks fine-grained credit assignment . When the rewards are used to assign credit across an entire trajectory, it becomes difficult to identify which specific decisions contributed positively or negatively to the final result. Effective multi-turn reasoning requires more precise, turn-level credit assignment to enable the agent to refine individual steps, rather than treating all actions as equally responsible for success or failure. The lack of fine-grained credit assignment ultimately limits the performance and adaptability of multi-turn LLM agents.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "Li et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib20"
            },
            {
              "text": "Qian et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib27"
            },
            {
              "text": "Wang et al., 2025a",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib37"
            },
            {
              "text": "Labs, 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib19"
            },
            {
              "text": "Wang et al., 2025b",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib39"
            },
            {
              "text": "Zhang et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib51"
            },
            {
              "text": "Singh et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib34"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -5.209748547315273
        },
        {
          "evidence_id": "2505.11821v1:S1.p5",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S1.p5",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S1.p5",
          "section": "1 Introduction",
          "section_url": "https://arxiv.org/html/2505.11821v1#S1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Inspired by recent work on credit assignment ( Pignatelli et al., 2023 ) for pure text reasoning tasks ( Shao et al., 2024 ; Cui et al., 2025 ; Cheng et al., 2025 ) , in this paper, we introduce a fine-grained turn-level credit assignment strategy for multi-turn LLM agent training. Compared with textual reasoning tasks like mathematical problem solving, multi-turn agent interactive tasks present a more intuitive setting to highlight the importance of fine-grained credit assignment. The key contributions are as follows: • We propose modeling multi-turn long-horizon reasoning tasks in LLM agents as Markov Decision Processes (MDPs), which naturally capture the sequential decision-making structure of such problems. To train multi-turn LLM agents effectively within the MDP framework, we present a fine-grained turn-level advantage estimation strategy using both outcome and turn-level rewards. In this work, we instantiate our approach within the GRPO algorithm. Notably, our strategy is general and can be compatible with a wide range of RL methods. • To highlight the importance of credit assignment mechanisms in multi-turn reasoning, we construct an agent that performs question answering using a Wikipedia search tool. The agent operates in multiple steps: reasoning, search, and answer summarization. It learns to leverage the Wikipedia search engine to retrieve relevant information in support of its final answer through RL training. Figure 1 illustrates the multi-turn agent workflow and compares baselines of trajectory-level advantage estimation with our proposed GRPO-based variant. • Experimental results on multi-turn reasoning and search tasks show that compared with baselines using trajectory-level advantage estimation, our MDP formulation and fine-grained turn-level credit assignment significantly improve the multi-turn reasoning performance of LLM agents in complex decision-making tasks. In particular, our method achieves 100% success in tool invocation and 50% accuracy in exact answer matching, significantly outperforming baselines, which fail to invoke tools and achieve only 20–30% exact match accuracy. Additionally, we find that our method promotes more stable and consistent tool use during training, whereas baselines with coarse-grained trajectory-level credit assignment often forget to call tools and exhibit higher variance. These findings further highlight the critical role of precise credit assignment in effective multi-turn agent training.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "Pignatelli et al., 2023",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib26"
            },
            {
              "text": "Shao et al., 2024",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib32"
            },
            {
              "text": "Cui et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib9"
            },
            {
              "text": "Cheng et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib8"
            },
            {
              "text": "1",
              "url": "https://arxiv.org/html/2505.11821v1#S0.F1"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -5.254984706141114
        },
        {
          "evidence_id": "2505.11821v1:S2.SS2.p3",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S2.SS2.p3",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S2.SS2.p3",
          "section": "2 Related Work / 2.2 RL for LLMs",
          "section_url": "https://arxiv.org/html/2505.11821v1#S2.SS2",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Recently, the credit assignment problem ( Pignatelli et al., 2023 ) in RL has received increasing attention in the context of LLM reasoning ( Shao et al., 2024 ; Cui et al., 2025 ; Cheng et al., 2025 ) . Dense process rewards offer an appealing alternative to sparse outcome-level rewards for training LLMs with RL. PRIME ( Cui et al., 2025 ) proposes online process reward model (PRM) updates using only policy rollouts and outcome labels through implicit process rewards. By fusing token-level dense implicit process rewards with sparse outcome rewards to estimate advantages, PRIME boosts the performance of various RL algorithms, including GRPO and RLOO. PURE ( Cheng et al., 2025 ) identifies that summation-based credit assignment can cause LLMs to exploit high-reward steps, leading them to prioritize verbose thinking over actual problem-solving. To address this, PURE introduces min-form credit assignment, which mitigates reward hacking associated with PRMs.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "Pignatelli et al., 2023",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib26"
            },
            {
              "text": "Shao et al., 2024",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib32"
            },
            {
              "text": "Cui et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib9"
            },
            {
              "text": "Cheng et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib8"
            },
            {
              "text": "Cui et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib9"
            },
            {
              "text": "Cheng et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib8"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -5.17372232184783
        },
        {
          "evidence_id": "2505.11821v1:S2.SS3.p2",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S2.SS3.p2",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S2.SS3.p2",
          "section": "2 Related Work / 2.3 RL for LLM Agents",
          "section_url": "https://arxiv.org/html/2505.11821v1#S2.SS3",
          "block_classes": [
            "ltx_para"
          ],
          "text": "However, these approaches typically formulate the agent tasks as bandit problems even when turn-level rewards are involved ( Li et al., 2025 ; Qian et al., 2025 ; Wang et al., 2025a ; Labs, 2025 ; Wang et al., 2025b ; Zhang et al., 2025 ; Singh et al., 2025 ) . They compute advantages at the trajectory level by summing outcome and turn-level rewards. None of these methods considers fine-grained turn-level credit assignment across multiple decision steps to enhance multi-turn reasoning in LLM agents.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "Li et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib20"
            },
            {
              "text": "Qian et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib27"
            },
            {
              "text": "Wang et al., 2025a",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib37"
            },
            {
              "text": "Labs, 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib19"
            },
            {
              "text": "Wang et al., 2025b",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib39"
            },
            {
              "text": "Zhang et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib51"
            },
            {
              "text": "Singh et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib34"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -3.9625270218985644
        },
        {
          "evidence_id": "2505.11821v1:S3.SS1.p1",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S3.SS1.p1",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S3.SS1.p1",
          "section": "3 Multi-Turn Tool-Calling LLM Agent System / 3.1 Task Formulation",
          "section_url": "https://arxiv.org/html/2505.11821v1#S3.SS1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "To emphasize the importance of fine-grained credit assignment in multi-turn agent interactions, we formulate the task under the MDP framework, involving multiple steps of reasoning, tool use, and answer summarization for question answering. Specifically, our tool-use environment is modeled on a Wikipedia search setup, where the agent learns to leverage a Wikipedia search engine to retrieve relevant information and generate accurate answers. The goal is to improve the agent’s performance through effective integration of external tool use. Without tool calling, the agent must rely solely on its internal knowledge to answer questions, which can limit accuracy, especially for fact-based queries requiring up-to-date or domain-specific information.",
          "equations": [],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -3.515650026216531
        },
        {
          "evidence_id": "2505.11821v1:S3.SS1.p2",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S3.SS1.p2",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S3.SS1.p2",
          "section": "3 Multi-Turn Tool-Calling LLM Agent System / 3.1 Task Formulation",
          "section_url": "https://arxiv.org/html/2505.11821v1#S3.SS1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "To clearly illustrate the impact of credit assignment, we design a simplified two-turn tool-use environment in which the LLM agent can interact with the search tool environment for a maximum of two turns. In this setup, the agent is allowed to call the Wikipedia search engine at most once before submitting an answer to the question. Figure 1 illustrates the pipeline of the multi-turn, tool-calling LLM agent system. Given a system prompt and a question, the LLM agent first performs a reasoning step and issues a tool call, specifying both the tool name and a query derived from its reasoning. The external tool environment processes the query and returns a search result. Based on the retrieved result, the agent performs a second round of reasoning to summarize the information and generate the final answer. The whole process can be summarized as $\\texttt{reasoning}\\rightarrow\\texttt{search}\\rightarrow\\texttt{result}\\rightarrow\\texttt{reasoning}\\rightarrow\\texttt{answer}$ These steps are explicitly outlined in the system prompt, which also enforces strict constraints, such as allowing only a single tool invocation and requiring the use of specific XML-like tags (e.g., <reasoning> , <tool> , <result> , <answer> ) to delineate each stage of the interaction. The full system prompt is provided in Appendix A . Table 1 presents an example rollout in which the agent successfully calls the search tool. If the tool name or argument format is incorrect, the tool environment returns an error message, indicated by the response beginning with “Error:”. If the agent fails to include a tool-calling command in the first reasoning step, the tool environment will not be invoked. If the XML format or tag usage is incorrect—for example, if tags are missing, nested improperly, or misnamed—the environment may fail to parse the agent’s response, resulting in an error or a skipped tool invocation. Additional rollout examples where the agent fails to call the tool correctly are provided in Appendix B .",
          "equations": [
            {
              "anchor": "S3.Ex1.m1",
              "latex": "\\texttt{reasoning}\\rightarrow\\texttt{search}\\rightarrow\\texttt{result}\\rightarrow\\texttt{reasoning}\\rightarrow\\texttt{answer}",
              "display": "block"
            }
          ],
          "tables": [],
          "links": [
            {
              "text": "1",
              "url": "https://arxiv.org/html/2505.11821v1#S0.F1"
            },
            {
              "text": "A",
              "url": "https://arxiv.org/html/2505.11821v1#A1"
            },
            {
              "text": "1",
              "url": "https://arxiv.org/html/2505.11821v1#S3.T1"
            },
            {
              "text": "B",
              "url": "https://arxiv.org/html/2505.11821v1#A2"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -2.0041467842281477
        },
        {
          "evidence_id": "2505.11821v1:S3.p1",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S3.p1",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S3.p1",
          "section": "3 Multi-Turn Tool-Calling LLM Agent System",
          "section_url": "https://arxiv.org/html/2505.11821v1#S3",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Before presenting our fine-grained turn-level credit assignment for various RL algorithms, we first describe the experimental environment of the multi-turn tool-calling LLM agent system.",
          "equations": [],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -5.019206638704204
        },
        {
          "evidence_id": "2505.11821v1:S4.SS2.p2",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S4.SS2.p2",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S4.SS2.p2",
          "section": "4 Methodology / 4.2 Proposed Method: Turn-Level Credit Assignment for Multi-Turn LLM Agents",
          "section_url": "https://arxiv.org/html/2505.11821v1#S4.SS2",
          "block_classes": [
            "ltx_para"
          ],
          "text": "We now introduce the detailed implementations of turn-level advantage estimation tailored to our two-turn LLM agent setting. We adopt GPRO as a representative algorithm to derive the turn-level credit assignment strategy, referring to the resulting approach as Multi-Turn GPRO (MT-GPRO). Inspired by ( Cui et al., 2025 ; Cheng et al., 2025 ) , in MT-GPRO, the advantages in the first and second turns can be computed as $\\hat{A}^{\\text{MT-GRPO}}_{i,1}=\\hat{A}^{T}_{i}+\\lambda\\hat{A}^{O}_{i},\\quad\\hat{A}^{\\text{MT-GRPO}}_{i,2}=\\hat{A}^{O}_{i}.$ (9) where $\\lambda$ is the turn-level advantage coefficient, $\\hat{A}^{T}_{i}$ and $\\hat{A}^{O}_{i}$ are computed: $\\hat{A}^{T}_{i}=\\frac{R^{T}_{i}-\\text{mean}(\\{R^{T}_{i}\\}_{i=1}^{G})}{\\text{std}(\\{R^{T}_{i}\\}_{i=1}^{G})},\\quad\\hat{A}^{O}_{i}=\\frac{R^{O}_{i}-\\text{mean}(\\{R^{O}_{i}\\}_{i=1}^{G})}{\\text{std}(\\{R^{O}_{i}\\}_{i=1}^{G})}.$ (10)",
          "equations": [
            {
              "anchor": "S4.E9.m1",
              "latex": "\\hat{A}^{\\text{MT-GRPO}}_{i,1}=\\hat{A}^{T}_{i}+\\lambda\\hat{A}^{O}_{i},\\quad\\hat{A}^{\\text{MT-GRPO}}_{i,2}=\\hat{A}^{O}_{i}.",
              "display": "block"
            },
            {
              "anchor": "S4.SS2.p2.m1",
              "latex": "\\lambda",
              "display": "inline"
            },
            {
              "anchor": "S4.SS2.p2.m2",
              "latex": "\\hat{A}^{T}_{i}",
              "display": "inline"
            },
            {
              "anchor": "S4.SS2.p2.m3",
              "latex": "\\hat{A}^{O}_{i}",
              "display": "inline"
            },
            {
              "anchor": "S4.E10.m1",
              "latex": "\\hat{A}^{T}_{i}=\\frac{R^{T}_{i}-\\text{mean}(\\{R^{T}_{i}\\}_{i=1}^{G})}{\\text{std}(\\{R^{T}_{i}\\}_{i=1}^{G})},\\quad\\hat{A}^{O}_{i}=\\frac{R^{O}_{i}-\\text{mean}(\\{R^{O}_{i}\\}_{i=1}^{G})}{\\text{std}(\\{R^{O}_{i}\\}_{i=1}^{G})}.",
              "display": "block"
            }
          ],
          "tables": [],
          "links": [
            {
              "text": "Cui et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib9"
            },
            {
              "text": "Cheng et al., 2025",
              "url": "https://arxiv.org/html/2505.11821v1#bib.bib8"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -2.944841559304301
        },
        {
          "evidence_id": "2505.11821v1:S4.p1",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S4.p1",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S4.p1",
          "section": "4 Methodology",
          "section_url": "https://arxiv.org/html/2505.11821v1#S4",
          "block_classes": [
            "ltx_para"
          ],
          "text": "In this section, we first review existing trajectory-level advantage estimation implementations for multi-turn LLM agent training and discuss their limitations, and then present the fine-grained turn-level credit assignment for various RL algorithms.",
          "equations": [],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -4.824992065804896
        },
        {
          "evidence_id": "2505.11821v1:S5.SS1.p1",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S5.SS1.p1",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S5.SS1.p1",
          "section": "5 Experiments / 5.1 Evaluated Methods",
          "section_url": "https://arxiv.org/html/2505.11821v1#S5.SS1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "We compare our proposed MT-GPRO with vanilla GRPO. • GRPO : original GRPO with trajectory-level advantage estimation – GRPO-OR : GRPO using only outcome rewards – GRPO-MR : GRPO using merged outcome and turn-level rewards • MT-GRPO (ours): GPRO variant with turn-level advantage estimation using both outcome and turn-level rewards These configurations allow us to assess the influence of turn-level verifiable rewards and credit assignment on the dynamics of the LLM agent.",
          "equations": [],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -4.075858223536623
        },
        {
          "evidence_id": "2505.11821v1:S5.SS3.p1",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S5.SS3.p1",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S5.SS3.p1",
          "section": "5 Experiments / 5.3 Main Results",
          "section_url": "https://arxiv.org/html/2505.11821v1#S5.SS3",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Figure 2 shows reward component curves during training across various algorithms. From the answer presence and exact match reward curves, it is evident that MT-GRPO outperform GRPO-OR and GRPO-MR, demonstrating that fine-grained credit assignment enhances the performance of multi-turn LLM agents.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "2",
              "url": "https://arxiv.org/html/2505.11821v1#S5.F2"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -4.6023846324847355
        },
        {
          "evidence_id": "2505.11821v1:S5.SS3.p3",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S5.SS3.p3",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S5.SS3.p3",
          "section": "5 Experiments / 5.3 Main Results",
          "section_url": "https://arxiv.org/html/2505.11821v1#S5.SS3",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Figures 3 , 4 , and 5 in Appendix D illustrate reward component curves during training with different algorithms, where shaded regions represent the range between the maximum and minimum values across 10 runs, showcasing the variability in learning performance. Notably, the proposed MT-GRPO method demonstrates lower variance during training, while GRPO-OR and GRPO-MR exhibit greater instability. An interesting observation is that the tool execution curve of MT-GRPO temporarily drops sharply around step 230–250 but subsequently recovers and stabilizes. This demonstrates that even if the agent forgets to call search tools in the middle of the training, it eventually learns to incorporate them in the final stages. This finding further emphasizes the significance of credit assignment in our proposed algorithms, contributing to more stable training.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "3",
              "url": "https://arxiv.org/html/2505.11821v1#A4.F3"
            },
            {
              "text": "4",
              "url": "https://arxiv.org/html/2505.11821v1#A4.F4"
            },
            {
              "text": "5",
              "url": "https://arxiv.org/html/2505.11821v1#A4.F5"
            },
            {
              "text": "D",
              "url": "https://arxiv.org/html/2505.11821v1#A4"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -3.3391313948821932
        },
        {
          "evidence_id": "2505.11821v1:S5.SS3.p4",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S5.SS3.p4",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S5.SS3.p4",
          "section": "5 Experiments / 5.3 Main Results",
          "section_url": "https://arxiv.org/html/2505.11821v1#S5.SS3",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Table 2 presents the validation reward scores across different models. MT-GRPO achieves the highest performance in all reward metrics. Compared to GRPO-MR, which reaches 0.3724 in final search answer and 0.3346 in exact match, MT-GRPO demonstrates clear improvements, especially in exact match with a margin of +0.1664. In contrast, GRPO-OR performs poorly across all metrics, scoring 0 in turn-level rewards and only 0.04 in XML format. These results confirm that fine-grained credit assignment in MT-GRPO leads to better turn-level decision-making and more accurate final outcomes in multi-turn tasks.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "2",
              "url": "https://arxiv.org/html/2505.11821v1#S5.T2"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -3.657436656758806
        },
        {
          "evidence_id": "2505.11821v1:S5.p1",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S5.p1",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S5.p1",
          "section": "5 Experiments",
          "section_url": "https://arxiv.org/html/2505.11821v1#S5",
          "block_classes": [
            "ltx_para"
          ],
          "text": "In this section, we describe the experimental setup and present the main results to analyze the impact of credit assignment on training LLM agents for multi-turn tool-use tasks.",
          "equations": [],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -4.9692018202501975
        },
        {
          "evidence_id": "2505.11821v1:S6.p1",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S6.p1",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S6.p1",
          "section": "6 Conclusion and Future Work",
          "section_url": "https://arxiv.org/html/2505.11821v1#S6",
          "block_classes": [
            "ltx_para"
          ],
          "text": "In this work, we investigate the role of credit assignment in RL algorithms for enhancing the multi-turn reasoning capabilities of LLM agents. By constructing a two-turn tool-use environment, we demonstrate that trajectory-level advantage functions in existing RL algorithms like GRPO fail to effectively capture the individual contributions of actions within a trajectory. To address this limitation, we propose novel variants of the GRPO algorithm that enable turn-level credit assignment, tailored for multi-turn reasoning tasks. Through experiments on a Wikipedia search task, where the LLM agent learns to utilize a search engine to answer questions from the TriviaQA dataset, the results show that the proposed methods significantly improve both tool execution success rates and answer correctness compared to existing baselines. More specifically, our method achieves 100% success in tool execution and 50% accuracy in exact answer matching, significantly outperforming baselines, which fail to call tools and achieve only 20–30% exact match accuracy. These results highlight the critical importance of turn-level credit assignment in advancing the multi-turn reasoning capabilities of LLM agents.",
          "equations": [],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -4.886499599826897
        },
        {
          "evidence_id": "2505.11821v1:S6.p2",
          "arxiv": "2505.11821",
          "arxiv_version": "2505.11821v1",
          "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
          "authors": [
            "Zeng, Siliang",
            "Wei, Quan",
            "Brown, William",
            "Frunza, Oana",
            "Nevmyvaka, Yuriy",
            "Hong, Mingyi"
          ],
          "citation_date": "2025/05/17",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2505.11821v1#S6.p2",
          "paper_url": "https://arxiv.org/abs/2505.11821v1",
          "anchor": "S6.p2",
          "section": "6 Conclusion and Future Work",
          "section_url": "https://arxiv.org/html/2505.11821v1#S6",
          "block_classes": [
            "ltx_para"
          ],
          "text": "The current work primarily focuses on the two-turn tool-use environment, which serves as a simplified testbed to demonstrate the importance of credit assignment in multi-turn reasoning tasks. For future work, we aim to extend our methods to more complex multi-turn tool-use tasks involving longer horizons and interactions. Additionally, we plan to explore more flexible RL training pipelines and frameworks that do not rely on predefined turn-level verifiable rewards, enabling broader applicability in multi-turn reasoning tasks.",
          "equations": [],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -3.9468493009856314
        }
      ]
    },
    {
      "source": {
        "arxiv": "2602.22817",
        "arxiv_version": "2602.22817v1",
        "paper_url": "https://arxiv.org/abs/2602.22817v1",
        "html_url": "https://arxiv.org/html/2602.22817v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Hierarchy-of-Groups Policy Optimization for Long-Horizon Agentic Tasks",
        "authors": [
          "He, Shuo",
          "Feng, Lang",
          "Wei, Qi",
          "Cheng, Xin",
          "Feng, Lei",
          "An, Bo"
        ],
        "citation_date": "2026/02/26",
        "evidence_count": 64
      },
      "matching_block_count": 2,
      "representative_block": {
        "evidence_id": "2602.22817v1:S3.p3",
        "arxiv": "2602.22817",
        "arxiv_version": "2602.22817v1",
        "title": "Hierarchy-of-Groups Policy Optimization for Long-Horizon Agentic Tasks",
        "authors": [
          "He, Shuo",
          "Feng, Lang",
          "Wei, Qi",
          "Cheng, Xin",
          "Feng, Lei",
          "An, Bo"
        ],
        "citation_date": "2026/02/26",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "source_url": "https://arxiv.org/html/2602.22817v1#S3.p3",
        "paper_url": "https://arxiv.org/abs/2602.22817v1",
        "anchor": "S3.p3",
        "section": "3 Preliminaries",
        "section_url": "https://arxiv.org/html/2602.22817v1#S3",
        "block_classes": [
          "ltx_para"
        ],
        "text": "Group-based reinforcement learning. Unlike PPO ( Schulman et al., 2017 ) , which estimates advantages using an additional value function, group-based reinforcement learning (RL) algorithms such as GRPO ( Shao et al., 2024 ) compute advantages directly from the statistics of a sampled group of trajectories $G_{\\tau}$ . Specifically, GRPO was originally designed for single-turn tasks under a trajectory-wise policy optimization framework. To extend it to long-horizon tasks, we adapt it to the stepwise setting and calculate the trajectory-level advantage as: $\\displaystyle A^{T}(\\tau_{i})=\\left(R({\\tau_{i}})-1/|G_{\\tau}|\\sum\\nolimits_{j\\in G_{\\tau}}R({\\tau_{j}})\\right)/\\sigma_{G_{\\tau}},$ (1) where $\\sigma_{G_{\\tau}}$ denotes the standard deviation of rewards within the group $G_{\\tau}$ . This trajectory-level computation assigns the same advantage value to every step in trajectory $\\tau_{i}$ , thereby overlooking the finer credit assignment required within a trajectory. To address this limitation, one can instead adopt a step-level group relative advantage estimator ( Feng et al., 2025b ) . Here, steps with identical current states $\\tilde{\\bm{s}_{i}}$ across all group trajectories are clustered into step-level groups $G_{\\tilde{\\bm{s}_{i}}}$ , and their advantages are computed as: $\\displaystyle A^{S}(\\tilde{\\bm{s}_{i}})=\\left(R(\\tilde{\\bm{s}_{i}})-1/|G_{\\tilde{\\bm{s}_{i}}}|\\sum\\nolimits_{j\\in G_{{\\tilde{\\bm{s}_{i}}}}}R(\\tilde{\\bm{s}_{j}})\\right)/\\sigma_{G_{{\\tilde{\\bm{s}_{i}}}}}.$ (2) Compared to Eq. ( 1 ), the step-level estimator in Eq. ( 2 ) provides more fine-grained and effective credit assignment across steps within the same trajectory.",
        "equations": [
          {
            "anchor": "S3.p3.m1",
            "latex": "G_{\\tau}",
            "display": "inline"
          },
          {
            "anchor": "S3.E1.m1",
            "latex": "\\displaystyle A^{T}(\\tau_{i})=\\left(R({\\tau_{i}})-1/|G_{\\tau}|\\sum\\nolimits_{j\\in G_{\\tau}}R({\\tau_{j}})\\right)/\\sigma_{G_{\\tau}},",
            "display": "inline"
          },
          {
            "anchor": "S3.p3.m2",
            "latex": "\\sigma_{G_{\\tau}}",
            "display": "inline"
          },
          {
            "anchor": "S3.p3.m3",
            "latex": "G_{\\tau}",
            "display": "inline"
          },
          {
            "anchor": "S3.p3.m4",
            "latex": "\\tau_{i}",
            "display": "inline"
          },
          {
            "anchor": "S3.p3.m5",
            "latex": "\\tilde{\\bm{s}_{i}}",
            "display": "inline"
          },
          {
            "anchor": "S3.p3.m6",
            "latex": "G_{\\tilde{\\bm{s}_{i}}}",
            "display": "inline"
          },
          {
            "anchor": "S3.E2.m1",
            "latex": "\\displaystyle A^{S}(\\tilde{\\bm{s}_{i}})=\\left(R(\\tilde{\\bm{s}_{i}})-1/|G_{\\tilde{\\bm{s}_{i}}}|\\sum\\nolimits_{j\\in G_{{\\tilde{\\bm{s}_{i}}}}}R(\\tilde{\\bm{s}_{j}})\\right)/\\sigma_{G_{{\\tilde{\\bm{s}_{i}}}}}.",
            "display": "inline"
          }
        ],
        "tables": [],
        "links": [
          {
            "text": "Schulman et al., 2017",
            "url": "https://arxiv.org/html/2602.22817v1#bib.bib2"
          },
          {
            "text": "Shao et al., 2024",
            "url": "https://arxiv.org/html/2602.22817v1#bib.bib35"
          },
          {
            "text": "Feng et al., 2025b",
            "url": "https://arxiv.org/html/2602.22817v1#bib.bib54"
          },
          {
            "text": "1",
            "url": "https://arxiv.org/html/2602.22817v1#S3.E1"
          },
          {
            "text": "2",
            "url": "https://arxiv.org/html/2602.22817v1#S3.E2"
          }
        ],
        "untranscribed_graphics": 0,
        "bm25_score": -3.5181520886074926
      },
      "matching_blocks": [
        {
          "evidence_id": "2602.22817v1:S1.p3",
          "arxiv": "2602.22817",
          "arxiv_version": "2602.22817v1",
          "title": "Hierarchy-of-Groups Policy Optimization for Long-Horizon Agentic Tasks",
          "authors": [
            "He, Shuo",
            "Feng, Lang",
            "Wei, Qi",
            "Cheng, Xin",
            "Feng, Lei",
            "An, Bo"
          ],
          "citation_date": "2026/02/26",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2602.22817v1#S1.p3",
          "paper_url": "https://arxiv.org/abs/2602.22817v1",
          "anchor": "S1.p3",
          "section": "1 Introduction",
          "section_url": "https://arxiv.org/html/2602.22817v1#S1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "To address this issue, recent research has shifted toward the stepwise policy optimization framework ( Feng et al., 2025b ; Luo et al., 2025c ; Chen et al., 2025b ; Team, 2025 ; Yu et al., 2025b ; Wang et al., 2025c ) , which treats each step within a rollout trajectory independently while leveraging a memory module to retain historical context. This design allows for flexible context management and highly scalable RL training. A comparison of the two frameworks is illustrated in Figure 1 . (a). Building on the stepwise framework, group-based RL methods such as GRPO ( Shao et al., 2024 ) can be adapted into stepwise group-based variants for long-horizon agentic tasks. Furthermore, to enable finer-grained credit assignment, GiGPO ( Feng et al., 2025b ) extends GRPO by estimating additional step-level advantages within groups where all steps share the same current state.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "Feng et al., 2025b",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib54"
            },
            {
              "text": "Luo et al., 2025c",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib55"
            },
            {
              "text": "Chen et al., 2025b",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib69"
            },
            {
              "text": "Team, 2025",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib70"
            },
            {
              "text": "Yu et al., 2025b",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib88"
            },
            {
              "text": "Wang et al., 2025c",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib89"
            },
            {
              "text": "1",
              "url": "https://arxiv.org/html/2602.22817v1#S1.F1"
            },
            {
              "text": "Shao et al., 2024",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib35"
            },
            {
              "text": "Feng et al., 2025b",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib54"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -3.2414797683648384
        },
        {
          "evidence_id": "2602.22817v1:S3.p3",
          "arxiv": "2602.22817",
          "arxiv_version": "2602.22817v1",
          "title": "Hierarchy-of-Groups Policy Optimization for Long-Horizon Agentic Tasks",
          "authors": [
            "He, Shuo",
            "Feng, Lang",
            "Wei, Qi",
            "Cheng, Xin",
            "Feng, Lei",
            "An, Bo"
          ],
          "citation_date": "2026/02/26",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2602.22817v1#S3.p3",
          "paper_url": "https://arxiv.org/abs/2602.22817v1",
          "anchor": "S3.p3",
          "section": "3 Preliminaries",
          "section_url": "https://arxiv.org/html/2602.22817v1#S3",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Group-based reinforcement learning. Unlike PPO ( Schulman et al., 2017 ) , which estimates advantages using an additional value function, group-based reinforcement learning (RL) algorithms such as GRPO ( Shao et al., 2024 ) compute advantages directly from the statistics of a sampled group of trajectories $G_{\\tau}$ . Specifically, GRPO was originally designed for single-turn tasks under a trajectory-wise policy optimization framework. To extend it to long-horizon tasks, we adapt it to the stepwise setting and calculate the trajectory-level advantage as: $\\displaystyle A^{T}(\\tau_{i})=\\left(R({\\tau_{i}})-1/|G_{\\tau}|\\sum\\nolimits_{j\\in G_{\\tau}}R({\\tau_{j}})\\right)/\\sigma_{G_{\\tau}},$ (1) where $\\sigma_{G_{\\tau}}$ denotes the standard deviation of rewards within the group $G_{\\tau}$ . This trajectory-level computation assigns the same advantage value to every step in trajectory $\\tau_{i}$ , thereby overlooking the finer credit assignment required within a trajectory. To address this limitation, one can instead adopt a step-level group relative advantage estimator ( Feng et al., 2025b ) . Here, steps with identical current states $\\tilde{\\bm{s}_{i}}$ across all group trajectories are clustered into step-level groups $G_{\\tilde{\\bm{s}_{i}}}$ , and their advantages are computed as: $\\displaystyle A^{S}(\\tilde{\\bm{s}_{i}})=\\left(R(\\tilde{\\bm{s}_{i}})-1/|G_{\\tilde{\\bm{s}_{i}}}|\\sum\\nolimits_{j\\in G_{{\\tilde{\\bm{s}_{i}}}}}R(\\tilde{\\bm{s}_{j}})\\right)/\\sigma_{G_{{\\tilde{\\bm{s}_{i}}}}}.$ (2) Compared to Eq. ( 1 ), the step-level estimator in Eq. ( 2 ) provides more fine-grained and effective credit assignment across steps within the same trajectory.",
          "equations": [
            {
              "anchor": "S3.p3.m1",
              "latex": "G_{\\tau}",
              "display": "inline"
            },
            {
              "anchor": "S3.E1.m1",
              "latex": "\\displaystyle A^{T}(\\tau_{i})=\\left(R({\\tau_{i}})-1/|G_{\\tau}|\\sum\\nolimits_{j\\in G_{\\tau}}R({\\tau_{j}})\\right)/\\sigma_{G_{\\tau}},",
              "display": "inline"
            },
            {
              "anchor": "S3.p3.m2",
              "latex": "\\sigma_{G_{\\tau}}",
              "display": "inline"
            },
            {
              "anchor": "S3.p3.m3",
              "latex": "G_{\\tau}",
              "display": "inline"
            },
            {
              "anchor": "S3.p3.m4",
              "latex": "\\tau_{i}",
              "display": "inline"
            },
            {
              "anchor": "S3.p3.m5",
              "latex": "\\tilde{\\bm{s}_{i}}",
              "display": "inline"
            },
            {
              "anchor": "S3.p3.m6",
              "latex": "G_{\\tilde{\\bm{s}_{i}}}",
              "display": "inline"
            },
            {
              "anchor": "S3.E2.m1",
              "latex": "\\displaystyle A^{S}(\\tilde{\\bm{s}_{i}})=\\left(R(\\tilde{\\bm{s}_{i}})-1/|G_{\\tilde{\\bm{s}_{i}}}|\\sum\\nolimits_{j\\in G_{{\\tilde{\\bm{s}_{i}}}}}R(\\tilde{\\bm{s}_{j}})\\right)/\\sigma_{G_{{\\tilde{\\bm{s}_{i}}}}}.",
              "display": "inline"
            }
          ],
          "tables": [],
          "links": [
            {
              "text": "Schulman et al., 2017",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib2"
            },
            {
              "text": "Shao et al., 2024",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib35"
            },
            {
              "text": "Feng et al., 2025b",
              "url": "https://arxiv.org/html/2602.22817v1#bib.bib54"
            },
            {
              "text": "1",
              "url": "https://arxiv.org/html/2602.22817v1#S3.E1"
            },
            {
              "text": "2",
              "url": "https://arxiv.org/html/2602.22817v1#S3.E2"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -3.5181520886074926
        }
      ]
    },
    {
      "source": {
        "arxiv": "2605.08013",
        "arxiv_version": "2605.08013v1",
        "paper_url": "https://arxiv.org/abs/2605.08013v1",
        "html_url": "https://arxiv.org/html/2605.08013v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
        "authors": [
          "Su, Haoyang",
          "Wen, Ying"
        ],
        "citation_date": "2026/05/08",
        "evidence_count": 74
      },
      "matching_block_count": 6,
      "representative_block": {
        "evidence_id": "2605.08013v1:S2.SS2.p1",
        "arxiv": "2605.08013",
        "arxiv_version": "2605.08013v1",
        "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
        "authors": [
          "Su, Haoyang",
          "Wen, Ying"
        ],
        "citation_date": "2026/05/08",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "source_url": "https://arxiv.org/html/2605.08013v1#S2.SS2.p1",
        "paper_url": "https://arxiv.org/abs/2605.08013v1",
        "anchor": "S2.SS2.p1",
        "section": "2 Related Work / 2.2 Agentic Reinforcement Learning",
        "section_url": "https://arxiv.org/html/2605.08013v1#S2.SS2",
        "block_classes": [
          "ltx_para"
        ],
        "text": "Standard RL fine-tuning for large language models distributes credit at token granularity within a single generation [ 36 , 43 , 45 , 69 , 3 ] , while credit assignment across multi-turn environment interaction remains less settled [ 70 , 46 , 66 , 29 , 56 , 79 , 22 ] .",
        "equations": [],
        "tables": [],
        "links": [
          {
            "text": "36",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib8"
          },
          {
            "text": "43",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib7"
          },
          {
            "text": "45",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib5"
          },
          {
            "text": "69",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib4"
          },
          {
            "text": "3",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib6"
          },
          {
            "text": "70",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib36"
          },
          {
            "text": "46",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib11"
          },
          {
            "text": "66",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib12"
          },
          {
            "text": "29",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib13"
          },
          {
            "text": "56",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib72"
          },
          {
            "text": "79",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib65"
          },
          {
            "text": "22",
            "url": "https://arxiv.org/html/2605.08013v1#bib.bib66"
          }
        ],
        "untranscribed_graphics": 0,
        "bm25_score": -4.733414048211432
      },
      "matching_blocks": [
        {
          "evidence_id": "2605.08013v1:A4.p4",
          "arxiv": "2605.08013",
          "arxiv_version": "2605.08013v1",
          "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
          "authors": [
            "Su, Haoyang",
            "Wen, Ying"
          ],
          "citation_date": "2026/05/08",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2605.08013v1#A4.p4",
          "paper_url": "https://arxiv.org/abs/2605.08013v1",
          "anchor": "A4.p4",
          "section": "Appendix D Training Dynamics across Agentic RL Methods",
          "section_url": "https://arxiv.org/html/2605.08013v1#A4",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Token-level surrogates plateau under coarse credit assignment. GSPO maintains a well-conditioned surrogate KL envelope and stable policy entropy throughout training, making it the most stable baseline at the optimization level. Its training success and reward trajectories saturate after roughly $120$ updates and remain on a plateau for the rest of the budget. The sequence-level importance ratio is normalized for single-turn language modelling, but does not separate subgoal completion within a trajectory from terminal reward. Once the easy mass of the dataset has been fitted, the policy receives little additional shaping signal from interior recovery steps.",
          "equations": [
            {
              "anchor": "A4.p4.m1",
              "latex": "120",
              "display": "inline"
            }
          ],
          "tables": [],
          "links": [],
          "untranscribed_graphics": 0,
          "bm25_score": -3.6981126675772966
        },
        {
          "evidence_id": "2605.08013v1:A4.p5",
          "arxiv": "2605.08013",
          "arxiv_version": "2605.08013v1",
          "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
          "authors": [
            "Su, Haoyang",
            "Wen, Ying"
          ],
          "citation_date": "2026/05/08",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2605.08013v1#A4.p5",
          "paper_url": "https://arxiv.org/abs/2605.08013v1",
          "anchor": "A4.p5",
          "section": "Appendix D Training Dynamics across Agentic RL Methods",
          "section_url": "https://arxiv.org/html/2605.08013v1#A4",
          "block_classes": [
            "ltx_para"
          ],
          "text": "$\\mathrm{A}^{3}$ retains a stable surrogate trajectory while continuing to gain. Across the same horizon, $\\mathrm{A}^{3}$ keeps the surrogate KL inside an envelope comparable to GSPO and avoids the late training shock observed for HGPO, GiGPO, and RetroAgent. Its policy entropy decays more gradually than the memory based baseline and remains above the saturation level reached by GSPO, while the success and reward panels continue to grow throughout the second half of training. This pattern is consistent with the multi-granularity credit assignment of Section 3 , where the episode backbone supplies terminal signal, the turn-level action sub-chain residual adds local shaping, and the tree advantage redistributes credit within sibling rollouts.",
          "equations": [
            {
              "anchor": "A4.p5.m1",
              "latex": "\\mathrm{A}^{3}",
              "display": "inline"
            },
            {
              "anchor": "A4.p5.m2",
              "latex": "\\mathrm{A}^{3}",
              "display": "inline"
            }
          ],
          "tables": [],
          "links": [
            {
              "text": "3",
              "url": "https://arxiv.org/html/2605.08013v1#S3"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -3.4910434588031762
        },
        {
          "evidence_id": "2605.08013v1:S1.p3",
          "arxiv": "2605.08013",
          "arxiv_version": "2605.08013v1",
          "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
          "authors": [
            "Su, Haoyang",
            "Wen, Ying"
          ],
          "citation_date": "2026/05/08",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2605.08013v1#S1.p3",
          "paper_url": "https://arxiv.org/abs/2605.08013v1",
          "anchor": "S1.p3",
          "section": "1 Introduction",
          "section_url": "https://arxiv.org/html/2605.08013v1#S1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Trajectory-level supervised fine-tuning constructs annotated traces and trains the model to imitate them [ 42 , 6 , 64 , 47 ] , yet the resulting policy is bounded by the narrow coverage of its training data, and recent analysis confirms that such imitation increases memorization of patterns tied to the interface rather than real task understanding [ 14 ] . Reinforcement learning (RL) addresses this limitation by letting the agent explore and optimize toward task-level rewards [ 44 , 7 , 39 ] . Tool-oriented RL methods decompose rewards into format validity, parameter accuracy, and tool selection correctness [ 28 ] , or learn context control and execution structure to limit context growth during long interaction [ 15 ] . A complementary critic-free GRPO family, including GiGPO and HGPO [ 10 , 16 ] , uses observation-anchored or state-grouped normalization to refine credit assignment. These methods assume repeated states for within-state normalization, an assumption weakened by the large CLI and LLM state space, where nearly every observation can be unique. Existing paradigms therefore leave both partial workspace observation and sparse action credit unresolved for CLI agent learning.",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "42",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib37"
            },
            {
              "text": "6",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib38"
            },
            {
              "text": "64",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib39"
            },
            {
              "text": "47",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib40"
            },
            {
              "text": "14",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib41"
            },
            {
              "text": "44",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib3"
            },
            {
              "text": "7",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib10"
            },
            {
              "text": "39",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib61"
            },
            {
              "text": "28",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib42"
            },
            {
              "text": "15",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib43"
            },
            {
              "text": "10",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib1"
            },
            {
              "text": "16",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib2"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -2.910474226352946
        },
        {
          "evidence_id": "2605.08013v1:S2.SS2.p1",
          "arxiv": "2605.08013",
          "arxiv_version": "2605.08013v1",
          "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
          "authors": [
            "Su, Haoyang",
            "Wen, Ying"
          ],
          "citation_date": "2026/05/08",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2605.08013v1#S2.SS2.p1",
          "paper_url": "https://arxiv.org/abs/2605.08013v1",
          "anchor": "S2.SS2.p1",
          "section": "2 Related Work / 2.2 Agentic Reinforcement Learning",
          "section_url": "https://arxiv.org/html/2605.08013v1#S2.SS2",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Standard RL fine-tuning for large language models distributes credit at token granularity within a single generation [ 36 , 43 , 45 , 69 , 3 ] , while credit assignment across multi-turn environment interaction remains less settled [ 70 , 46 , 66 , 29 , 56 , 79 , 22 ] .",
          "equations": [],
          "tables": [],
          "links": [
            {
              "text": "36",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib8"
            },
            {
              "text": "43",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib7"
            },
            {
              "text": "45",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib5"
            },
            {
              "text": "69",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib4"
            },
            {
              "text": "3",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib6"
            },
            {
              "text": "70",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib36"
            },
            {
              "text": "46",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib11"
            },
            {
              "text": "66",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib12"
            },
            {
              "text": "29",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib13"
            },
            {
              "text": "56",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib72"
            },
            {
              "text": "79",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib65"
            },
            {
              "text": "22",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib66"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -4.733414048211432
        },
        {
          "evidence_id": "2605.08013v1:S2.SS2.p3",
          "arxiv": "2605.08013",
          "arxiv_version": "2605.08013v1",
          "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
          "authors": [
            "Su, Haoyang",
            "Wen, Ying"
          ],
          "citation_date": "2026/05/08",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2605.08013v1#S2.SS2.p3",
          "paper_url": "https://arxiv.org/abs/2605.08013v1",
          "anchor": "S2.SS2.p3",
          "section": "2 Related Work / 2.2 Agentic Reinforcement Learning",
          "section_url": "https://arxiv.org/html/2605.08013v1#S2.SS2",
          "block_classes": [
            "ltx_para"
          ],
          "text": "Recent agentic RL methods refine credit assignment in several directions. GSPO keeps the group relative objective at sequence scope [ 74 ] , while GiGPO and HGPO anchor advantages on repeated states or hierarchical state groups [ 10 , 16 ] . Turn-level RL and GTPO add rewards at each turn through MDP reformulation or execution signals [ 57 , 8 ] . IGPO, ZeroSearch, and StepSearch use information gain to guide search trajectories [ 51 , 48 , 77 ] . iStar and SPA-RL learn process or progress estimators, while rStar resamples trajectories and RetroAgent adds retrospective feedback from an LLM judge [ 30 , 52 , 40 , 73 ] . SkillRL, SKILL0, SLEA-RL, MemRL, and EvolveR introduce skill stores, retrieval, memory, or principle repositories into the policy loop [ 61 , 32 , 53 , 72 , 58 ] . Across these lines, credit assignment often depends on sequence normalization, repeated state anchors, learned estimators, retrieved context, or external judges. $\\mathrm{A}^{3}$ instead uses shell syntax directly in the advantage, without auxiliary models or state anchoring, at computational overhead comparable to conventional agentic RL.",
          "equations": [
            {
              "anchor": "S2.SS2.p3.m1",
              "latex": "\\mathrm{A}^{3}",
              "display": "inline"
            }
          ],
          "tables": [],
          "links": [
            {
              "text": "74",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib56"
            },
            {
              "text": "10",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib1"
            },
            {
              "text": "16",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib2"
            },
            {
              "text": "57",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib30"
            },
            {
              "text": "8",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib31"
            },
            {
              "text": "51",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib47"
            },
            {
              "text": "48",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib53"
            },
            {
              "text": "77",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib54"
            },
            {
              "text": "30",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib29"
            },
            {
              "text": "52",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib32"
            },
            {
              "text": "40",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib34"
            },
            {
              "text": "73",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib35"
            },
            {
              "text": "61",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib48"
            },
            {
              "text": "32",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib49"
            },
            {
              "text": "53",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib50"
            },
            {
              "text": "72",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib51"
            },
            {
              "text": "58",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib52"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -4.298962605309136
        },
        {
          "evidence_id": "2605.08013v1:S3.SS1.p1",
          "arxiv": "2605.08013",
          "arxiv_version": "2605.08013v1",
          "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
          "authors": [
            "Su, Haoyang",
            "Wen, Ying"
          ],
          "citation_date": "2026/05/08",
          "license": "CC BY 4.0",
          "license_url": "https://creativecommons.org/licenses/by/4.0/",
          "source_url": "https://arxiv.org/html/2605.08013v1#S3.SS1.p1",
          "paper_url": "https://arxiv.org/abs/2605.08013v1",
          "anchor": "S3.SS1.p1",
          "section": "3 Method / 3.1 AST measure for CLI agent actions",
          "section_url": "https://arxiv.org/html/2605.08013v1#S3.SS1",
          "block_classes": [
            "ltx_para"
          ],
          "text": "CLI agent actions are executable shell programs rather than free-form text. Their parse structure provides a compact basis for comparing action intent during credit assignment and is amenable to accelerated batch computation. We quantify action intent by comparing AST signatures. Let $\\mathrm{AST}(a)$ denote the Tree-sitter grammar for bash [ 5 ] applied to an action string $a$ . The map $\\mathrm{Lin}$ performs a fixed preorder traversal of $\\mathrm{AST}(a)$ and appends tokens at each visit according to deterministic rules. Control structure nodes contribute tokens in $\\mathcal{A}_{K}$ with kinds $\\kappa\\in\\mathcal{T}_{\\mathrm{ctrl}}$ . Each command node contributes one token in $\\mathcal{A}_{V}$ for the canonical verb and a finite sequence of tokens in $\\mathcal{A}_{W}$ for literals after normalization. The full signature is the concatenation of these contributions in visit order, an element of $\\mathcal{A}^{\\ast}$ with $\\mathcal{A}=\\mathcal{A}_{K}\\cup\\mathcal{A}_{V}\\cup\\mathcal{A}_{W}$ . We summarize this signature map in ( 1 ). $\\sigma(a)=\\mathrm{Lin}(\\mathrm{AST}(a))\\in\\mathcal{A}^{\\ast}.$ (1) Pairwise distance between actions is normalized Levenshtein distance [ 25 ] on signatures, as in ( 2 ). $d(a_{i},a_{j})=\\frac{\\mathrm{Lev}\\bigl(\\sigma(a_{i}),\\sigma(a_{j})\\bigr)}{\\max\\bigl(|\\sigma(a_{i})|,|\\sigma(a_{j})|\\bigr)}\\in[0,1].$ (2) This distance compares shell actions by structural form rather than surface paths or literal values, and the complete action pair comparison is illustrated in Fig. 2 .",
          "equations": [
            {
              "anchor": "S3.SS1.p1.m1",
              "latex": "\\mathrm{AST}(a)",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m2",
              "latex": "a",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m3",
              "latex": "\\mathrm{Lin}",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m4",
              "latex": "\\mathrm{AST}(a)",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m5",
              "latex": "\\mathcal{A}_{K}",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m6",
              "latex": "\\kappa\\in\\mathcal{T}_{\\mathrm{ctrl}}",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m7",
              "latex": "\\mathcal{A}_{V}",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m8",
              "latex": "\\mathcal{A}_{W}",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m9",
              "latex": "\\mathcal{A}^{\\ast}",
              "display": "inline"
            },
            {
              "anchor": "S3.SS1.p1.m10",
              "latex": "\\mathcal{A}=\\mathcal{A}_{K}\\cup\\mathcal{A}_{V}\\cup\\mathcal{A}_{W}",
              "display": "inline"
            },
            {
              "anchor": "S3.E1.m1",
              "latex": "\\sigma(a)=\\mathrm{Lin}(\\mathrm{AST}(a))\\in\\mathcal{A}^{\\ast}.",
              "display": "block"
            },
            {
              "anchor": "S3.E2.m1",
              "latex": "d(a_{i},a_{j})=\\frac{\\mathrm{Lev}\\bigl(\\sigma(a_{i}),\\sigma(a_{j})\\bigr)}{\\max\\bigl(|\\sigma(a_{i})|,|\\sigma(a_{j})|\\bigr)}\\in[0,1].",
              "display": "block"
            }
          ],
          "tables": [],
          "links": [
            {
              "text": "5",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib76"
            },
            {
              "text": "1",
              "url": "https://arxiv.org/html/2605.08013v1#S3.E1"
            },
            {
              "text": "25",
              "url": "https://arxiv.org/html/2605.08013v1#bib.bib75"
            },
            {
              "text": "2",
              "url": "https://arxiv.org/html/2605.08013v1#S3.E2"
            },
            {
              "text": "2",
              "url": "https://arxiv.org/html/2605.08013v1#S3.F2"
            }
          ],
          "untranscribed_graphics": 0,
          "bm25_score": -2.38794608383455
        }
      ]
    }
  ]
}
