{
  "title": "Agentic reinforcement learning research",
  "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/index.html",
  "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/index.json",
  "corpus": {
    "scope": {
      "declared_source_count": 13,
      "indexed_source_count": 8,
      "metadata_only_source_count": 5,
      "comprehensive": false,
      "query_language": "en",
      "topics": [
        "agent_reinforcement_learning",
        "credit_assignment",
        "cli_agents",
        "software_benchmarks"
      ]
    },
    "source_catalog": [
      {
        "arxiv": "2405.15793",
        "arxiv_version": "2405.15793v1",
        "paper_url": "https://arxiv.org/abs/2405.15793v1",
        "html_url": "https://arxiv.org/html/2405.15793v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "SWE-agent: Agent-Computer Interfaces Enable Automated Software Engineering",
        "authors": [
          "Yang, John",
          "Jimenez, Carlos E.",
          "Wettig, Alexander",
          "Lieret, Kilian",
          "Yao, Shunyu",
          "Narasimhan, Karthik",
          "Press, Ofir"
        ],
        "citation_date": "2024/05/06",
        "evidence_count": 91
      },
      {
        "arxiv": "2310.06770",
        "arxiv_version": "2310.06770v1",
        "paper_url": "https://arxiv.org/abs/2310.06770v1",
        "html_url": "https://arxiv.org/html/2310.06770v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "SWE-bench: Can Language Models Resolve Real-World GitHub Issues?",
        "authors": [
          "Jimenez, Carlos E.",
          "Yang, John",
          "Wettig, Alexander",
          "Yao, Shunyu",
          "Pei, Kexin",
          "Press, Ofir",
          "Narasimhan, Karthik"
        ],
        "citation_date": "2023/10/10",
        "evidence_count": 135
      },
      {
        "arxiv": "2601.11868",
        "arxiv_version": "2601.11868v1",
        "paper_url": "https://arxiv.org/abs/2601.11868v1",
        "html_url": "https://arxiv.org/html/2601.11868v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces",
        "authors": [
          "Merrill, Mike A.",
          "Shaw, Alexander G.",
          "Carlini, Nicholas",
          "Li, Boxuan",
          "Raj, Harsh",
          "Bercovich, Ivan",
          "Shi, Lin",
          "Shin, Jeong Yeon",
          "Walshe, Thomas",
          "Buchanan, E. Kelly",
          "Shen, Junhong",
          "Ye, Guanghao",
          "Lin, Haowei",
          "Poulos, Jason",
          "Wang, Maoyu",
          "Nezhurina, Marianna",
          "Jitsev, Jenia",
          "Lu, Di",
          "Mastromichalakis, Orfeas Menis",
          "Xu, Zhiwei",
          "Chen, Zizhao",
          "Liu, Yue",
          "Zhang, Robert",
          "Chen, Leon Liangyu",
          "Kashyap, Anurag",
          "Uslu, Jan-Lucas",
          "Li, Jeffrey",
          "Wu, Jianbo",
          "Yan, Minghao",
          "Bian, Song",
          "Sharma, Vedang",
          "Sun, Ke",
          "Dillmann, Steven",
          "Anand, Akshay",
          "Lanpouthakoun, Andrew",
          "Koopah, Bardia",
          "Hu, Changran",
          "Guha, Etash",
          "Dreiman, Gabriel H. S.",
          "Zhu, Jiacheng",
          "Krauth, Karl",
          "Zhong, Li",
          "Muennighoff, Niklas",
          "Amanfu, Robert",
          "Tan, Shangyin",
          "Pimpalgaonkar, Shreyas",
          "Aggarwal, Tushar",
          "Lin, Xiangning",
          "Lan, Xin",
          "Zhao, Xuandong",
          "Liang, Yiqing",
          "Wang, Yuanli",
          "Wang, Zilong",
          "Zhou, Changzhi",
          "Heineman, David",
          "Liu, Hange",
          "Trivedi, Harsh",
          "Yang, John",
          "Lin, Junhong",
          "Shetty, Manish",
          "Yang, Michael",
          "Omi, Nabil",
          "Raoof, Negin",
          "Li, Shanda",
          "Zhuo, Terry Yue",
          "Lin, Wuwei",
          "Dai, Yiwei",
          "Wang, Yuxin",
          "Chai, Wenhao",
          "Zhou, Shang",
          "Wahdany, Dariush",
          "She, Ziyu",
          "Hu, Jiaming",
          "Dong, Zhikang",
          "Zhu, Yuxuan",
          "Cui, Sasha",
          "Saiyed, Ahson",
          "Kolbeinsson, Arinbjörn",
          "Hu, Jesse",
          "Rytting, Christopher Michael",
          "Marten, Ryan",
          "Wang, Yixin",
          "Dimakis, Alex",
          "Konwinski, Andy",
          "Schmidt, Ludwig"
        ],
        "citation_date": "2026/01/17",
        "evidence_count": 160
      },
      {
        "arxiv": "2607.22724",
        "arxiv_version": "2607.22724v1",
        "paper_url": "https://arxiv.org/abs/2607.22724v1",
        "html_url": "https://arxiv.org/html/2607.22724v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "Progress-conditioned Group Policy Optimization for Long-Horizon Agentic Tasks",
        "authors": [
          "Yang, Kaibing",
          "Cai, Guangfeng",
          "Yang, Shengtian",
          "He, Shuo",
          "Li, Yu",
          "Liu, Mengyi",
          "Chen, Pengwei",
          "Xu, Jun",
          "Feng, Lei"
        ],
        "citation_date": "2026/07/22",
        "evidence_count": 0
      },
      {
        "arxiv": "2505.11821",
        "arxiv_version": "2505.11821v1",
        "paper_url": "https://arxiv.org/abs/2505.11821v1",
        "html_url": "https://arxiv.org/html/2505.11821v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Reinforcing Multi-Turn Reasoning in LLM Agents via Turn-Level Credit Assignment",
        "authors": [
          "Zeng, Siliang",
          "Wei, Quan",
          "Brown, William",
          "Frunza, Oana",
          "Nevmyvaka, Yuriy",
          "Hong, Mingyi"
        ],
        "citation_date": "2025/05/17",
        "evidence_count": 51
      },
      {
        "arxiv": "2402.03300",
        "arxiv_version": "2402.03300v1",
        "paper_url": "https://arxiv.org/abs/2402.03300v1",
        "html_url": "https://arxiv.org/html/2402.03300v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models",
        "authors": [
          "Shao, Zhihong",
          "Wang, Peiyi",
          "Zhu, Qihao",
          "Xu, Runxin",
          "Song, Junxiao",
          "Zhang, Mingchuan",
          "Li, Y. K.",
          "Wu, Y.",
          "Guo, Daya"
        ],
        "citation_date": "2024/02/05",
        "evidence_count": 0
      },
      {
        "arxiv": "2602.22817",
        "arxiv_version": "2602.22817v1",
        "paper_url": "https://arxiv.org/abs/2602.22817v1",
        "html_url": "https://arxiv.org/html/2602.22817v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Hierarchy-of-Groups Policy Optimization for Long-Horizon Agentic Tasks",
        "authors": [
          "He, Shuo",
          "Feng, Lang",
          "Wei, Qi",
          "Cheng, Xin",
          "Feng, Lei",
          "An, Bo"
        ],
        "citation_date": "2026/02/26",
        "evidence_count": 64
      },
      {
        "arxiv": "2407.16741",
        "arxiv_version": "2407.16741v1",
        "paper_url": "https://arxiv.org/abs/2407.16741v1",
        "html_url": "https://arxiv.org/html/2407.16741v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "OpenDevin: An Open Platform for AI Software Developers as Generalist Agents",
        "authors": [
          "Wang, Xingyao",
          "Li, Boxuan",
          "Song, Yufan",
          "Xu, Frank F.",
          "Tang, Xiangru",
          "Zhuge, Mingchen",
          "Pan, Jiayi",
          "Song, Yueqi",
          "Li, Bowen",
          "Singh, Jaskirat",
          "Tran, Hoang H.",
          "Li, Fuqiang",
          "Ma, Ren",
          "Zheng, Mingzhang",
          "Qian, Bill",
          "Shao, Yanjun",
          "Muennighoff, Niklas",
          "Zhang, Yizhe",
          "Hui, Binyuan",
          "Lin, Junyang",
          "Brennan, Robert",
          "Peng, Hao",
          "Ji, Heng",
          "Neubig, Graham"
        ],
        "citation_date": "2024/07/23",
        "evidence_count": 107
      },
      {
        "arxiv": "2502.18449",
        "arxiv_version": "2502.18449v1",
        "paper_url": "https://arxiv.org/abs/2502.18449v1",
        "html_url": "https://arxiv.org/html/2502.18449v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "SWE-RL : Advancing LLM Reasoning via Reinforcement Learning on Open Software Evolution",
        "authors": [
          "Wei, Yuxiang",
          "Duchenne, Olivier",
          "Copet, Jade",
          "Carbonneaux, Quentin",
          "Zhang, Lingming",
          "Fried, Daniel",
          "Synnaeve, Gabriel",
          "Singh, Rishabh",
          "Wang, Sida I."
        ],
        "citation_date": "2025/02/25",
        "evidence_count": 0
      },
      {
        "arxiv": "2605.08013",
        "arxiv_version": "2605.08013v1",
        "paper_url": "https://arxiv.org/abs/2605.08013v1",
        "html_url": "https://arxiv.org/html/2605.08013v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "Learning CLI Agents with Structured Action Credit under Selective Observation",
        "authors": [
          "Su, Haoyang",
          "Wen, Ying"
        ],
        "citation_date": "2026/05/08",
        "evidence_count": 74
      },
      {
        "arxiv": "2503.09516",
        "arxiv_version": "2503.09516v1",
        "paper_url": "https://arxiv.org/abs/2503.09516v1",
        "html_url": "https://arxiv.org/html/2503.09516v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "Search-R1: Training LLMs to Reason and Leverage Search Engines with Reinforcement Learning",
        "authors": [
          "Jin, Bowen",
          "Zeng, Hansi",
          "Yue, Zhenrui",
          "Wang, Dong",
          "Zamani, Hamed",
          "Han, Jiawei"
        ],
        "citation_date": "2025/03/12",
        "evidence_count": 0
      },
      {
        "arxiv": "2504.20073",
        "arxiv_version": "2504.20073v1",
        "paper_url": "https://arxiv.org/abs/2504.20073v1",
        "html_url": "https://arxiv.org/html/2504.20073v1",
        "license": "CC BY 4.0",
        "license_url": "https://creativecommons.org/licenses/by/4.0/",
        "index_mode": "original_html_body",
        "exclusion_reason_code": null,
        "title": "RAGEN: Understanding Self-Evolution in LLM Agents via Multi-Turn Reinforcement Learning",
        "authors": [
          "Wang, Zihan",
          "Wang, Kangrui",
          "Wang, Qineng",
          "Zhang, Pingyue",
          "Li, Linjie",
          "Yang, Zhengyuan",
          "Yu, Kefan",
          "Nguyen, Minh Nhat",
          "Liu, Licheng",
          "Gottlieb, Eli",
          "Lam, Monica",
          "Lu, Yiping",
          "Cho, Kyunghyun",
          "Wu, Jiajun",
          "Fei-Fei, Li",
          "Wang, Lijuan",
          "Choi, Yejin",
          "Li, Manling"
        ],
        "citation_date": "2025/04/24",
        "evidence_count": 115
      },
      {
        "arxiv": "2505.10978",
        "arxiv_version": "2505.10978v1",
        "paper_url": "https://arxiv.org/abs/2505.10978v1",
        "html_url": "https://arxiv.org/html/2505.10978v1",
        "license": "arXiv.org perpetual non-exclusive license",
        "license_url": "https://arxiv.org/licenses/nonexclusive-distrib/1.0/license.html",
        "index_mode": "metadata_only",
        "exclusion_reason_code": "no_third_party_fulltext_redistribution_grant",
        "title": "Group-in-Group Policy Optimization for LLM Agent Training",
        "authors": [
          "Feng, Lang",
          "Xue, Zhenghai",
          "Liu, Tingcong",
          "An, Bo"
        ],
        "citation_date": "2025/05/16",
        "evidence_count": 0
      }
    ],
    "service_links": {
      "Topics": "https://hoyant-su-agentic-rl.hf.space/topics/index.html",
      "Comparisons": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/index.html",
      "Method filters": "https://hoyant-su-agentic-rl.hf.space/methods/facets",
      "Method catalog": "https://hoyant-su-agentic-rl.hf.space/topics/methods.json",
      "MCP": "https://hoyant-su-agentic-rl.hf.space/gradio_api/mcp/",
      "Tool schema": "https://hoyant-su-agentic-rl.hf.space/gradio_api/mcp/schema",
      "Skill index": "https://hoyant-su-agentic-rl.hf.space/.well-known/agent-skills/index.json"
    }
  },
  "topics": [
    {
      "slug": "credit-assignment",
      "focus": "literature",
      "title": "Credit assignment in agent reinforcement learning",
      "query": "\"credit assignment\"",
      "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/credit-assignment.html",
      "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/credit-assignment.json",
      "paper_count": 3,
      "matching_block_count": 24,
      "all_matching_blocks_included": true
    },
    {
      "slug": "cli-agents",
      "focus": "benchmark",
      "title": "Terminal benchmark task inspection",
      "query": "CLI OR \"command line\" OR \"command-line\" OR \"terminal agent\" OR \"terminal benchmark\"",
      "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/cli-agents.html",
      "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/cli-agents.json",
      "paper_count": 5,
      "matching_block_count": 35,
      "all_matching_blocks_included": true
    }
  ],
  "comparisons": {
    "title": "Agentic RL method and benchmark comparisons",
    "scope": "Qualitative comparisons of methods, information requirements and task design in the cited paper versions.",
    "paper_count": 16,
    "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/index.html",
    "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/index.json",
    "topics": [
      {
        "title": "Credit assignment methods for multi-turn agents",
        "question": "Which feedback and cross-action comparisons produce the policy's credit signal?",
        "scope": "Qualitative comparisons of methods, information requirements and task design in the cited paper versions.",
        "keywords": [
          "credit assignment",
          "advantage estimation",
          "multi-turn reinforcement learning",
          "sparse rewards"
        ],
        "dimensions": [
          "Credit unit",
          "Feedback",
          "Comparison mechanism",
          "Additional requirements"
        ],
        "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/credit-assignment-methods.html",
        "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/credit-assignment-methods.json",
        "csv_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/credit-assignment-methods.csv",
        "bibtex_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/credit-assignment-methods.bib",
        "paper_count": 8
      },
      {
        "title": "Memory, context management and selective observation in agents",
        "question": "What information is retained, when is it selected, and how is selection trained?",
        "scope": "Qualitative comparisons of methods, information requirements and task design in the cited paper versions.",
        "keywords": [
          "agent memory",
          "context management",
          "selective observation",
          "partial observability",
          "context budget"
        ],
        "dimensions": [
          "Information retained",
          "Operation and timing",
          "Training or selection rule",
          "Task setting"
        ],
        "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/agent-memory-observation.html",
        "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/agent-memory-observation.json",
        "csv_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/agent-memory-observation.csv",
        "bibtex_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/agent-memory-observation.bib",
        "paper_count": 4
      },
      {
        "title": "Terminal and software agent benchmark comparison",
        "question": "How do task origin, interaction interface and success criteria differ?",
        "scope": "Qualitative comparisons of methods, information requirements and task design in the cited paper versions.",
        "keywords": [
          "CLI agent benchmark",
          "terminal benchmark",
          "execution feedback",
          "software agent evaluation",
          "ShellOps"
        ],
        "dimensions": [
          "Task source",
          "Interaction interface",
          "Success criteria",
          "Evaluation scope"
        ],
        "html_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/terminal-benchmark-design.html",
        "json_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/terminal-benchmark-design.json",
        "csv_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/terminal-benchmark-design.csv",
        "bibtex_url": "https://hoyant-su-agentic-rl.hf.space/topics/comparisons/terminal-benchmark-design.bib",
        "paper_count": 6
      }
    ]
  }
}
