Home / Current Issue / Paper 1722594
Multi-Agent Reinforcement Learning for Dynamic Inventory Rebalancing and Last-Mile Fulfillment Under Supply Chain Disruptions
Subject area: Science,Engineering and Technology · Area of research: Machine Learning, AI
Abstract
Supply chain disruptions propagate rapidly through multi-echelon networks, and most learning-based approaches stop at prediction rather than acting on it. This paper advances from disruption forecasting toward autonomous mitigation by framing dynamic inventory rebalancing and last-mile fulfillment as a cooperative multi-agent reinforcement learning problem. Each facility in the network is an independent agent that jointly decides (i) replenishment and lateral transshipment quantities to rebalance inventory across echelons, and (ii) fulfillment assignments that reroute customer orders through available last-mile capacity when a disruption degrades primary routes. The agents are trained under a centralized-training–decentralized-execution paradigm with a graph-neural-network state encoder and a clipped proximal-policy-optimization core and are exposed during training to a stochastic disruption generator so that mitigation policies are learned proactively rather than reactively. Experiments on a three-echelon network under supplier-failure, hub-failure, and transport-disruption scenarios show that the learned policy sustains service levels, shortens recovery time, and reduces total disruption cost relative to base-stock and single-agent baselines. The results demonstrate that prediction must be coupled with autonomous optimization to deliver practical resilience.
Keywords
multi-agent reinforcement learning, inventory rebalancing, last-mile fulfillment, supply chain disruptions, resilience optimization, autonomous decision-making.
How to cite this paper
@article{1722594,
author = {Sohail Sayed, Nauman Sayed},
title = {Multi-Agent Reinforcement Learning for Dynamic Inventory Rebalancing and Last-Mile Fulfillment Under Supply Chain Disruptions},
journal = {Iconic Research And Engineering Journals},
year = {2025},
volume = {8},
number = {9},
pages = {2070-2078},
issn = {2456-8880},
url = {https://www.irejournals.com/formatedpaper/1722594.pdf},
abstract = {Supply chain disruptions propagate rapidly through multi-echelon networks, and most learning-based approaches stop at prediction rather than acting on it. This paper advances from disruption forecasting toward autonomous mitigation by framing dynamic inventory rebalancing and last-mile fulfillment as a cooperative multi-agent reinforcement learning problem. Each facility in the network is an independent agent that jointly decides (i) replenishment and lateral transshipment quantities to rebalance inventory across echelons, and (ii) fulfillment assignments that reroute customer orders through available last-mile capacity when a disruption degrades primary routes. The agents are trained under a centralized-training–decentralized-execution paradigm with a graph-neural-network state encoder and a clipped proximal-policy-optimization core and are exposed during training to a stochastic disruption generator so that mitigation policies are learned proactively rather than reactively. Experiments on a three-echelon network under supplier-failure, hub-failure, and transport-disruption scenarios show that the learned policy sustains service levels, shortens recovery time, and reduces total disruption cost relative to base-stock and single-agent baselines. The results demonstrate that prediction must be coupled with autonomous optimization to deliver practical resilience.},
keywords = {multi-agent reinforcement learning, inventory rebalancing, last-mile fulfillment, supply chain disruptions, resilience optimization, autonomous decision-making.},
month = {March},
}