<?xml version="1.0" encoding="UTF-8"?><urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:news="http://www.google.com/schemas/sitemap-news/0.9" xmlns:xhtml="http://www.w3.org/1999/xhtml" xmlns:image="http://www.google.com/schemas/sitemap-image/1.1" xmlns:video="http://www.google.com/schemas/sitemap-video/1.1"><url><loc>https://lily-feng.github.io/Reinforcement-Learning</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/01-foundations/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/02-multi-armed-bandits/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/02-multi-armed-bandits/action-value-methods/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/02-multi-armed-bandits/epsilon-greedy/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/02-multi-armed-bandits/gradient-bandit/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/02-multi-armed-bandits/nonstationary-problems/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/02-multi-armed-bandits/optimistic-initial-values/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/02-multi-armed-bandits/upper-confidence-bound/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/01-agent-env-interface/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/02-goals-and-rewards/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/03-returns-and-episodes/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/04-unified-notation/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/05-policies-and-value-functions/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/06-optimal-policies-and-value-functions/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/03-markov-decision-processes/07-optimality-and-approximation/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/04-dynamic-programming/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/05-monte-carlo-methods/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/06-temporal-difference-learning/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/07-n-step-and-eligibility-traces/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/08-planning-and-learning/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/09-function-approximation/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/10-deep-q-networks/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/11-policy-gradient-methods/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/chapters/12-advanced-policy-optimization/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/demos/dynamic-programming/gridworld/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/demos/monte-carlo-methods/returns/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/demos/multi-armed-bandits/epsilon-greedy/</loc></url><url><loc>https://lily-feng.github.io/Reinforcement-Learning/getting-started/</loc></url></urlset>