<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9" xmlns:image="http://www.google.com/schemas/sitemap-image/1.1">
<url>
<loc>https://learnrlfast.com/</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/og-image.png</image:loc>
</image:image>
<changefreq>weekly</changefreq>
<priority>1</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics</loc>
<changefreq>weekly</changefreq>
<priority>0.9</priority>
</url>
<url>
<loc>https://learnrlfast.com/about</loc>
<changefreq>monthly</changefreq>
<priority>0.6</priority>
</url>
<url>
<loc>https://learnrlfast.com/editorial-policy</loc>
<changefreq>monthly</changefreq>
<priority>0.5</priority>
</url>
<url>
<loc>https://learnrlfast.com/privacy</loc>
<changefreq>yearly</changefreq>
<priority>0.3</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics/rl-foundations</loc>
<changefreq>weekly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics/value-and-bellman-reasoning</loc>
<changefreq>weekly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics/learning-from-experience</loc>
<changefreq>weekly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics/policy-learning-and-deep-rl</loc>
<changefreq>weekly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics/rl-experiments-and-evaluation</loc>
<changefreq>weekly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics/multi-agent-reinforcement-learning</loc>
<changefreq>weekly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/topics/advanced-policy-optimization</loc>
<changefreq>weekly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/actor-critic-methods-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/actor-critic-methods-explained/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/average-reward-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/average-reward-reinforcement-learning/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/behavior-cloning-and-covariate-shift</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/behavior-cloning-and-covariate-shift/assets/images/hero-machine-learning/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/belief-states-and-updates-in-pomdps</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/belief-states-and-updates-in-pomdps/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/bellman-backups-and-dynamic-programming</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/bellman-backups-and-dynamic-programming/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/bellman-equation-intuition</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/bellman-equation-intuition/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/centralized-training-decentralized-execution</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/centralized-training-decentralized-execution/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/choosing-planning-horizon-under-model-error</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/choosing-planning-horizon-under-model-error/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/collecting-preference-data-for-reward-models</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/collecting-preference-data-for-reward-models/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/combining-demonstrations-with-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/combining-demonstrations-with-reinforcement-learning/assets/images/hero-machine-learning/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/communication-and-coordination-in-multi-agent-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/communication-and-coordination-in-multi-agent-rl/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/constrained-reinforcement-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/constrained-reinforcement-learning-explained/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/contextual-bandits-and-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/contextual-bandits-and-reinforcement-learning/assets/images/hero-computer-lab/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/continuous-action-policy-parameterization</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/continuous-action-policy-parameterization/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/cooperative-vs-competitive-multi-agent-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/cooperative-vs-competitive-multi-agent-rl/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/credit-assignment-in-multi-agent-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/credit-assignment-in-multi-agent-rl/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/credit-assignment-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/credit-assignment-in-reinforcement-learning/assets/images/hero-dune/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/dataset-support-and-coverage-in-offline-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/dataset-support-and-coverage-in-offline-rl/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/deadly-triad-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/deadly-triad-reinforcement-learning/assets/images/hero-self-driving-car/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/deep-reinforcement-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/deep-reinforcement-learning-explained/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/deterministic-policy-gradients-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/deterministic-policy-gradients-explained/assets/images/hero-pelican/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/distributional-reinforcement-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/distributional-reinforcement-learning-explained/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/dyna-style-reinforcement-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/dyna-style-reinforcement-learning-explained/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/eligibility-traces-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/eligibility-traces-reinforcement-learning/assets/images/hero-diving/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/episodes-terminals-and-horizons</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/episodes-terminals-and-horizons/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/evaluating-exploration-beyond-novelty</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/evaluating-exploration-beyond-novelty/assets/images/hero-bird/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/evaluating-multi-agent-policies-and-exploitability</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/evaluating-multi-agent-policies-and-exploitability/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/exploration-bonuses-count-novelty-and-uncertainty</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/exploration-bonuses-count-novelty-and-uncertainty/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/exploration-exploitation-tradeoff</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/exploration-exploitation-tradeoff/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/extrapolation-error-in-offline-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/extrapolation-error-in-offline-rl/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/generalization-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/generalization-in-reinforcement-learning/assets/images/hero-computer-lab/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/generalized-advantage-estimation-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/generalized-advantage-estimation-explained/assets/images/hero-fish/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/how-to-design-a-reinforcement-learning-environment</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/how-to-design-a-reinforcement-learning-environment/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/how-to-evaluate-reinforcement-learning-agents</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/how-to-evaluate-reinforcement-learning-agents/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/imitation-learning-vs-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/imitation-learning-vs-reinforcement-learning/assets/images/hero-self-driving-car/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/importance-sampling-doubly-robust-off-policy-evaluation</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/importance-sampling-doubly-robust-off-policy-evaluation/assets/images/hero-dune/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/intrinsic-motivation-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/intrinsic-motivation-in-reinforcement-learning/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/invalid-actions-and-action-masking-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/invalid-actions-and-action-masking-rl/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/inverse-reinforcement-learning-reward-ambiguity</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/inverse-reinforcement-learning-reward-ambiguity/assets/images/hero-self-driving-car/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/learning-rate-discount-factor-rl-experiment</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/learning-rate-discount-factor-rl-experiment/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/markov-decision-process-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/markov-decision-process-explained/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/math-prerequisites-for-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/math-prerequisites-for-reinforcement-learning/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/maximum-entropy-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/maximum-entropy-reinforcement-learning/assets/images/hero-school-of-fish/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/model-based-vs-model-free-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/model-based-vs-model-free-reinforcement-learning/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/model-error-compounding-in-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/model-error-compounding-in-rl/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/model-predictive-control-for-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/model-predictive-control-for-reinforcement-learning/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/monte-carlo-vs-temporal-difference-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/monte-carlo-vs-temporal-difference-learning/assets/images/hero-flying-fish/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/multi-agent-reinforcement-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/multi-agent-reinforcement-learning-explained/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/non-stationarity-in-multi-agent-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/non-stationarity-in-multi-agent-reinforcement-learning/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/off-policy-evaluation-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/off-policy-evaluation-in-reinforcement-learning/assets/images/hero-self-driving-car/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/offline-reinforcement-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/offline-reinforcement-learning-explained/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/opponent-modeling-in-multi-agent-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/opponent-modeling-in-multi-agent-rl/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/options-and-temporally-extended-actions</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/options-and-temporally-extended-actions/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/parameterized-policies-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/parameterized-policies-in-reinforcement-learning/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/partial-observability-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/partial-observability-in-reinforcement-learning/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/planning-with-a-learned-model</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/planning-with-a-learned-model/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/policies-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/policies-in-reinforcement-learning/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/policy-gradient-methods-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/policy-gradient-methods-explained/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/policy-improvement-and-generalized-policy-iteration</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/policy-improvement-and-generalized-policy-iteration/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/prediction-vs-control-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/prediction-vs-control-reinforcement-learning/assets/images/hero-beach/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/preference-based-reinforcement-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/preference-based-reinforcement-learning-explained/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/proximal-policy-optimization-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/proximal-policy-optimization-explained/assets/images/hero-sky/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/q-learning-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/q-learning-explained/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/q-learning-overestimation-bias</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/q-learning-overestimation-bias/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/recurrent-policies-and-memory-in-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/recurrent-policies-and-memory-in-rl/assets/images/hero-beach/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reinforcement-learning-baselines-and-ablations</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reinforcement-learning-baselines-and-ablations/assets/images/hero-seagull/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reinforcement-learning-failure-modes</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reinforcement-learning-failure-modes/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reinforcement-learning-for-llms</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reinforcement-learning-for-llms/assets/images/hero-vacation/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reinforcement-learning-learning-roadmap</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reinforcement-learning-learning-roadmap/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reinforcement-learning-terminology-reference</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reinforcement-learning-terminology-reference/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reinforcement-learning-vs-supervised-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reinforcement-learning-vs-supervised-learning/assets/images/hero-machine-learning/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reward-model-evaluation-and-overoptimization</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reward-model-evaluation-and-overoptimization/assets/images/hero-fish/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reward-models-from-human-preferences</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reward-models-from-human-preferences/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/reward-shaping-and-sparse-rewards</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/reward-shaping-and-sparse-rewards/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/rewards-returns-and-discounting</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/rewards-returns-and-discounting/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/risk-sensitive-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/risk-sensitive-reinforcement-learning/assets/images/hero-self-driving-car/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/safe-exploration-and-action-shielding</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/safe-exploration-and-action-shielding/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/safe-policy-improvement-from-logged-data</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/safe-policy-improvement-from-logged-data/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/sarsa-vs-q-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/sarsa-vs-q-learning/assets/images/hero-computer-lab/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/self-play-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/self-play-in-reinforcement-learning/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/sim-to-real-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/sim-to-real-reinforcement-learning/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/soft-actor-critic-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/soft-actor-critic-explained/assets/images/hero-sunset/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/stabilizing-deep-q-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/stabilizing-deep-q-learning/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/states-observations-and-actions</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/states-observations-and-actions/assets/images/hero-self-driving-car/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/tabular-reinforcement-learning-experiment</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/tabular-reinforcement-learning-experiment/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/tabular-vs-function-approximation-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/tabular-vs-function-approximation-rl/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/td3-twin-delayed-deep-deterministic-policy-gradients</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/td3-twin-delayed-deep-deterministic-policy-gradients/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/transition-dynamics-and-reward-models</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/transition-dynamics-and-reward-models/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/trust-region-policy-optimization-explained</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/trust-region-policy-optimization-explained/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/value-based-vs-policy-based-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/value-based-vs-policy-based-reinforcement-learning/assets/images/hero-robot/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/value-decomposition-multi-agent-rl</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/value-decomposition-multi-agent-rl/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/value-functions-in-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/value-functions-in-reinforcement-learning/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/value-vs-q-value-vs-advantage</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/value-vs-q-value-vs-advantage/assets/images/hero-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/what-is-reinforcement-learning</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/what-is-reinforcement-learning/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.8</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/when-deep-rl-is-the-wrong-tool</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/when-deep-rl-is-the-wrong-tool/assets/images/hero-gambling/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/when-hierarchical-reinforcement-learning-helps</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/when-hierarchical-reinforcement-learning-helps/assets/images/hero-chess/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
<url>
<loc>https://learnrlfast.com/blog/why-bellman-updates-converge</loc>
<image:image>
<image:loc>https://assets.worldmonger.com/sites/learnrlfast/articles/why-bellman-updates-converge/assets/images/hero-neural-network/image.webp</image:loc>
</image:image>
<lastmod>2026-09-12T00:00:00.000Z</lastmod>
<changefreq>monthly</changefreq>
<priority>0.7</priority>
</url>
</urlset>
