{"schema_version":"1.0","language":"zh-CN","generated_from":"NEWVAR 从零开始学AI structured Markdown","write_access":false,"full_text_access":false,"content_boundary":"This endpoint exposes metadata, objectives, prerequisites, labs and source summaries. It does not expose chapter body text or answer text.","chapter":{"slug":"mdp-bellman","title":"MDP、回报与贝尔曼方程","english_title":"MDPs and Bellman equations","chapter_number":65,"volume":3,"volume_title":"第三卷：主要AI分支","month":17,"unit_title":"强化学习与控制","review_status":"draft","updated_at":"2026-08-06","estimated_hours":8,"focus":"理解状态、动作、转移、奖励、折扣和马尔可夫假设。","objectives":["理解状态、动作、转移、奖励、折扣和马尔可夫假设。","解释核心数学关系并完成最小实现","运行单变量实验并记录失败样例","完成工程线或研究线至少一项迁移任务"],"prerequisites":["flows-energy-evaluation"],"exercise_count":3,"tracks":["engineering","research"],"human_url":"/learn/mdp-bellman"},"labs":[{"id":"lab-mdp-bellman","chapter_slug":"mdp-bellman","title":"MDP、回报与贝尔曼方程：单变量实验","kind":"discount_factor","data_status":"教学模拟","variable":{"label":"折扣因子","unit":"γ","min":0,"max":1,"step":0.01,"default":0.95},"objective":"只改变“折扣因子”，观察结果、代价和风险如何一起变化，理解理解状态、动作、转移、奖励、折扣和马尔可夫假设。","static_fallback":"静态替代：把折扣因子分别设为0γ、0.95γ和1γ，手工比较三组教学模拟输出。","keyboard":"聚焦滑块后使用方向键微调，Page Up/Page Down大步调整，Home/End到达边界。","measurement_note":"页面数值由公开公式生成，只用于教学，不代表真实模型性能或现实世界因果效果。","source_ids":["S040","S041","S042","S043"]}],"project":{"id":"project-17","month":17,"title":"训练一个可解释的网格世界智能体","brief":"用教学模拟环境记录奖励、策略、探索、种子、学习曲线和失败轨迹，不把单次成功当结论。","deliverables":["问题与边界说明","可运行最小实现","实验记录与失败分析","来源与许可清单","风险、隐私与人工闸门","复现README"]},"sources":[{"id":"S040","author":"Richard Sutton, Andrew Barto","title":"Reinforcement Learning: An Introduction","source_level":"作者开放教材","published_at":"2018","accessed_at":"2026-08-06","doi_or_url":"http://incompleteideas.net/book/the-book-2nd.html"},{"id":"S041","author":"UC Berkeley","title":"CS 285: Deep Reinforcement Learning","source_level":"大学课程一手资料","published_at":"持续更新","accessed_at":"2026-08-06","doi_or_url":"https://rail.eecs.berkeley.edu/deeprlcourse/"},{"id":"S042","author":"Volodymyr Mnih et al.","title":"Human-level control through deep reinforcement learning","source_level":"同行评审论文","published_at":"2015","accessed_at":"2026-08-06","doi_or_url":"https://doi.org/10.1038/nature14236"},{"id":"S043","author":"John Schulman et al.","title":"Proximal Policy Optimization Algorithms","source_level":"研究论文预印本","published_at":"2017","accessed_at":"2026-08-06","doi_or_url":"https://arxiv.org/abs/1707.06347"}]}