{"schema_version":"1.0","language":"zh-CN","generated_from":"NEWVAR 从零开始学AI structured Markdown","write_access":false,"full_text_access":false,"content_boundary":"This endpoint exposes metadata, objectives, prerequisites, labs and source summaries. It does not expose chapter body text or answer text.","chapter":{"slug":"inference-compression-serving","title":"推理、量化、蒸馏与服务","english_title":"Inference and serving","chapter_number":87,"volume":4,"volume_title":"第四卷：前沿、系统与研究","month":22,"unit_title":"数据、规模化训练、推理和MLOps","review_status":"draft","updated_at":"2026-08-06","estimated_hours":10,"focus":"理解批处理、缓存、量化、剪枝、蒸馏、吞吐、延迟和质量折衷。","objectives":["理解批处理、缓存、量化、剪枝、蒸馏、吞吐、延迟和质量折衷。","解释核心数学关系并完成最小实现","运行单变量实验并记录失败样例","完成工程线或研究线至少一项迁移任务"],"prerequisites":["distributed-training"],"exercise_count":3,"tracks":["engineering","research"],"human_url":"/learn/inference-compression-serving"},"labs":[{"id":"lab-inference-compression-serving","chapter_slug":"inference-compression-serving","title":"推理、量化、蒸馏与服务：单变量实验","kind":"quantization_bits","data_status":"教学模拟","variable":{"label":"量化位宽","unit":"bit","min":2,"max":16,"step":1,"default":8},"objective":"只改变“量化位宽”，观察结果、代价和风险如何一起变化，理解理解批处理、缓存、量化、剪枝、蒸馏、吞吐、延迟和质量折衷。","static_fallback":"静态替代：把量化位宽分别设为2bit、8bit和16bit，手工比较三组教学模拟输出。","keyboard":"聚焦滑块后使用方向键微调，Page Up/Page Down大步调整，Home/End到达边界。","measurement_note":"页面数值由公开公式生成，只用于教学，不代表真实模型性能或现实世界因果效果。","source_ids":["S050","S061","S062","S063","S064","S065","S066","S067"]}],"project":{"id":"project-22","month":22,"title":"部署一个有监控和回滚的小模型服务","brief":"建立数据版本、训练记录、离线评估、CPU推理、延迟预算、漂移告警和回滚说明。","deliverables":["问题与边界说明","可运行最小实现","实验记录与失败分析","来源与许可清单","风险、隐私与人工闸门","复现README"]},"sources":[{"id":"S050","author":"Stanford University","title":"CS336: Language Modeling from Scratch","source_level":"大学课程一手资料","published_at":"2026","accessed_at":"2026-08-06","doi_or_url":"https://cs336.stanford.edu/"},{"id":"S061","author":"PyTorch","title":"Fully Sharded Data Parallel","source_level":"官方文档","published_at":"持续更新","accessed_at":"2026-08-06","doi_or_url":"https://docs.pytorch.org/docs/stable/fsdp.html"},{"id":"S062","author":"Mohammad Shoeybi et al.","title":"Megatron-LM","source_level":"研究论文预印本","published_at":"2019","accessed_at":"2026-08-06","doi_or_url":"https://arxiv.org/abs/1909.08053"},{"id":"S063","author":"Samyam Rajbhandari et al.","title":"ZeRO","source_level":"同行评审论文预印本","published_at":"2019","accessed_at":"2026-08-06","doi_or_url":"https://arxiv.org/abs/1910.02054"},{"id":"S064","author":"Tri Dao et al.","title":"FlashAttention","source_level":"同行评审论文预印本","published_at":"2022","accessed_at":"2026-08-06","doi_or_url":"https://arxiv.org/abs/2205.14135"},{"id":"S065","author":"Woosuk Kwon et al.","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","source_level":"同行评审论文预印本","published_at":"2023","accessed_at":"2026-08-06","doi_or_url":"https://arxiv.org/abs/2309.06180"},{"id":"S066","author":"MLflow Project","title":"MLflow Documentation","source_level":"官方文档","published_at":"持续更新","accessed_at":"2026-08-06","doi_or_url":"https://mlflow.org/docs/latest/"},{"id":"S067","author":"Kubernetes","title":"Kubernetes Documentation","source_level":"官方文档","published_at":"持续更新","accessed_at":"2026-08-06","doi_or_url":"https://kubernetes.io/docs/home/"}]}