[{"data":1,"prerenderedAt":293},["ShallowReactive",2],{"navigation_docs_en":3,"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02":114,"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-surround":288},[4],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":45},"AI Engineering",null,"\u002Fen\u002Fai-engineering","en\u002F1.ai-engineering",[10,46,77],{"title":11,"icon":12,"path":13,"stem":14,"children":15,"page":45},"Introduction to Building AI Applications with Foundation Models","i-lucide-brain-circuit","\u002Fen\u002Fai-engineering\u002Fintro","en\u002F1.ai-engineering\u002F1.intro",[16,20,25,30,35,40],{"title":11,"path":17,"stem":18,"icon":19},"\u002Fen\u002Fai-engineering\u002Fintro\u002Fch01","en\u002F1.ai-engineering\u002F1.intro\u002Fch01","i-lucide-sparkles",{"title":21,"path":22,"stem":23,"icon":24},"The Rise of AI Engineering","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch011-the-rise-of-ai-engineering","en\u002F1.ai-engineering\u002F1.intro\u002Fch011-the-rise-of-ai-engineering","i-lucide-history",{"title":26,"path":27,"stem":28,"icon":29},"Foundation Model Use Cases","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch012-foundation-model-use-cases","en\u002F1.ai-engineering\u002F1.intro\u002Fch012-foundation-model-use-cases","i-lucide-layout-grid",{"title":31,"path":32,"stem":33,"icon":34},"Planning AI Applications","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch013-planning-ai-applications","en\u002F1.ai-engineering\u002F1.intro\u002Fch013-planning-ai-applications","i-lucide-clipboard-list",{"title":36,"path":37,"stem":38,"icon":39},"The AI Engineering Stack","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch014-the-ai-engineering-stack","en\u002F1.ai-engineering\u002F1.intro\u002Fch014-the-ai-engineering-stack","i-lucide-layers",{"title":41,"path":42,"stem":43,"icon":44},"Summary","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch015-summary","en\u002F1.ai-engineering\u002F1.intro\u002Fch015-summary","i-lucide-flag",false,{"title":47,"icon":6,"path":48,"stem":49,"children":50,"page":45},"Understanding Foundation Models","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models","en\u002F1.ai-engineering\u002F2.understanding-foundation-models",[51,54,59,64,69,74],{"title":47,"path":52,"stem":53,"icon":12},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02",{"title":55,"path":56,"stem":57,"icon":58},"Training Data","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-1-training-data","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-1-training-data","i-lucide-database",{"title":60,"path":61,"stem":62,"icon":63},"Modeling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-2-modeling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-2-modeling","i-lucide-network",{"title":65,"path":66,"stem":67,"icon":68},"Post-Training","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-3-post-training","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-3-post-training","i-lucide-sliders-horizontal",{"title":70,"path":71,"stem":72,"icon":73},"Sampling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-4-sampling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-4-sampling","i-lucide-dices",{"title":41,"path":75,"stem":76,"icon":44},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-5-summary","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-5-summary",{"title":78,"path":79,"stem":80,"children":81,"page":45},"Evaluation Methodology","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology","en\u002F1.ai-engineering\u002F3.evaluation-methodology",[82,86,91,96,101,106,111],{"title":78,"path":83,"stem":84,"icon":85},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03","i-lucide-clipboard-check",{"title":87,"path":88,"stem":89,"icon":90},"Challenges of Evaluating Foundation Models","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","i-lucide-shield-alert",{"title":92,"path":93,"stem":94,"icon":95},"Understanding Language Modeling Metrics","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","i-lucide-sigma",{"title":97,"path":98,"stem":99,"icon":100},"Exact Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-3-exact-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-3-exact-evaluation","i-lucide-check-check",{"title":102,"path":103,"stem":104,"icon":105},"AI as a Judge","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-4-ai-as-a-judge","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-4-ai-as-a-judge","i-lucide-scale",{"title":107,"path":108,"stem":109,"icon":110},"Ranking Models with Comparative Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","i-lucide-trophy",{"title":41,"path":112,"stem":113,"icon":44},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-6-summary","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-6-summary",{"id":115,"title":47,"body":116,"description":282,"extension":283,"links":6,"meta":284,"navigation":285,"path":52,"seo":286,"stem":53,"__hash__":287},"docs_en\u002Fen\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02.md",{"type":117,"value":118,"toc":273},"minimark",[119,134,139,143,151,154,158,181,185,189,192,209,213,216,233,241,245,252,260,263,267,270],[120,121,122,126],"u-page-hero",{},[123,124,47],"template",{"v-slot:title":125},"",[123,127,128,129,133],{"v-slot:description":125},"To build applications with foundation models, you first need ",[130,131,132],"strong",{},"foundation models",". While you don't need to know how to develop a model to use it, a high-level understanding will help you decide what model to use and how to adapt it to your needs.",[135,136,138],"h2",{"id":137},"what-this-chapter-can-and-cant-do","What This Chapter Can and Can't Do",[140,141,142],"p",{},"Training a foundation model is an incredibly complex and costly process. Those who know how to do this well are likely prevented by confidentiality agreements from disclosing the secret sauce.",[144,145,146,147,150],"warning",{},"This chapter won't be able to tell you how to build a model to compete with ChatGPT. Instead, it'll focus on ",[130,148,149],{},"design decisions with consequential impact on downstream applications",".",[140,152,153],{},"With the growing lack of transparency in the training process of foundation models, it's difficult to know all the design decisions that go into making a model. In general, however, differences in foundation models can be traced back to decisions about training data, model architecture and size, and how they are post-trained to align with human preferences.",[135,155,157],{"id":156},"the-decisions-that-shape-a-model","The Decisions That Shape a Model",[159,160,161,169,173,178],"card-group",{},[162,163,164,165,168],"card",{"icon":58,"title":55},"Since models learn from data, their training data reveals a great deal about their ",[130,166,167],{},"capabilities and limitations",". This chapter begins with how model developers curate training data, focusing on the distribution of training data.",[162,170,172],{"icon":63,"title":171},"Architecture","Given the dominance of the transformer architecture, it might seem that model architecture is less of a choice. You might be wondering what makes the transformer architecture so special that it continues to dominate.",[162,174,177],{"icon":175,"title":176},"i-lucide-maximize","Size","Whenever a new model is released, one of the first things people want to know is its size. This chapter will explore how a model developer might determine the appropriate size for their model.",[162,179,180],{"icon":68,"title":65},"Pre-training makes a model capable, but not necessarily safe or easy to use. Post-training aligns the model with human preferences, which has a significant impact on the model's usability.",[182,183,184],"note",{},"Chapter 8 explores dataset engineering techniques in detail, including data quality evaluation and data synthesis.",[135,186,188],{"id":187},"why-architecture-still-matters","Why Architecture Still Matters",[140,190,191],{},"Given the dominance of the transformer architecture, it might seem that model architecture is less of a choice.",[193,194,195,201,205],"accordion",{},[196,197,200],"accordion-item",{"icon":198,"label":199},"i-lucide-circle-help","What makes the transformer architecture so special that it continues to dominate?","This chapter will address this question.",[196,202,204],{"icon":198,"label":203},"How long until another architecture takes over?","This chapter will address this question, too.",[196,206,208],{"icon":198,"label":207},"What might this new architecture look like?","This chapter will also address what a future alternative architecture might look like.",[135,210,212],{"id":211},"from-capability-to-usability","From Capability to Usability",[140,214,215],{},"As mentioned in Chapter 1, a model's training process is often divided into pre-training and post-training.",[159,217,218,226],{},[162,219,221,222,225],{"icon":19,"title":220},"Pre-Training","Pre-training makes a model ",[130,223,224],{},"capable",", but not necessarily safe or easy to use.",[162,227,229,230,150],{"icon":228,"title":65},"i-lucide-user-check","Post-training aims to align the model with ",[130,231,232],{},"human preferences",[140,234,235,236,240],{},"But what exactly is ",[237,238,239],"em",{},"human preference","? How can it be represented in a way that a model can learn? The way a model developer aligns their model has a significant impact on the model's usability, and will be discussed in this chapter.",[135,242,244],{"id":243},"the-underrated-role-of-sampling","The Underrated Role of Sampling",[140,246,247,248,251],{},"While most people understand the impact of training on a model's performance, the impact of ",[237,249,250],{},"sampling"," is often overlooked. Sampling is how a model chooses an output from all possible options. It is perhaps one of the most underrated concepts in AI.",[253,254,255,256,259],"tip",{},"Sampling explains many seemingly baffling AI behaviors, including ",[130,257,258],{},"hallucinations and inconsistencies",". Choosing the right sampling strategy can also significantly boost a model's performance with relatively little effort.",[140,261,262],{},"For this reason, sampling is the section that I was the most excited to write about in this chapter.",[135,264,266],{"id":265},"how-to-use-this-chapter","How to Use This Chapter",[182,268,269],{},"Concepts covered in this chapter are fundamental for understanding the rest of the book. However, because these concepts are fundamental, you might already be familiar with them.",[140,271,272],{},"Feel free free to skip any concept that you're confident about. If you encounter a confusing concept later on, you can revisit this chapter.",{"title":125,"searchDepth":274,"depth":274,"links":275},2,[276,277,278,279,280,281],{"id":137,"depth":274,"text":138},{"id":156,"depth":274,"text":157},{"id":187,"depth":274,"text":188},{"id":211,"depth":274,"text":212},{"id":243,"depth":274,"text":244},{"id":265,"depth":274,"text":266},"A guide to how training data, architecture, size, post-training, and sampling shape foundation model behavior.","md",{},{"icon":12},{"title":47,"description":282},"ZqqcQ2Fgc22Bjz50TxD9joHKkaw7QUCjjeyKL_clKGI",[289,291],{"title":41,"path":42,"stem":43,"description":290,"icon":44,"children":-1},"A recap of how foundation models gave rise to AI engineering, the application patterns enabled, and the framework this book provides.",{"title":55,"path":56,"stem":57,"description":292,"icon":58,"children":-1},"How training data quality, language coverage, and domain coverage shape foundation model capability, cost, and reliability.",1788871974598]