[{"data":1,"prerenderedAt":546},["ShallowReactive",2],{"navigation_docs_en":3,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03":114,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-surround":541},[4],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":45},"AI Engineering",null,"\u002Fen\u002Fai-engineering","en\u002F1.ai-engineering",[10,46,77],{"title":11,"icon":12,"path":13,"stem":14,"children":15,"page":45},"Introduction to Building AI Applications with Foundation Models","i-lucide-brain-circuit","\u002Fen\u002Fai-engineering\u002Fintro","en\u002F1.ai-engineering\u002F1.intro",[16,20,25,30,35,40],{"title":11,"path":17,"stem":18,"icon":19},"\u002Fen\u002Fai-engineering\u002Fintro\u002Fch01","en\u002F1.ai-engineering\u002F1.intro\u002Fch01","i-lucide-sparkles",{"title":21,"path":22,"stem":23,"icon":24},"The Rise of AI Engineering","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch011-the-rise-of-ai-engineering","en\u002F1.ai-engineering\u002F1.intro\u002Fch011-the-rise-of-ai-engineering","i-lucide-history",{"title":26,"path":27,"stem":28,"icon":29},"Foundation Model Use Cases","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch012-foundation-model-use-cases","en\u002F1.ai-engineering\u002F1.intro\u002Fch012-foundation-model-use-cases","i-lucide-layout-grid",{"title":31,"path":32,"stem":33,"icon":34},"Planning AI Applications","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch013-planning-ai-applications","en\u002F1.ai-engineering\u002F1.intro\u002Fch013-planning-ai-applications","i-lucide-clipboard-list",{"title":36,"path":37,"stem":38,"icon":39},"The AI Engineering Stack","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch014-the-ai-engineering-stack","en\u002F1.ai-engineering\u002F1.intro\u002Fch014-the-ai-engineering-stack","i-lucide-layers",{"title":41,"path":42,"stem":43,"icon":44},"Summary","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch015-summary","en\u002F1.ai-engineering\u002F1.intro\u002Fch015-summary","i-lucide-flag",false,{"title":47,"icon":6,"path":48,"stem":49,"children":50,"page":45},"Understanding Foundation Models","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models","en\u002F1.ai-engineering\u002F2.understanding-foundation-models",[51,54,59,64,69,74],{"title":47,"path":52,"stem":53,"icon":12},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02",{"title":55,"path":56,"stem":57,"icon":58},"Training Data","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-1-training-data","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-1-training-data","i-lucide-database",{"title":60,"path":61,"stem":62,"icon":63},"Modeling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-2-modeling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-2-modeling","i-lucide-network",{"title":65,"path":66,"stem":67,"icon":68},"Post-Training","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-3-post-training","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-3-post-training","i-lucide-sliders-horizontal",{"title":70,"path":71,"stem":72,"icon":73},"Sampling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-4-sampling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-4-sampling","i-lucide-dices",{"title":41,"path":75,"stem":76,"icon":44},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-5-summary","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-5-summary",{"title":78,"path":79,"stem":80,"children":81,"page":45},"Evaluation Methodology","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology","en\u002F1.ai-engineering\u002F3.evaluation-methodology",[82,86,91,96,101,106,111],{"title":78,"path":83,"stem":84,"icon":85},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03","i-lucide-clipboard-check",{"title":87,"path":88,"stem":89,"icon":90},"Challenges of Evaluating Foundation Models","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","i-lucide-shield-alert",{"title":92,"path":93,"stem":94,"icon":95},"Understanding Language Modeling Metrics","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","i-lucide-sigma",{"title":97,"path":98,"stem":99,"icon":100},"Exact Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-3-exact-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-3-exact-evaluation","i-lucide-check-check",{"title":102,"path":103,"stem":104,"icon":105},"AI as a Judge","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-4-ai-as-a-judge","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-4-ai-as-a-judge","i-lucide-scale",{"title":107,"path":108,"stem":109,"icon":110},"Ranking Models with Comparative Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","i-lucide-trophy",{"title":41,"path":112,"stem":113,"icon":44},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-6-summary","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-6-summary",{"id":115,"title":78,"body":116,"description":535,"extension":536,"links":6,"meta":537,"navigation":538,"path":83,"seo":539,"stem":84,"__hash__":540},"docs_en\u002Fen\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03.md",{"type":117,"value":118,"toc":516},"minimark",[119,134,139,177,185,189,200,210,214,221,248,252,259,269,286,298,302,310,339,348,356,360,369,385,389,396,423,426,433,440,460,463,467],[120,121,122,126],"u-page-hero",{},[123,124,78],"template",{"v-slot:title":125},"",[123,127,128,129,133],{"v-slot:description":125},"The more AI is used, the more opportunity there is for ",[130,131,132],"strong",{},"catastrophic failure",". We've already seen many failures in the short time that foundation models have been around.",[135,136,138],"h2",{"id":137},"failures-we-have-already-seen","Failures We Have Already Seen",[140,141,142,157,167],"card-group",{},[143,144,149,150,156],"card",{"icon":145,"title":146,"target":147,"to":148},"i-lucide-bot","Chatbot Encouraged Suicide","_blank","https:\u002F\u002Fwww.vice.com\u002Fen\u002Farticle\u002Fman-dies-by-suicide-after-talking-with-ai-chatbot-widow-says\u002F","A man committed suicide after being ",[151,152,155],"a",{"href":148,"rel":153},[154],"nofollow","encouraged by a chatbot",".",[143,158,162,163,156],{"icon":159,"title":160,"target":147,"to":161},"i-lucide-gavel","Hallucinated Court Evidence","https:\u002F\u002Ffortune.com\u002F2023\u002F06\u002F23\u002Flawyers-fined-filing-chatgpt-hallucinations-in-court\u002F","Lawyers submitted ",[151,164,166],{"href":161,"rel":165},[154],"false evidence hallucinated by AI",[143,168,172,173,156],{"icon":169,"title":170,"target":147,"to":171},"i-lucide-plane","Airline Chatbot Misled a Passenger","https:\u002F\u002Fwww.cio.com\u002Farticle\u002F190888\u002F5-famous-analytics-and-ai-disasters.html","Air Canada was ordered to pay damages when its AI chatbot ",[151,174,176],{"href":171,"rel":175},[154],"gave a passenger false information",[178,179,180,181,184],"caution",{},"Without a way to quality control AI outputs, the ",[130,182,183],{},"risk of AI might outweigh its benefits"," for many applications.",[135,186,188],{"id":187},"the-biggest-hurdle","The Biggest Hurdle",[190,191,192,193,196,197,156],"p",{},"As teams rush to adopt AI, many quickly realize that the biggest hurdle to bringing AI applications to reality is ",[130,194,195],{},"evaluation",". For some applications, figuring out evaluation can take up the ",[130,198,199],{},"majority of the development effort",[201,202,203,204,209],"note",{},"In December 2023, ",[151,205,208],{"href":206,"rel":207},"https:\u002F\u002Fx.com\u002Fgdb\u002Fstatus\u002F1733553161884127435",[154],"Greg Brockman, an OpenAI cofounder, tweeted"," that \"evals are surprisingly often all you need.\"",[135,211,213],{"id":212},"two-chapters-on-evaluation","Two Chapters on Evaluation",[190,215,216,217,220],{},"Due to the importance and complexity of evaluation, this book has ",[130,218,219],{},"two chapters"," on it.",[140,222,223,235],{},[143,224,227,228,231,232,156],{"icon":225,"title":226},"i-lucide-flask-conical","This Chapter — Methods","Different evaluation methods used to evaluate ",[130,229,230],{},"open-ended models",", how these methods work, and their ",[130,233,234],{},"limitations",[143,236,239,240,243,244,247],{"icon":237,"title":238},"i-lucide-workflow","Next Chapter — Application","How to use these methods to ",[130,241,242],{},"select models"," for your application and ",[130,245,246],{},"build an evaluation pipeline"," to evaluate your application.",[135,249,251],{"id":250},"evaluation-in-the-context-of-a-whole-system","Evaluation in the Context of a Whole System",[190,253,254,255,258],{},"While I discuss evaluation in its own chapters, evaluation has to be considered in the context of a ",[130,256,257],{},"whole system",", not in isolation.",[190,260,261,262,265,266,156],{},"Evaluation aims to ",[130,263,264],{},"mitigate risks"," and ",[130,267,268],{},"uncover opportunities",[140,270,271,279],{},[143,272,274,275,278],{"icon":90,"title":273},"Mitigate Risks","To mitigate risks, you first need to identify the places where your system is ",[130,276,277],{},"likely to fail"," and design your evaluation around them.",[143,280,283,284,156],{"icon":281,"title":282},"i-lucide-lightbulb","Uncover Opportunities","Evaluation also aims to ",[130,285,268],{},[287,288,289,290,293,294,297],"warning",{},"Often, this may require ",[130,291,292],{},"redesigning your system"," to enhance visibility into its failures. Without a clear understanding of where your system fails, ",[130,295,296],{},"no amount of evaluation metrics or tools"," can make the system robust.",[135,299,301],{"id":300},"why-people-skip-systematic-evaluation","Why People Skip Systematic Evaluation",[190,303,304,305,309],{},"Before diving into evaluation methods, it's important to acknowledge the challenges of evaluating foundation models. Because evaluation is difficult, many people settle for ",[306,307,308],"em",{},"word of mouth"," (e.g., someone says that the model X is good) or eyeballing the results.",[140,311,312,331],{},[143,313,316,317,320,321,326,327,330],{"icon":314,"title":315},"i-lucide-messages-square","Word of Mouth","Someone says that ",[130,318,319],{},"model X is good",". A 2023 study by ",[151,322,325],{"href":323,"rel":324},"https:\u002F\u002Fa16z.com\u002Fgenerative-ai-enterprise-2024\u002F",[154],"a16z"," showed that ",[130,328,329],{},"6 out of 70"," decision makers evaluated models by word of mouth.",[143,332,335,336,156],{"icon":333,"title":334},"i-lucide-eye","Eyeballing the Results","Also known as a ",[306,337,338],{},"vibe check",[287,340,341,342,265,345,156],{},"This creates even more ",[130,343,344],{},"risk",[130,346,347],{},"slows application iteration",[349,350,351,352,355],"tip",{},"Instead, we need to invest in ",[130,353,354],{},"systematic evaluation"," to make the results more reliable.",[135,357,359],{"id":358},"language-modeling-metrics","Language Modeling Metrics",[190,361,362,363,265,366,156],{},"Since many foundation models have a language model component, this chapter will provide a quick overview of the metrics used to evaluate language models, including ",[130,364,365],{},"cross entropy",[130,367,368],{},"perplexity",[140,370,371,379],{},[143,372,374,375,378],{"icon":95,"title":373},"Cross Entropy","Essential for guiding the ",[130,376,377],{},"training and finetuning"," of language models, and frequently used in many evaluation methods.",[143,380,374,383,378],{"icon":381,"title":382},"i-lucide-gauge","Perplexity",[130,384,377],{},[135,386,388],{"id":387},"open-ended-models-need-different-practices","Open-Ended Models Need Different Practices",[190,390,391,392,395],{},"Evaluating foundation models is especially challenging because they are ",[130,393,394],{},"open-ended",", and I'll cover best practices for how to tackle these.",[140,397,398,411],{},[143,399,402,403,406,407,410],{"icon":400,"title":401},"i-lucide-users","Human Evaluators","Using human evaluators remains a ",[130,404,405],{},"necessary option"," for many applications. Given how ",[130,408,409],{},"slow and expensive"," human annotations can be, the goal is to automate the process.",[143,412,415,416,265,419,422],{"icon":413,"title":414},"i-lucide-cpu","Automatic Evaluation","This book focuses on automatic evaluation, which includes both ",[130,417,418],{},"exact",[130,420,421],{},"subjective"," evaluation.",[135,424,102],{"id":425},"ai-as-a-judge",[190,427,428,429,432],{},"The rising star of subjective evaluation is ",[130,430,431],{},"AI as a judge"," — the approach of using AI to evaluate AI responses.",[201,434,435,436,439],{},"It's subjective because the score depends on ",[130,437,438],{},"what model and prompt"," the AI judge uses.",[140,441,442,451],{},[143,443,446,447,450],{"icon":444,"title":445},"i-lucide-trending-up","Rapid Traction","This approach is gaining ",[130,448,449],{},"rapid traction"," in the industry.",[143,452,455,456,459],{"icon":453,"title":454},"i-lucide-shield-off","Intense Opposition","It also invites intense opposition from those who believe that ",[130,457,458],{},"AI isn't trustworthy enough"," for this important task.",[349,461,462],{},"I'm especially excited to go deeper into this discussion, and I hope you will be, too.",[135,464,466],{"id":465},"what-this-chapter-covers","What This Chapter Covers",[468,469,471,475,484,487,495,498,504,507,510,513],"steps",{"level":470},"3",[472,473,87],"h3",{"id":474},"challenges-of-evaluating-foundation-models",[190,476,477,478,480,481,483],{},"Why evaluating foundation models is hard, including the limits of ",[306,479,308],{}," and vibe checks, and best practices for ",[130,482,394],{}," models.",[472,485,92],{"id":486},"understanding-language-modeling-metrics",[190,488,489,490,265,492,494],{},"A quick overview of ",[130,491,365],{},[130,493,368],{}," — metrics essential for training, finetuning, and many evaluation methods.",[472,496,97],{"id":497},"exact-evaluation",[190,499,500,501,503],{},"One half of automatic evaluation: ",[130,502,418],{}," methods, alongside subjective evaluation.",[472,505,102],{"id":506},"ai-as-a-judge-1",[190,508,509],{},"The rising star of subjective evaluation — using AI to evaluate AI responses, including why it is taking off and why it is contested.",[472,511,107],{"id":512},"ranking-models-with-comparative-evaluation",[190,514,515],{},"Ranking models with comparative evaluation.",{"title":125,"searchDepth":517,"depth":517,"links":518},2,[519,520,521,522,523,524,525,526,527],{"id":137,"depth":517,"text":138},{"id":187,"depth":517,"text":188},{"id":212,"depth":517,"text":213},{"id":250,"depth":517,"text":251},{"id":300,"depth":517,"text":301},{"id":358,"depth":517,"text":359},{"id":387,"depth":517,"text":388},{"id":425,"depth":517,"text":102},{"id":465,"depth":517,"text":466,"children":528},[529,531,532,533,534],{"id":474,"depth":530,"text":87},3,{"id":486,"depth":530,"text":92},{"id":497,"depth":530,"text":97},{"id":506,"depth":530,"text":102},{"id":512,"depth":530,"text":107},"How to evaluate open-ended foundation models: language-modeling metrics, exact and subjective methods, AI as a judge, and their limitations.","md",{},{"icon":85},{"title":78,"description":535},"_7v-ohB0C9k-x8Sel0x10CS-Ms2C6vvXgtBMeWcd-ak",[542,544],{"title":41,"path":75,"stem":76,"description":543,"icon":44,"children":-1},"A recap of how training data, modeling choices, post-training, and sampling shape foundation model behavior.",{"title":87,"path":88,"stem":89,"description":545,"icon":90,"children":-1},"Why evaluating foundation models is harder than traditional ML — intelligence, open-ended outputs, black boxes, saturating benchmarks, and expanding scope.",1788871970211]