[{"data":1,"prerenderedAt":1201},["ShallowReactive",2],{"navigation_docs_en":3,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-4-ai-as-a-judge":114,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-4-ai-as-a-judge-surround":1196},[4],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":45},"AI Engineering",null,"\u002Fen\u002Fai-engineering","en\u002F1.ai-engineering",[10,46,77],{"title":11,"icon":12,"path":13,"stem":14,"children":15,"page":45},"Introduction to Building AI Applications with Foundation Models","i-lucide-brain-circuit","\u002Fen\u002Fai-engineering\u002Fintro","en\u002F1.ai-engineering\u002F1.intro",[16,20,25,30,35,40],{"title":11,"path":17,"stem":18,"icon":19},"\u002Fen\u002Fai-engineering\u002Fintro\u002Fch01","en\u002F1.ai-engineering\u002F1.intro\u002Fch01","i-lucide-sparkles",{"title":21,"path":22,"stem":23,"icon":24},"The Rise of AI Engineering","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch011-the-rise-of-ai-engineering","en\u002F1.ai-engineering\u002F1.intro\u002Fch011-the-rise-of-ai-engineering","i-lucide-history",{"title":26,"path":27,"stem":28,"icon":29},"Foundation Model Use Cases","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch012-foundation-model-use-cases","en\u002F1.ai-engineering\u002F1.intro\u002Fch012-foundation-model-use-cases","i-lucide-layout-grid",{"title":31,"path":32,"stem":33,"icon":34},"Planning AI Applications","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch013-planning-ai-applications","en\u002F1.ai-engineering\u002F1.intro\u002Fch013-planning-ai-applications","i-lucide-clipboard-list",{"title":36,"path":37,"stem":38,"icon":39},"The AI Engineering Stack","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch014-the-ai-engineering-stack","en\u002F1.ai-engineering\u002F1.intro\u002Fch014-the-ai-engineering-stack","i-lucide-layers",{"title":41,"path":42,"stem":43,"icon":44},"Summary","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch015-summary","en\u002F1.ai-engineering\u002F1.intro\u002Fch015-summary","i-lucide-flag",false,{"title":47,"icon":6,"path":48,"stem":49,"children":50,"page":45},"Understanding Foundation Models","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models","en\u002F1.ai-engineering\u002F2.understanding-foundation-models",[51,54,59,64,69,74],{"title":47,"path":52,"stem":53,"icon":12},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02",{"title":55,"path":56,"stem":57,"icon":58},"Training Data","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-1-training-data","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-1-training-data","i-lucide-database",{"title":60,"path":61,"stem":62,"icon":63},"Modeling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-2-modeling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-2-modeling","i-lucide-network",{"title":65,"path":66,"stem":67,"icon":68},"Post-Training","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-3-post-training","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-3-post-training","i-lucide-sliders-horizontal",{"title":70,"path":71,"stem":72,"icon":73},"Sampling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-4-sampling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-4-sampling","i-lucide-dices",{"title":41,"path":75,"stem":76,"icon":44},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-5-summary","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-5-summary",{"title":78,"path":79,"stem":80,"children":81,"page":45},"Evaluation Methodology","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology","en\u002F1.ai-engineering\u002F3.evaluation-methodology",[82,86,91,96,101,106,111],{"title":78,"path":83,"stem":84,"icon":85},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03","i-lucide-clipboard-check",{"title":87,"path":88,"stem":89,"icon":90},"Challenges of Evaluating Foundation Models","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","i-lucide-shield-alert",{"title":92,"path":93,"stem":94,"icon":95},"Understanding Language Modeling Metrics","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","i-lucide-sigma",{"title":97,"path":98,"stem":99,"icon":100},"Exact Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-3-exact-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-3-exact-evaluation","i-lucide-check-check",{"title":102,"path":103,"stem":104,"icon":105},"AI as a Judge","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-4-ai-as-a-judge","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-4-ai-as-a-judge","i-lucide-scale",{"title":107,"path":108,"stem":109,"icon":110},"Ranking Models with Comparative Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","i-lucide-trophy",{"title":41,"path":112,"stem":113,"icon":44},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-6-summary","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-6-summary",{"id":115,"title":102,"body":116,"description":1190,"extension":1191,"links":6,"meta":1192,"navigation":1193,"path":103,"seo":1194,"stem":104,"__hash__":1195},"docs_en\u002Fen\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-4-ai-as-a-judge.md",{"type":117,"value":118,"toc":1170},"minimark",[119,134,151,158,161,172,188,193,196,215,218,221,256,259,266,271,274,278,282,285,442,445,457,460,465,541,545,549,552,575,591,599,602,612,658,661,667,671,674,678,681,684,688,691,706,712,716,723,728,824,827,830,833,836,839,842,846,849,867,870,882,885,888,891,895,899,902,961,964,967,970,973,977,980,1004,1007,1010,1013,1025,1045,1053,1078,1084,1087,1137,1146,1149,1155,1160,1163,1166],[120,121,122,126],"u-page-hero",{},[123,124,102],"template",{"v-slot:title":125},"",[123,127,128,129,133],{"v-slot:description":125},"The challenges of evaluating open-ended responses have led many teams to fall back on ",[130,131,132],"strong",{},"human evaluation",". As AI has successfully been used to automate many challenging tasks, can AI automate evaluation as well?",[135,136,137,138,142,143,146,147,150],"p",{},"The approach of using AI to evaluate AI is called ",[139,140,141],"em",{},"AI as a judge"," or ",[139,144,145],{},"LLM as a judge",". An AI model that is used to evaluate other AI models is called an ",[139,148,149],{},"AI judge",".",[152,153,154,155,157],"note",{},"The term ",[139,156,149],{}," is not to be confused with the use case where AI is used as a judge in court.",[135,159,160],{},"While the idea of using AI to automate evaluation has been around for a long time, it only became practical when AI models became capable of doing so, which was around 2020 with the release of GPT-3.",[152,162,163,164,171],{},"In 2017, I presented at a NeurIPS workshop ",[165,166,170],"a",{"href":167,"rel":168},"https:\u002F\u002Fx.com\u002Fchipro\u002Fstatus\u002F937384141791698944",[169],"nofollow","MEWR"," (Machine translation Evaluation metric Without Reference text), an evaluation method that leverages stronger language models to automatically evaluate machine translations. Sadly, I never pursued this line of research because life got in the way.",[135,173,174,175,183,184,187],{},"As of this writing, AI as a judge has become one of the most, if not the most, common methods for evaluating AI models in production. Most demos of AI evaluation startups I saw in 2023 and 2024 leveraged AI as a judge in one way or another. ",[165,176,179,180],{"href":177,"rel":178},"https:\u002F\u002Fwww.langchain.com\u002Fblog\u002Flangchain-state-of-ai-2023",[169],"LangChain's ",[139,181,182],{},"State of AI"," report in 2023 noted that ",[130,185,186],{},"58%"," of evaluations on their platform were done by AI judges. AI as a judge is also an active area of research.",[189,190,192],"h2",{"id":191},"why-ai-as-a-judge","Why AI as a Judge?",[135,194,195],{},"AI judges are fast, easy to use, and relatively cheap compared to human evaluators. They can also work without reference data, which means they can be used in production environments where there is no reference data.",[197,198,199,205,210],"card-group",{},[200,201,204],"card",{"icon":202,"title":203},"i-lucide-zap","Fast, Easy, Relatively Cheap","Compared with human evaluators, AI judges are faster, easier to use, and relatively cheap.",[200,206,209],{"icon":207,"title":208},"i-lucide-database-off","No Reference Data Required","They can work without reference data, so they can be used in production environments where none exists.",[200,211,214],{"icon":212,"title":213},"i-lucide-list-checks","Any Criteria You Can Ask","You can ask AI models to judge an output based on any criteria: correctness, repetitiveness, toxicity, wholesomeness, hallucinations, and more.",[135,216,217],{},"This is similar to how you can ask a person to give their opinion about anything. You might think, \"But you can't always trust people's opinions.\" That's true, and you can't always trust AI's judgments, either. However, as each AI model is an aggregation of the masses, it's possible for AI models to make judgments representative of the masses. With the right prompt for the right model, you can get reasonably good judgments on a wide range of topics.",[135,219,220],{},"Studies have shown that certain AI judges are strongly correlated to human evaluators.",[197,222,223,242],{},[200,224,227,228,233,234,237,238,241],{"icon":225,"title":226},"i-lucide-users","GPT-4 and Humans","In 2023, ",[165,229,232],{"href":230,"rel":231},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2306.05685",[169],"Zheng et al."," found that on their evaluation benchmark, MT-Bench, the agreement between GPT-4 and humans reached ",[130,235,236],{},"85%",", which is even higher than the agreement among humans (",[130,239,240],{},"81%",").",[200,243,245,246,251,252,255],{"icon":110,"title":244},"AlpacaEval and Chat Arena","AlpacaEval authors (",[165,247,250],{"href":248,"rel":249},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2404.04475",[169],"Dubois et al., 2023",") also found that their AI judges have a near perfect (",[130,253,254],{},"0.98",") correlation with LMSYS's Chat Arena leaderboard, which is evaluated by humans.",[135,257,258],{},"Not only can AI evaluate a response, but it can also explain its decision, which can be especially useful when you want to audit your evaluation results. Figure 3-7 shows an example of GPT-4 explaining its judgment.",[135,260,261],{},[262,263],"img",{"alt":264,"src":265},"Figure 3-7. Not only can AI judges score, they also can explain their decisions.",".\u002Fmedia\u002Ffig-3-7.png",[267,268,269],"blockquote",{},[135,270,264],{},[135,272,273],{},"Its flexibility makes AI as a judge useful for a wide range of applications, and for some applications, it's the only automatic evaluation option.",[275,276,277],"tip",{},"Even when AI judgments aren't as good as human judgments, they might still be good enough to guide an application's development and provide sufficient confidence to get a project off the ground.",[189,279,281],{"id":280},"how-to-use-ai-as-a-judge","How to Use AI as a Judge",[135,283,284],{},"There are many ways you can use AI to make judgments. For example, you can use AI to evaluate the quality of a response by itself, compare that response to reference data, or compare that response to another response. Here are naive example prompts for these three approaches:",[286,287,289,294,297,361,365,368,401,405,408],"steps",{"level":288},"3",[290,291,293],"h3",{"id":292},"evaluate-a-response-by-itself","Evaluate a Response by Itself",[135,295,296],{},"Evaluate the quality of a response by itself, given the original question:",[298,299,303],"pre",{"className":300,"code":301,"language":302,"meta":125,"style":125},"language-prompt shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","\"Given the following question and answer, evaluate how good the answer is for the question. Use the score from 1 to 5.\n\n- 1 means very bad.\n- 5 means very good.\n\nQuestion: [QUESTION]\nAnswer: [ANSWER]\n\nScore:\"\n","prompt",[304,305,306,314,321,327,333,338,344,350,355],"code",{"__ignoreMap":125},[307,308,311],"span",{"class":309,"line":310},"line",1,[307,312,313],{},"\"Given the following question and answer, evaluate how good the answer is for the question. Use the score from 1 to 5.\n",[307,315,317],{"class":309,"line":316},2,[307,318,320],{"emptyLinePlaceholder":319},true,"\n",[307,322,324],{"class":309,"line":323},3,[307,325,326],{},"- 1 means very bad.\n",[307,328,330],{"class":309,"line":329},4,[307,331,332],{},"- 5 means very good.\n",[307,334,336],{"class":309,"line":335},5,[307,337,320],{"emptyLinePlaceholder":319},[307,339,341],{"class":309,"line":340},6,[307,342,343],{},"Question: [QUESTION]\n",[307,345,347],{"class":309,"line":346},7,[307,348,349],{},"Answer: [ANSWER]\n",[307,351,353],{"class":309,"line":352},8,[307,354,320],{"emptyLinePlaceholder":319},[307,356,358],{"class":309,"line":357},9,[307,359,360],{},"Score:\"\n",[290,362,364],{"id":363},"compare-to-a-reference-response","Compare to a Reference Response",[135,366,367],{},"Compare a generated response to a reference response to evaluate whether the generated response is the same as the reference response. This can be an alternative approach to human-designed similarity measurements:",[298,369,371],{"className":300,"code":370,"language":302,"meta":125,"style":125},"\"Given the following question, reference answer, and generated answer, evaluate whether this generated answer is the same as the reference answer.\nOutput True or False.\n\nQuestion: [QUESTION]\nReference answer: [REFERENCE ANSWER]\nGenerated answer: [GENERATED ANSWER]\"\n",[304,372,373,378,383,387,391,396],{"__ignoreMap":125},[307,374,375],{"class":309,"line":310},[307,376,377],{},"\"Given the following question, reference answer, and generated answer, evaluate whether this generated answer is the same as the reference answer.\n",[307,379,380],{"class":309,"line":316},[307,381,382],{},"Output True or False.\n",[307,384,385],{"class":309,"line":323},[307,386,320],{"emptyLinePlaceholder":319},[307,388,389],{"class":309,"line":329},[307,390,343],{},[307,392,393],{"class":309,"line":335},[307,394,395],{},"Reference answer: [REFERENCE ANSWER]\n",[307,397,398],{"class":309,"line":340},[307,399,400],{},"Generated answer: [GENERATED ANSWER]\"\n",[290,402,404],{"id":403},"compare-two-generated-responses","Compare Two Generated Responses",[135,406,407],{},"Compare two generated responses and determine which one is better or predict which one users will likely prefer. This is helpful for generating preference data for post-training alignment (discussed in Chapter 2), test-time compute (discussed in Chapter 2), and ranking models using comparative evaluation (discussed in the next section):",[298,409,411],{"className":300,"code":410,"language":302,"meta":125,"style":125},"\"Given the following question and two answers, evaluate which answer is better.\nOutput A or B.\nQuestion: [QUESTION]\nA: [FIRST ANSWER]\nB: [SECOND ANSWER]\nThe better answer is:\"\n",[304,412,413,418,423,427,432,437],{"__ignoreMap":125},[307,414,415],{"class":309,"line":310},[307,416,417],{},"\"Given the following question and two answers, evaluate which answer is better.\n",[307,419,420],{"class":309,"line":316},[307,421,422],{},"Output A or B.\n",[307,424,425],{"class":309,"line":323},[307,426,343],{},[307,428,429],{"class":309,"line":329},[307,430,431],{},"A: [FIRST ANSWER]\n",[307,433,434],{"class":309,"line":335},[307,435,436],{},"B: [SECOND ANSWER]\n",[307,438,439],{"class":309,"line":340},[307,440,441],{},"The better answer is:\"\n",[135,443,444],{},"A general-purpose AI judge can be asked to evaluate a response based on any criteria.",[197,446,447,452],{},[200,448,451],{"icon":449,"title":450},"i-lucide-drama","Roleplaying Chatbot","If you're building a roleplaying chatbot, you might want to evaluate if a chatbot's response is consistent with the role users want it to play, such as \"Does this response sound like something Gandalf would say?\"",[200,453,456],{"icon":454,"title":455},"i-lucide-image","Promotional Product Photos","If you're building an application to generate promotional product photos, you might want to ask \"From 1 to 5, how would you rate the trustworthiness of the product in this image?\"",[135,458,459],{},"Table 3-3 shows common built-in AI as a judge criteria offered by some AI tools.",[267,461,462],{},[135,463,464],{},"Table 3-3. Examples of built-in AI as a judge criteria offered by some AI tools, as of September 2024. Note that as these tools evolve, these built-in criteria will change.",[466,467,468,481],"table",{},[469,470,471],"thead",{},[472,473,474,478],"tr",{},[475,476,477],"th",{},"AI Tools",[475,479,480],{},"Built-in criteria",[482,483,484,499,513,527],"tbody",{},[472,485,486,496],{},[487,488,489],"td",{},[165,490,493],{"href":491,"rel":492},"https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Ffoundry\u002Fconcepts\u002Fbuilt-in-evaluators",[169],[130,494,495],{},"Azure AI Studio",[487,497,498],{},"Groundedness, relevance, coherence, fluency, similarity",[472,500,501,510],{},[487,502,503],{},[165,504,507],{"href":505,"rel":506},"https:\u002F\u002Fmlflow.org\u002Fdocs\u002Flatest\u002Fapi_reference\u002Fpython_api\u002Fmlflow.metrics.html#generative-ai-metrics",[169],[130,508,509],{},"MLflow.metrics",[487,511,512],{},"Faithfulness, relevance",[472,514,515,524],{},[487,516,517],{},[165,518,521],{"href":519,"rel":520},"https:\u002F\u002Fdocs.langchain.com\u002Foss\u002Fpython\u002Flangchain\u002Foverview",[169],[130,522,523],{},"LangChain Criteria Evaluation",[487,525,526],{},"Conciseness, relevance, correctness, coherence, harmfulness, maliciousness, helpfulness, controversiality, misogyny, insensitivity, criminality",[472,528,529,538],{},[487,530,531],{},[165,532,535],{"href":533,"rel":534},"https:\u002F\u002Fdocs.ragas.io\u002Fen\u002Flatest\u002Fconcepts\u002Fmetrics\u002Findex.html",[169],[130,536,537],{},"Ragas",[487,539,540],{},"Faithfulness, answer relevance",[542,543,544],"warning",{},"It's essential to remember that AI as a judge criteria aren't standardized. Azure AI Studio's relevance scores might be very different from MLflow's relevance scores. These scores depend on the judge's underlying model and prompt.",[189,546,548],{"id":547},"how-to-prompt-an-ai-judge","How to Prompt an AI Judge",[135,550,551],{},"How to prompt an AI judge is similar to how to prompt any AI application. In general, a judge's prompt should clearly explain the following:",[286,553,554,558,561,565,568,572],{"level":288},[290,555,557],{"id":556},"the-task","The Task",[135,559,560],{},"The task the model is to perform, such as to evaluate the relevance between a generated answer and the question.",[290,562,564],{"id":563},"the-criteria","The Criteria",[135,566,567],{},"The criteria the model should follow to evaluate, such as \"Your primary focus should be on determining whether the generated answer contains sufficient information to address the given question according to the ground truth answer\". The more detailed the instruction, the better.",[290,569,571],{"id":570},"the-scoring-system","The Scoring System",[135,573,574],{},"The scoring system, which can be one of these:",[197,576,577,582,587],{},[200,578,581],{"icon":579,"title":580},"i-lucide-tags","Classification","Such as good\u002Fbad or relevant\u002Firrelevant\u002Fneutral.",[200,583,586],{"icon":584,"title":585},"i-lucide-hash","Discrete Numerical Values","Such as 1 to 5. Discrete numerical values can be considered a special case of classification, where each class has a numerical interpretation instead of a semantic interpretation.",[200,588,590],{"icon":68,"title":589},"Continuous Numerical Values","Such as between 0 and 1, e.g., when you want to evaluate the degree of similarity.",[152,592,593,596],{},[135,594,595],{},"Language models are generally better with text than with numbers. It's been reported that AI judges work better with classification than with numerical scoring systems.",[135,597,598],{},"For numerical scoring systems, discrete scoring seems to work better than continuous scoring. Empirically, the wider the range for discrete scoring, the worse the model seems to get. Typical discrete scoring systems are between 1 and 5.",[135,600,601],{},"Prompts with examples have been shown to perform better. If you use a scoring system between 1 and 5, include examples of what a response with a score of 1, 2, 3, 4, or 5 looks like, and if possible, why a response receives a certain score. Best practices for prompting are discussed in Chapter 5.",[135,603,604,605,611],{},"Here's part of the prompt used for the criteria ",[165,606,608],{"href":491,"rel":607},[169],[139,609,610],{},"relevance"," by Azure AI Studio. It explains the task, the criteria, the scoring system, an example of an input with a low score, and a justification for why this input has a low score. Part of the prompt was removed for brevity.",[298,613,615],{"className":300,"code":614,"language":302,"meta":125,"style":125},"Your task is to score the relevance between a generated answer and the question based on the ground truth answer in the range between 1 and 5, and please also provide the scoring reason.\n\nYour primary focus should be on determining whether the generated answer contains sufficient information to address the given question according to the ground truth answer. …\n\nIf the generated answer contradicts the ground truth answer, it will receive a low score of 1-2.\n\nFor example, for the question \"Is the sky blue?\" the ground truth answer is \"Yes, the sky is blue.\" and the generated answer is \"No, the sky is not blue.\"\n\nIn this example, the generated answer contradicts the ground truth answer by stating that the sky is not blue, when in fact it is blue. This inconsistency would result in a low score of 1–2, and the reason for the low score would reflect the contradiction between the generated answer and the ground truth answer.\n",[304,616,617,622,626,631,635,640,644,649,653],{"__ignoreMap":125},[307,618,619],{"class":309,"line":310},[307,620,621],{},"Your task is to score the relevance between a generated answer and the question based on the ground truth answer in the range between 1 and 5, and please also provide the scoring reason.\n",[307,623,624],{"class":309,"line":316},[307,625,320],{"emptyLinePlaceholder":319},[307,627,628],{"class":309,"line":323},[307,629,630],{},"Your primary focus should be on determining whether the generated answer contains sufficient information to address the given question according to the ground truth answer. …\n",[307,632,633],{"class":309,"line":329},[307,634,320],{"emptyLinePlaceholder":319},[307,636,637],{"class":309,"line":335},[307,638,639],{},"If the generated answer contradicts the ground truth answer, it will receive a low score of 1-2.\n",[307,641,642],{"class":309,"line":340},[307,643,320],{"emptyLinePlaceholder":319},[307,645,646],{"class":309,"line":346},[307,647,648],{},"For example, for the question \"Is the sky blue?\" the ground truth answer is \"Yes, the sky is blue.\" and the generated answer is \"No, the sky is not blue.\"\n",[307,650,651],{"class":309,"line":352},[307,652,320],{"emptyLinePlaceholder":319},[307,654,655],{"class":309,"line":357},[307,656,657],{},"In this example, the generated answer contradicts the ground truth answer by stating that the sky is not blue, when in fact it is blue. This inconsistency would result in a low score of 1–2, and the reason for the low score would reflect the contradiction between the generated answer and the ground truth answer.\n",[135,659,660],{},"Figure 3-8 shows an example of an AI judge that evaluates the quality of an answer given a question.",[135,662,663],{},[262,664],{"alt":665,"src":666},"Figure 3-8. An example of an AI judge that evaluates the quality of an answer given a question.",".\u002Fmedia\u002Ffig-3-8.png",[267,668,669],{},[135,670,665],{},[275,672,673],{},"An AI judge is not just a model — it's a system that includes both a model and a prompt. Altering the model, the prompt, or the model's sampling parameters results in a different judge.",[189,675,677],{"id":676},"limitations-of-ai-as-a-judge","Limitations of AI as a Judge",[135,679,680],{},"Despite the many advantages of AI as a judge, many teams are hesitant to adopt this approach. Using AI to evaluate AI seems tautological. The probabilistic nature of AI makes it seem too unreliable to act as an evaluator. AI judges can potentially introduce nontrivial costs and latency to an application.",[542,682,683],{},"Given these limitations, some teams see AI as a judge as a fallback option when they don't have any other way of evaluating their systems, especially in production.",[290,685,687],{"id":686},"inconsistency","Inconsistency",[135,689,690],{},"For an evaluation method to be trustworthy, its results should be consistent. Yet AI judges, like all AI applications, are probabilistic. The same judge, on the same input, can output different scores if prompted differently. Even the same judge, prompted with the same instruction, can output different scores if run twice. This inconsistency makes it hard to reproduce or trust evaluation results.",[135,692,693,694,698,699,702,703,150],{},"It's possible to get an AI judge to be more consistent. Chapter 2 discusses how to do so with sampling variables. ",[165,695,697],{"href":230,"rel":696},[169],"Zheng et al. (2023)"," showed that including evaluation examples in the prompt can increase the consistency of GPT-4 from ",[130,700,701],{},"65%"," to ",[130,704,705],{},"77.5%",[542,707,708,709,150],{},"They acknowledged that high consistency may not imply high accuracy — the judge might consistently make the same mistakes. On top of that, including more examples makes prompts longer, and longer prompts mean higher inference costs. In Zheng et al.'s experiment, including more examples in their prompts caused their GPT-4 spending to ",[130,710,711],{},"quadruple",[290,713,715],{"id":714},"criteria-ambiguity","Criteria Ambiguity",[135,717,718,719,722],{},"Unlike many human-designed metrics, AI as a judge metrics aren't standardized, making it easy to misinterpret and misuse them. As of this writing, the open source tools MLflow, Ragas, and LlamaIndex all have the built-in criterion ",[139,720,721],{},"faithfulness"," to measure how faithful a generated output is to the given context, but their instructions and scoring systems are all different. As shown in Table 3-4, MLflow uses a scoring system from 1 to 5, Ragas uses 0 and 1, whereas LlamaIndex's prompt asks the judge to output YES and NO.",[267,724,725],{},[135,726,727],{},"Table 3-4. Different tools can have very difficult default prompts for the same criteria.",[466,729,730,743],{},[469,731,732],{},[472,733,734,737,740],{},[475,735,736],{},"Tool",[475,738,739],{},"Prompt (partially omitted for brevity)",[475,741,742],{},"Scoring system",[482,744,745,774,790],{},[472,746,747,756,771],{},[487,748,749],{},[165,750,753],{"href":751,"rel":752},"https:\u002F\u002Fgithub.com\u002Fmlflow\u002Fmlflow\u002Fblob\u002F5cdae7c4321015620032d02a3b84fb6127247392\u002Fmlflow\u002Fmetrics\u002Fgenai\u002Fprompts\u002Fv1.py",[169],[130,754,755],{},"MLflow",[487,757,758,759,762,764,765,767,768,770],{},"Faithfulness is only evaluated with the provided output and provided context, please ignore the provided input entirely when scoring faithfulness. Faithfulness assesses how much of the provided output is factually consistent with the provided context....",[760,761],"br",{},[760,763],{},"Faithfulness: Below are the details for different scores:",[760,766],{},"- Score 1: None of the claims in the output can be inferred from the provided context.",[760,769],{},"- Score 2: ...",[487,772,773],{},"1–5",[472,775,776,784,787],{},[487,777,778],{},[165,779,782],{"href":780,"rel":781},"https:\u002F\u002Fgithub.com\u002Fvibrantlabsai\u002Fragas\u002Fblob\u002Fb276f59c0d4eb4795dc28966bfbce14d5aacd140\u002Fsrc\u002Fragas\u002Fmetrics\u002F_faithfulness.py#L93C1-L94C1",[169],[130,783,537],{},[487,785,786],{},"Your task is to judge the faithfulness of a series of statements based on a given context. For each statement you must return verdict as 1 if the statement can be verified based on the context or 0 if the statement can not be verified based on the context.",[487,788,789],{},"0 and 1",[472,791,792,801,821],{},[487,793,794],{},[165,795,798],{"href":796,"rel":797},"https:\u002F\u002Fgithub.com\u002Frun-llama\u002Fllama_index\u002Fblob\u002Fmain\u002Fllama-index-core\u002Fllama_index\u002Fcore\u002Fevaluation\u002Ffaithfulness.py",[169],[130,799,800],{},"LlamaIndex",[487,802,803,804,806,807,809,810,812,814,815,817,818,820],{},"Please tell if a given piece of information is supported by the context.",[760,805],{},"You need to answer with either YES or NO.",[760,808],{},"Answer YES if any of the context supports the information, even if most of the context is unrelated. Some examples are provided below.",[760,811],{},[760,813],{},"Information: Apple pie is generally double-crusted.",[760,816],{},"Context: An apple pie is a fruit pie... It is generally double-crusted, with pastry both above and below the filling ...",[760,819],{},"Answer: YES",[487,822,823],{},"YES and NO",[135,825,826],{},"The faithfulness scores outputted by these three tools won't be comparable. If, given a (context, answer) pair, MLflow gives a faithfulness score of 3, Ragas outputs 1, and LlamaIndex outputs NO, which score would you use?",[135,828,829],{},"An application evolves over time, but the way it's evaluated ideally should be fixed. This way, evaluation metrics can be used to monitor the application's changes. However, AI judges are also AI applications, which means that they also can change over time.",[135,831,832],{},"Imagine that last month, your application's coherence score was 90%, and this month, this score is 92%. Does this mean that your application's coherence has improved? It's hard to answer this question unless you know for sure that the AI judges used in both cases are exactly the same. What if the judge's prompt this month is different from the one last month? Maybe you switched to a slightly better-performing prompt or a coworker fixed a typo in last month's prompt, and the judge this month is more lenient.",[135,834,835],{},"This can become especially confusing if the application and the AI judge are managed by different teams. The AI judge team might change the judges without informing the application team. As a result, the application team might mistakenly attribute the changes in the evaluation results to changes in the application, rather than the changes in the judges.",[152,837,838],{},"Do not trust any AI judge if you can't see the model and the prompt used for the judge.",[135,840,841],{},"Evaluation methods take time to standardize. As the field evolves and more guardrails are introduced, I hope that future AI judges will become a lot more standardized and reliable.",[290,843,845],{"id":844},"increased-costs-and-latency","Increased Costs and Latency",[135,847,848],{},"You can use AI judges to evaluate applications both during experimentation and in production. Many teams use AI judges as guardrails in production to reduce risks, showing users only generated responses deemed good by the AI judge. Using powerful models to evaluate responses can be expensive.",[197,850,851,860],{},[200,852,855,856,859],{"icon":853,"title":854},"i-lucide-copy","Generate and Evaluate with GPT-4","If you use GPT-4 to both generate and evaluate responses, you'll do twice as many GPT-4 calls, approximately ",[130,857,858],{},"doubling"," your API costs.",[200,861,863,864,150],{"icon":39,"title":862},"Three Evaluation Prompts","If you have three evaluation prompts because you want to evaluate three criteria — say, overall response quality, factual consistency, and toxicity — you'll increase your number of API calls ",[130,865,866],{},"four times",[152,868,869],{},"In some cases, evaluation can take up the majority of the budget, even more than response generation.",[135,871,872,873,877,878,881],{},"You can reduce costs by using weaker models as the judges (see ",[165,874,876],{"href":875},"#what-models-can-act-as-judges","\"What Models Can Act as Judges?\"","). You can also reduce costs with ",[139,879,880],{},"spot-checking",": evaluating only a subset of responses.",[152,883,884],{},"Spot-checking is the same as sampling.",[135,886,887],{},"Spot-checking means you might fail to catch some failures. The larger the percentage of samples you evaluate, the more confidence you will have in your evaluation results, but also the higher the costs. Finding the right balance between cost and confidence might take trial and error. This process is discussed further in Chapter 4. All things considered, AI judges are much cheaper than human evaluators.",[135,889,890],{},"Implementing AI judges in your production pipeline can add latency. If you evaluate responses before returning them to users, you face a trade-off: reduced risk but increased latency.",[892,893,894],"caution",{},"The added latency might make this option a nonstarter for applications with strict latency requirements.",[290,896,898],{"id":897},"biases-of-ai-as-a-judge","Biases of AI as a Judge",[135,900,901],{},"Human evaluators have biases, and so do AI judges. Different AI judges have different biases. This section will discuss some of the common ones. Being aware of your AI judges' biases helps you interpret their scores correctly and even mitigate these biases.",[197,903,904,926,940],{},[200,905,908,909,912,913,917,918,921,922,925],{"icon":906,"title":907},"i-lucide-heart","Self-Bias","AI judges tend to have ",[139,910,911],{},"self-bias",", where a model favors its own responses over the responses generated by other models. The same mechanism that helps a model compute the most likely response to generate will also give this response a high score. In ",[165,914,916],{"href":230,"rel":915},[169],"Zheng et al.'s 2023 experiment",", GPT-4 favors itself with a ",[130,919,920],{},"10%"," higher win rate, while Claude-v1 favors itself with a ",[130,923,924],{},"25%"," higher win rate.",[200,927,930,931,936,937,150],{"icon":928,"title":929},"i-lucide-list-ordered","First-Position Bias","Many AI models have first-position bias. An AI judge may favor the first answer in a pairwise comparison or the first in a list of options. This can be mitigated by repeating the same test multiple times with different orderings or with carefully crafted prompts. The position bias of AI is the opposite of that of humans. Humans tend to favor ",[165,932,935],{"href":933,"rel":934},"https:\u002F\u002Fwww.interconnects.ai\u002Fp\u002Fevaluating-open-llms",[169],"the answer they see last",", which is called ",[139,938,939],{},"recency bias",[200,941,944,945,948,949,954,955,960],{"icon":942,"title":943},"i-lucide-align-left","Verbosity Bias","Some AI judges have ",[139,946,947],{},"verbosity bias",", favoring lengthier answers, regardless of their quality. ",[165,950,953],{"href":951,"rel":952},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2307.03025",[169],"Wu and Aji (2023)"," found that both GPT-4 and Claude-1 prefer longer responses (~100 words) with factual errors over shorter, correct responses (~50 words). ",[165,956,959],{"href":957,"rel":958},"https:\u002F\u002Farxiv.org\u002Fpdf\u002F2310.10076",[169],"Saito et al. (2023)"," studied this bias for creative tasks and found that when the length difference is large enough (e.g., one response is twice as long as the other), the judge almost always prefers the longer one. Both Zheng et al. (2023) and Saito et al. (2023), however, discovered that GPT-4 is less prone to this bias than GPT-3.5, suggesting that this bias might go away as models become stronger.",[152,962,963],{},"Saito et al. (2023) found that humans tend to favor longer responses too, but to a much lesser extent.",[135,965,966],{},"On top of all these biases, AI judges have the same limitations as all AI applications, including privacy and IP. If you use a proprietary model as your judge, you'd need to send your data to this model. If the model provider doesn't disclose their training data, you won't know for sure if the judge is commercially safe to use.",[135,968,969],{},"Despite the limitations of the AI as a judge approach, its many advantages make me believe that its adoption will continue to grow.",[542,971,972],{},"However, AI judges should be supplemented with exact evaluation methods and\u002For human evaluation.",[189,974,976],{"id":975},"what-models-can-act-as-judges","What Models Can Act as Judges?",[135,978,979],{},"The judge can either be stronger, weaker, or the same as the model being judged. Each scenario has its pros and cons.",[197,981,982,987,992],{},[200,983,986],{"icon":984,"title":985},"i-lucide-arrow-up","Stronger Judge","At first glance, a stronger judge makes sense. Shouldn't the exam grader be more knowledgeable than the exam taker? Not only can stronger models make better judgments, but they can also help improve weaker models by guiding them to generate better responses.",[200,988,991],{"icon":989,"title":990},"i-lucide-arrow-down","Weaker Judge","One open question is whether the judge can be weaker than the model being judged. Some argue that judging is an easier task than generating. Anyone can have an opinion about whether a song is good, but not everyone can write a song. Weaker models should be able to judge the outputs of stronger models.",[200,993,996,997,142,1000,1003],{"icon":994,"title":995},"i-lucide-refresh-cw","Same Model (Self-Evaluation)","Using a model to judge itself, ",[139,998,999],{},"self-evaluation",[139,1001,1002],{},"self-critique",", sounds like cheating, especially because of self-bias. However, self-evaluation can be great for sanity checks. If a model thinks its own response is incorrect, the model might not be that reliable.",[135,1005,1006],{},"You might wonder: if you already have access to the stronger model, why bother using a weaker model to generate responses? The answer is cost and latency. You might not have the budget to use the stronger model to generate all responses, so you use it to evaluate a subset of responses. For example, you may use a cheap in-house model to generate responses and GPT-4 to evaluate 1% of the responses.",[135,1008,1009],{},"The stronger model also might be too slow for your application. You can use a fast model to generate responses while the stronger, but slower, model does evaluation in the background. If the strong model thinks that the weak model's response is bad, remedy actions might be taken, such as updating the response with that of the strong model. Note that the opposite pattern is also common. You use a strong model to generate responses, with a weak model running in the background to do evaluation.",[135,1011,1012],{},"Using the stronger model as a judge leaves us with two challenges.",[197,1014,1015,1020],{},[200,1016,1019],{"icon":1017,"title":1018},"i-lucide-crown","No Eligible Judge for the Strongest","The strongest model will be left with no eligible judge.",[200,1021,1024],{"icon":1022,"title":1023},"i-lucide-circle-help","Who Is Strongest?","We need an alternative evaluation method to determine which model is the strongest.",[135,1026,1027,1028,1033,1034,1033,1039,1044],{},"Beyond sanity checks, asking a model to evaluate itself can nudge a model to revise and improve its responses (",[165,1029,1032],{"href":1030,"rel":1031},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2210.03350",[169],"Press et al., 2022","; ",[165,1035,1038],{"href":1036,"rel":1037},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.11738",[169],"Gou et al., 2023",[165,1040,1043],{"href":1041,"rel":1042},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.08118",[169],"Valmeekam et al., 2023","). This example shows what self-evaluation might look like:",[152,1046,1047,1048,142,1050,150],{},"This technique is sometimes referred to as ",[139,1049,1002],{},[139,1051,1052],{},"self-ask",[298,1054,1056],{"className":300,"code":1055,"language":302,"meta":125,"style":125},"Prompt [from user]: What's 10+3?\nFirst response [from AI]: 30\nSelf-critique [from AI]: Is this answer correct?\nFinal response [from AI]: No it's not. The correct answer is 13.\n",[304,1057,1058,1063,1068,1073],{"__ignoreMap":125},[307,1059,1060],{"class":309,"line":310},[307,1061,1062],{},"Prompt [from user]: What's 10+3?\n",[307,1064,1065],{"class":309,"line":316},[307,1066,1067],{},"First response [from AI]: 30\n",[307,1069,1070],{"class":309,"line":323},[307,1071,1072],{},"Self-critique [from AI]: Is this answer correct?\n",[307,1074,1075],{"class":309,"line":329},[307,1076,1077],{},"Final response [from AI]: No it's not. The correct answer is 13.\n",[135,1079,1080,1083],{},[165,1081,697],{"href":230,"rel":1082},[169]," found that stronger models are better correlated to human preference, which makes people opt for the strongest models they can afford. However, this experiment was limited to general-purpose judges. One research direction that I'm excited about is small, specialized judges. Specialized judges are trained to make specific judgments, using specific criteria and following specific scoring systems. A small, specialized judge can be more reliable than larger, general-purpose judges for specific judgments.",[135,1085,1086],{},"Because there are many possible ways to use AI judges, there are many possible specialized AI judges. Here, I'll go over examples of three specialized judges: reward models, reference-based judges, and preference models.",[197,1088,1089,1104,1121],{},[200,1090,1093,1094,1099,1100,1103],{"icon":1091,"title":1092},"i-lucide-gift","Reward Model","A reward model takes in a (prompt, response) pair and scores how good the response is given the prompt. Reward models have been successfully used in RLHF for many years. ",[165,1095,1098],{"href":1096,"rel":1097},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2311.06720",[169],"Cappy"," is an example of a reward model developed by Google (2023). Given a pair of (prompt, response), Cappy produces a score between 0 and 1, indicating how correct the response is. Cappy is a lightweight scorer with ",[130,1101,1102],{},"360 million"," parameters, much smaller than general-purpose foundation models.",[200,1105,1108,1109,1114,1115,1120],{"icon":1106,"title":1107},"i-lucide-file-search","Reference-Based Judge","A reference-based judge evaluates the generated response with respect to one or more reference responses. This judge can output a similarity score or a quality score (how good the generated response is compared to the reference responses). For example, BLEURT (",[165,1110,1113],{"href":1111,"rel":1112},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2004.04696",[169],"Sellam et al., 2020",") takes in a (candidate response, reference response) pair and outputs a similarity score between the candidate and reference response. Prometheus (",[165,1116,1119],{"href":1117,"rel":1118},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.08491",[169],"Kim et al., 2023",") takes in (prompt, generated response, reference response, scoring rubric) and outputs a quality score between 1 and 5, assuming that the reference response gets a 5.",[200,1122,1125,1126,1131,1132,241],{"icon":1123,"title":1124},"i-lucide-git-compare","Preference Model","A preference model takes in (prompt, response 1, response 2) as input and outputs which of the two responses is better (preferred by users) for the given prompt. This is perhaps one of the more exciting directions for specialized judges. Being able to predict human preference opens up many possibilities. As discussed in Chapter 2, preference data is essential for aligning AI models to human preference, and it's challenging and expensive to obtain. Having a good human preference predictor can generally make evaluation easier and models safer to use. There have been many initiatives in building preference models, including PandaLM (",[165,1127,1130],{"href":1128,"rel":1129},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2306.05087",[169],"Wang et al., 2023",") and JudgeLM (",[165,1133,1136],{"href":1134,"rel":1135},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2310.17631",[169],"Zhu et al., 2023",[152,1138,1139,1140,1145],{},"The BLEURT score range is confusing. It's approximately ",[165,1141,1144],{"href":1142,"rel":1143},"https:\u002F\u002Fgithub.com\u002Fgoogle-research\u002Fbleurt\u002Fissues\u002F1",[169],"between -2.5 and 1.0",". This highlights the challenge of criteria ambiguity with AI judges: the score range can be arbitrary.",[135,1147,1148],{},"Figure 3-9 shows an example of how PandaLM works. It not only outputs which response is better but also explains its rationale.",[135,1150,1151],{},[262,1152],{"alt":1153,"src":1154},"Figure 3-9. An example output of PandaLM, given a human prompt and two generated responses. Picture from Wang et al. (2023), modified slightly for readability.",".\u002Fmedia\u002Ffig-3-9.png",[267,1156,1157],{},[135,1158,1159],{},"Figure 3-9. An example output of PandaLM, given a human prompt and two generated responses. Picture from Wang et al. (2023), modified slightly for readability. The original image is available under the Apache License 2.0.",[135,1161,1162],{},"Despite its limitations, the AI as a judge approach is versatile and powerful. Using cheaper models as judges makes it even more useful. Many of my colleagues, who were initially skeptical, have started to rely on it more in production.",[135,1164,1165],{},"AI as a judge is exciting, and the next approach we'll discuss is just as intriguing. It's inspired by game design, a fascinating field.",[1167,1168,1169],"style",{},"html .light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html.light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"title":125,"searchDepth":316,"depth":316,"links":1171},[1172,1173,1178,1183,1189],{"id":191,"depth":316,"text":192},{"id":280,"depth":316,"text":281,"children":1174},[1175,1176,1177],{"id":292,"depth":323,"text":293},{"id":363,"depth":323,"text":364},{"id":403,"depth":323,"text":404},{"id":547,"depth":316,"text":548,"children":1179},[1180,1181,1182],{"id":556,"depth":323,"text":557},{"id":563,"depth":323,"text":564},{"id":570,"depth":323,"text":571},{"id":676,"depth":316,"text":677,"children":1184},[1185,1186,1187,1188],{"id":686,"depth":323,"text":687},{"id":714,"depth":323,"text":715},{"id":844,"depth":323,"text":845},{"id":897,"depth":323,"text":898},{"id":975,"depth":316,"text":976},"Why AI judges took off, how to prompt them, their limits (inconsistency, cost, bias), and which models can judge.","md",{},{"icon":105},{"title":102,"description":1190},"kwm8eBsipqA4yRIiew7Tn_WxfmXvap2NhKsVDMopiCg",[1197,1199],{"title":97,"path":98,"stem":99,"description":1198,"icon":100,"children":-1},"How functional correctness, similarity against reference data, and embeddings produce exact scores for open-ended model outputs.",{"title":107,"path":108,"stem":109,"description":1200,"icon":110,"children":-1},"Rank models with pointwise scores or comparative votes. How Chatbot Arena works, and the scalability, quality, and absolute-performance limits of ranking.",1788871976056]