[{"data":1,"prerenderedAt":4883},["ShallowReactive",2],{"navigation_docs_en":3,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics":114,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics-surround":4878},[4],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":45},"AI Engineering",null,"\u002Fen\u002Fai-engineering","en\u002F1.ai-engineering",[10,46,77],{"title":11,"icon":12,"path":13,"stem":14,"children":15,"page":45},"Introduction to Building AI Applications with Foundation Models","i-lucide-brain-circuit","\u002Fen\u002Fai-engineering\u002Fintro","en\u002F1.ai-engineering\u002F1.intro",[16,20,25,30,35,40],{"title":11,"path":17,"stem":18,"icon":19},"\u002Fen\u002Fai-engineering\u002Fintro\u002Fch01","en\u002F1.ai-engineering\u002F1.intro\u002Fch01","i-lucide-sparkles",{"title":21,"path":22,"stem":23,"icon":24},"The Rise of AI Engineering","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch011-the-rise-of-ai-engineering","en\u002F1.ai-engineering\u002F1.intro\u002Fch011-the-rise-of-ai-engineering","i-lucide-history",{"title":26,"path":27,"stem":28,"icon":29},"Foundation Model Use Cases","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch012-foundation-model-use-cases","en\u002F1.ai-engineering\u002F1.intro\u002Fch012-foundation-model-use-cases","i-lucide-layout-grid",{"title":31,"path":32,"stem":33,"icon":34},"Planning AI Applications","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch013-planning-ai-applications","en\u002F1.ai-engineering\u002F1.intro\u002Fch013-planning-ai-applications","i-lucide-clipboard-list",{"title":36,"path":37,"stem":38,"icon":39},"The AI Engineering Stack","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch014-the-ai-engineering-stack","en\u002F1.ai-engineering\u002F1.intro\u002Fch014-the-ai-engineering-stack","i-lucide-layers",{"title":41,"path":42,"stem":43,"icon":44},"Summary","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch015-summary","en\u002F1.ai-engineering\u002F1.intro\u002Fch015-summary","i-lucide-flag",false,{"title":47,"icon":6,"path":48,"stem":49,"children":50,"page":45},"Understanding Foundation Models","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models","en\u002F1.ai-engineering\u002F2.understanding-foundation-models",[51,54,59,64,69,74],{"title":47,"path":52,"stem":53,"icon":12},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02",{"title":55,"path":56,"stem":57,"icon":58},"Training Data","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-1-training-data","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-1-training-data","i-lucide-database",{"title":60,"path":61,"stem":62,"icon":63},"Modeling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-2-modeling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-2-modeling","i-lucide-network",{"title":65,"path":66,"stem":67,"icon":68},"Post-Training","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-3-post-training","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-3-post-training","i-lucide-sliders-horizontal",{"title":70,"path":71,"stem":72,"icon":73},"Sampling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-4-sampling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-4-sampling","i-lucide-dices",{"title":41,"path":75,"stem":76,"icon":44},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-5-summary","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-5-summary",{"title":78,"path":79,"stem":80,"children":81,"page":45},"Evaluation Methodology","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology","en\u002F1.ai-engineering\u002F3.evaluation-methodology",[82,86,91,96,101,106,111],{"title":78,"path":83,"stem":84,"icon":85},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03","i-lucide-clipboard-check",{"title":87,"path":88,"stem":89,"icon":90},"Challenges of Evaluating Foundation Models","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","i-lucide-shield-alert",{"title":92,"path":93,"stem":94,"icon":95},"Understanding Language Modeling Metrics","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","i-lucide-sigma",{"title":97,"path":98,"stem":99,"icon":100},"Exact Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-3-exact-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-3-exact-evaluation","i-lucide-check-check",{"title":102,"path":103,"stem":104,"icon":105},"AI as a Judge","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-4-ai-as-a-judge","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-4-ai-as-a-judge","i-lucide-scale",{"title":107,"path":108,"stem":109,"icon":110},"Ranking Models with Comparative Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","i-lucide-trophy",{"title":41,"path":112,"stem":113,"icon":44},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-6-summary","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-6-summary",{"id":115,"title":92,"body":116,"description":4872,"extension":4873,"links":6,"meta":4874,"navigation":4875,"path":93,"seo":4876,"stem":94,"__hash__":4877},"docs_en\u002Fen\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics.md",{"type":117,"value":118,"toc":4862},"minimark",[119,142,146,151,155,158,193,196,212,215,219,223,227,233,250,253,260,264,283,286,289,292,295,312,425,686,895,1126,1218,1222,1284,1408,1671,1761,1764,1798,1923,1929,2073,2076,2163,2170,2213,2333,2367,2511,2519,2523,2526,2529,2566,2572,2575,2585,2818,2830,2833,2852,2855,2859,3135,4363,4859],[120,121,122,126],"u-page-hero",{},[123,124,92],"template",{"v-slot:title":125},"",[123,127,128,129,136,137,141],{"v-slot:description":125},"Foundation models evolved out of language models. Many foundation models still have language models as their main components. For these models, the performance of the language model component tends to be well correlated to the foundation model's performance on downstream applications (",[130,131,135],"a",{"href":132,"rel":133},"https:\u002F\u002Fproceedings.mlr.press\u002Fv202\u002Fliu23ao.html",[134],"nofollow","Liu et al., 2023","). Therefore, a rough understanding of language modeling metrics can be quite helpful in understanding ",[138,139,140],"strong",{},"downstream performance",".",[143,144,145],"note",{},"While there's a strong correlation, language modeling performance doesn't fully explain downstream performance. This is an active area of research.",[147,148,150],"h2",{"id":149},"why-language-modeling-metrics-matter","Why Language Modeling Metrics Matter",[152,153,154],"p",{},"As discussed in Chapter 1, language modeling has been around for decades, popularized by Claude Shannon in his 1951 paper \"Prediction and Entropy of Printed English\". The metrics used to guide the development of language models haven't changed much since then. Most autoregressive language models are trained using cross entropy or its relative, perplexity. When reading papers and model reports, you might also come across bits-per-character (BPC) and bits-per-byte (BPB); both are variations of cross entropy.",[152,156,157],{},"All four metrics — cross entropy, perplexity, BPC, and BPB — are closely related. If you know the value of one, you can compute the other three, given the necessary information.",[159,160,161,169,178,186],"card-group",{},[162,163,165,166,141],"card",{"icon":95,"title":164},"Cross Entropy","Most autoregressive language models are trained using ",[138,167,168],{},"cross entropy",[162,170,173,174,177],{"icon":171,"title":172},"i-lucide-gauge","Perplexity","A ",[138,175,176],{},"relative of cross entropy",". Most autoregressive language models are trained using cross entropy or its relative, perplexity.",[162,179,173,182,185],{"icon":180,"title":181},"i-lucide-type","Bits-per-Character (BPC)",[138,183,184],{},"variation of cross entropy"," you might come across in papers and model reports.",[162,187,190,191,185],{"icon":188,"title":189},"i-lucide-binary","Bits-per-Byte (BPB)","Another ",[138,192,184],{},[143,194,195],{},"While I refer to them as language modeling metrics, they can be used for any model that generates sequences of tokens, including non-text tokens.",[152,197,198,199,203,204,207,208,211],{},"Recall that a language model encodes statistical information (how likely a token is to appear in a given context) about languages. Statistically, given the context ",[200,201,202],"code",{},"\"I like drinking __\"",", the next word is more likely to be ",[200,205,206],{},"\"tea\""," than ",[200,209,210],{},"\"charcoal\"",". The more statistical information that a model can capture, the better it is at predicting the next token.",[152,213,214],{},"In ML lingo, a language model learns the distribution of its training data. The better this model learns, the better it is at predicting what comes next in the training data, and the lower its training cross entropy. As with any ML model, you care about its performance not just on the training data but also on your production data. In general, the closer your data is to a model's training data, the better the model can perform on your data.",[216,217,218],"warning",{},"Compared to the rest of the book, this section is math-heavy. If you find it confusing, feel free to skip the math part and focus on the discussion of how to interpret these metrics.",[220,221,222],"tip",{},"Even if you're not training or finetuning language models, understanding these metrics can help with evaluating which models to use for your application. These metrics can occasionally be used for certain evaluation and data deduplication techniques, as discussed throughout this book.",[147,224,226],{"id":225},"entropy","Entropy",[152,228,229,232],{},[230,231,226],"em",{}," measures how much information, on average, a token carries. The higher the entropy, the more information each token carries, and the more bits are needed to represent a token.",[143,234,235,244],{},[152,236,237,238,243],{},"As discussed in Chapter 1, a token can be a character, a word, or part of a word. When Claude Shannon introduced entropy in 1951, the tokens he worked with were characters. Here's entropy in ",[130,239,242],{"href":240,"rel":241},"https:\u002F\u002Fwww.princeton.edu\u002F~wbialek\u002Frome\u002Frefs\u002Fshannon_51.pdf",[134],"his own words",":",[245,246,247],"blockquote",{},[152,248,249],{},"The entropy is a statistical parameter which measures, in a certain sense, how much information is produced on the average for each letter of a text in the language. If the language is translated into binary digits (0 or 1) in the most efficient way, the entropy is the average number of binary digits required per letter of the original language.",[152,251,252],{},"Let's use a simple example to illustrate this. Imagine you want to create a language to describe positions within a square, as shown in Figure 3-4.",[152,254,255],{},[256,257],"img",{"alt":258,"src":259},"Figure 3-4. Two languages describe positions within a square. Compared to the language on the left (a), the tokens on the right (b) carry more information, but they need more bits to represent them.",".\u002Fmedia\u002Ffig-3-4.png",[245,261,262],{},[152,263,258],{},[159,265,266,274],{},[162,267,270,271,141],{"icon":268,"title":269},"i-lucide-square","Two Tokens — Entropy 1","If your language has only two tokens, shown as (a) in Figure 3-4, each token can tell you whether the position is upper or lower. Since there are only two tokens, one bit is sufficient to represent them. The entropy of this language is, therefore, ",[138,272,273],{},"1",[162,275,278,279,282],{"icon":276,"title":277},"i-lucide-grid-2x2","Four Tokens — Entropy 2","If your language has four tokens, shown as (b) in Figure 3-4, each token can give you a more specific position: upper-left, upper-right, lower-left, or lower-right. However, since there are now four tokens, you need two bits to represent them. The entropy of this language is ",[138,280,281],{},"2",". This language has higher entropy, since each token carries more information, but each token requires more bits to represent.",[152,284,285],{},"Intuitively, entropy measures how difficult it is to predict what comes next in a language. The lower a language's entropy (the less information a token of a language carries), the more predictable that language. In our previous example, the language with only two tokens is easier to predict than the language with four (you have to predict among only two possible tokens compared to four). This is similar to how, if you can perfectly predict what I will say next, what I say carries no new information.",[147,287,164],{"id":288},"cross-entropy",[152,290,291],{},"When you train a language model on a dataset, your goal is to get the model to learn the distribution of this training data. In other words, your goal is to get the model to predict what comes next in the training data. A language model's cross entropy on a dataset measures how difficult it is for the language model to predict what comes next in this dataset.",[152,293,294],{},"A model's cross entropy on the training data depends on two qualities:",[159,296,297,303],{},[162,298,300,301,141],{"icon":73,"title":299},"Training Data Predictability","Measured by the training data's ",[138,302,225],{},[162,304,307,308,311],{"icon":305,"title":306},"i-lucide-git-compare","Distribution Divergence","How the distribution captured by the language model ",[138,309,310],{},"diverges"," from the true distribution of the training data.",[152,313,314,315,362,363,393,394,424],{},"Entropy and cross entropy share the same mathematical notation, ",[316,317,320,342],"span",{"className":318},[319],"katex",[316,321,324],{"className":322},[323],"katex-mathml",[325,326,328],"math",{"xmlns":327},"http:\u002F\u002Fwww.w3.org\u002F1998\u002FMath\u002FMathML",[329,330,331,338],"semantics",{},[332,333,334],"mrow",{},[335,336,337],"mi",{},"H",[339,340,337],"annotation",{"encoding":341},"application\u002Fx-tex",[316,343,347],{"className":344,"ariaHidden":346},[345],"katex-html","true",[316,348,351,356],{"className":349},[350],"base",[316,352],{"className":353,"style":355},[354],"strut","height:0.6833em;",[316,357,337],{"className":358,"style":361},[359,360],"mord","mathnormal","margin-right:0.0813em;",". Let ",[316,364,366,380],{"className":365},[319],[316,367,369],{"className":368},[323],[325,370,371],{"xmlns":327},[329,372,373,378],{},[332,374,375],{},[335,376,377],{},"P",[339,379,377],{"encoding":341},[316,381,383],{"className":382,"ariaHidden":346},[345],[316,384,386,389],{"className":385},[350],[316,387],{"className":388,"style":355},[354],[316,390,377],{"className":391,"style":392},[359,360],"margin-right:0.1389em;"," be the true distribution of the training data, and ",[316,395,397,411],{"className":396},[319],[316,398,400],{"className":399},[323],[325,401,402],{"xmlns":327},[329,403,404,409],{},[332,405,406],{},[335,407,408],{},"Q",[339,410,408],{"encoding":341},[316,412,414],{"className":413,"ariaHidden":346},[345],[316,415,417,421],{"className":416},[350],[316,418],{"className":419,"style":420},[354],"height:0.8778em;vertical-align:-0.1944em;",[316,422,408],{"className":423},[359,360]," be the distribution learned by the language model. Accordingly, the following is true:",[426,427,428,483,683],"ul",{},[429,430,431,432,141],"li",{},"The training data's entropy is, therefore, ",[316,433,435,459],{"className":434},[319],[316,436,438],{"className":437},[323],[325,439,440],{"xmlns":327},[329,441,442,456],{},[332,443,444,446,451,453],{},[335,445,337],{},[447,448,450],"mo",{"stretchy":449},"false","(",[335,452,377],{},[447,454,455],{"stretchy":449},")",[339,457,458],{"encoding":341},"H(P)",[316,460,462],{"className":461,"ariaHidden":346},[345],[316,463,465,469,472,476,479],{"className":464},[350],[316,466],{"className":467,"style":468},[354],"height:1em;vertical-align:-0.25em;",[316,470,337],{"className":471,"style":361},[359,360],[316,473,450],{"className":474},[475],"mopen",[316,477,377],{"className":478,"style":392},[359,360],[316,480,455],{"className":481},[482],"mclose",[429,484,485,486,514,515,543,544,141],{},"The divergence of ",[316,487,489,502],{"className":488},[319],[316,490,492],{"className":491},[323],[325,493,494],{"xmlns":327},[329,495,496,500],{},[332,497,498],{},[335,499,408],{},[339,501,408],{"encoding":341},[316,503,505],{"className":504,"ariaHidden":346},[345],[316,506,508,511],{"className":507},[350],[316,509],{"className":510,"style":420},[354],[316,512,408],{"className":513},[359,360]," with respect to ",[316,516,518,531],{"className":517},[319],[316,519,521],{"className":520},[323],[325,522,523],{"xmlns":327},[329,524,525,529],{},[332,526,527],{},[335,528,377],{},[339,530,377],{"encoding":341},[316,532,534],{"className":533,"ariaHidden":346},[345],[316,535,537,540],{"className":536},[350],[316,538],{"className":539,"style":355},[354],[316,541,377],{"className":542,"style":392},[359,360]," can be measured using the Kullback–Leibler (KL) divergence, which is mathematically represented as ",[316,545,547,580],{"className":546},[319],[316,548,550],{"className":549},[323],[325,551,552],{"xmlns":327},[329,553,554,577],{},[332,555,556,566,568,570,573,575],{},[557,558,559,562],"msub",{},[335,560,561],{},"D",[563,564,565],"mtext",{},"KL",[447,567,450],{"stretchy":449},[335,569,377],{},[447,571,572],{},"∥",[335,574,408],{},[447,576,455],{"stretchy":449},[339,578,579],{"encoding":341},"D_{\\text{KL}}(P \\parallel Q)",[316,581,583,671],{"className":582,"ariaHidden":346},[345],[316,584,586,589,653,656,659,664,668],{"className":585},[350],[316,587],{"className":588,"style":468},[354],[316,590,592,596],{"className":591},[359],[316,593,561],{"className":594,"style":595},[359,360],"margin-right:0.0278em;",[316,597,600],{"className":598},[599],"msupsub",[316,601,605,644],{"className":602},[603,604],"vlist-t","vlist-t2",[316,606,609,639],{"className":607},[608],"vlist-r",[316,610,614],{"className":611,"style":613},[612],"vlist","height:0.3283em;",[316,615,617,622],{"style":616},"top:-2.55em;margin-left:-0.0278em;margin-right:0.05em;",[316,618],{"className":619,"style":621},[620],"pstrut","height:2.7em;",[316,623,629],{"className":624},[625,626,627,628],"sizing","reset-size6","size3","mtight",[316,630,632],{"className":631},[359,628],[316,633,636],{"className":634},[359,635,628],"text",[316,637,565],{"className":638},[359,628],[316,640,643],{"className":641},[642],"vlist-s","​",[316,645,647],{"className":646},[608],[316,648,651],{"className":649,"style":650},[612],"height:0.15em;",[316,652],{},[316,654,450],{"className":655},[475],[316,657,377],{"className":658,"style":392},[359,360],[316,660],{"className":661,"style":663},[662],"mspace","margin-right:0.2778em;",[316,665,572],{"className":666},[667],"mrel",[316,669],{"className":670,"style":663},[662],[316,672,674,677,680],{"className":673},[350],[316,675],{"className":676,"style":468},[354],[316,678,408],{"className":679},[359,360],[316,681,455],{"className":682},[482],[429,684,685],{},"The model's cross entropy with respect to the training data is therefore:",[152,687,688],{},[316,689,691,746],{"className":690},[319],[316,692,694],{"className":693},[323],[325,695,696],{"xmlns":327},[329,697,698,743],{},[332,699,700,702,704,706,709,711,713,716,718,720,722,724,727,733,735,737,739,741],{},[335,701,337],{},[447,703,450],{"stretchy":449},[335,705,377],{},[447,707,708],{"separator":346},",",[335,710,408],{},[447,712,455],{"stretchy":449},[447,714,715],{},"=",[335,717,337],{},[447,719,450],{"stretchy":449},[335,721,377],{},[447,723,455],{"stretchy":449},[447,725,726],{},"+",[557,728,729,731],{},[335,730,561],{},[563,732,565],{},[447,734,450],{"stretchy":449},[335,736,377],{},[447,738,572],{},[335,740,408],{},[447,742,455],{"stretchy":449},[339,744,745],{"encoding":341},"H(P, Q) = H(P) + D_{\\text{KL}}(P \\parallel Q)",[316,747,749,787,816,883],{"className":748,"ariaHidden":346},[345],[316,750,752,755,758,761,764,768,772,775,778,781,784],{"className":751},[350],[316,753],{"className":754,"style":468},[354],[316,756,337],{"className":757,"style":361},[359,360],[316,759,450],{"className":760},[475],[316,762,377],{"className":763,"style":392},[359,360],[316,765,708],{"className":766},[767],"mpunct",[316,769],{"className":770,"style":771},[662],"margin-right:0.1667em;",[316,773,408],{"className":774},[359,360],[316,776,455],{"className":777},[482],[316,779],{"className":780,"style":663},[662],[316,782,715],{"className":783},[667],[316,785],{"className":786,"style":663},[662],[316,788,790,793,796,799,802,805,809,813],{"className":789},[350],[316,791],{"className":792,"style":468},[354],[316,794,337],{"className":795,"style":361},[359,360],[316,797,450],{"className":798},[475],[316,800,377],{"className":801,"style":392},[359,360],[316,803,455],{"className":804},[482],[316,806],{"className":807,"style":808},[662],"margin-right:0.2222em;",[316,810,726],{"className":811},[812],"mbin",[316,814],{"className":815,"style":808},[662],[316,817,819,822,868,871,874,877,880],{"className":818},[350],[316,820],{"className":821,"style":468},[354],[316,823,825,828],{"className":824},[359],[316,826,561],{"className":827,"style":595},[359,360],[316,829,831],{"className":830},[599],[316,832,834,860],{"className":833},[603,604],[316,835,837,857],{"className":836},[608],[316,838,840],{"className":839,"style":613},[612],[316,841,842,845],{"style":616},[316,843],{"className":844,"style":621},[620],[316,846,848],{"className":847},[625,626,627,628],[316,849,851],{"className":850},[359,628],[316,852,854],{"className":853},[359,635,628],[316,855,565],{"className":856},[359,628],[316,858,643],{"className":859},[642],[316,861,863],{"className":862},[608],[316,864,866],{"className":865,"style":650},[612],[316,867],{},[316,869,450],{"className":870},[475],[316,872,377],{"className":873,"style":392},[359,360],[316,875],{"className":876,"style":663},[662],[316,878,572],{"className":879},[667],[316,881],{"className":882,"style":663},[662],[316,884,886,889,892],{"className":885},[350],[316,887],{"className":888,"style":468},[354],[316,890,408],{"className":891},[359,360],[316,893,455],{"className":894},[482],[152,896,897,898,514,926,954,955,1012,1013,514,1041,954,1069,141],{},"Cross entropy isn't symmetric. The cross entropy of ",[316,899,901,914],{"className":900},[319],[316,902,904],{"className":903},[323],[325,905,906],{"xmlns":327},[329,907,908,912],{},[332,909,910],{},[335,911,408],{},[339,913,408],{"encoding":341},[316,915,917],{"className":916,"ariaHidden":346},[345],[316,918,920,923],{"className":919},[350],[316,921],{"className":922,"style":420},[354],[316,924,408],{"className":925},[359,360],[316,927,929,942],{"className":928},[319],[316,930,932],{"className":931},[323],[325,933,934],{"xmlns":327},[329,935,936,940],{},[332,937,938],{},[335,939,377],{},[339,941,377],{"encoding":341},[316,943,945],{"className":944,"ariaHidden":346},[345],[316,946,948,951],{"className":947},[350],[316,949],{"className":950,"style":355},[354],[316,952,377],{"className":953,"style":392},[359,360]," — ",[316,956,958,982],{"className":957},[319],[316,959,961],{"className":960},[323],[325,962,963],{"xmlns":327},[329,964,965,979],{},[332,966,967,969,971,973,975,977],{},[335,968,337],{},[447,970,450],{"stretchy":449},[335,972,377],{},[447,974,708],{"separator":346},[335,976,408],{},[447,978,455],{"stretchy":449},[339,980,981],{"encoding":341},"H(P, Q)",[316,983,985],{"className":984,"ariaHidden":346},[345],[316,986,988,991,994,997,1000,1003,1006,1009],{"className":987},[350],[316,989],{"className":990,"style":468},[354],[316,992,337],{"className":993,"style":361},[359,360],[316,995,450],{"className":996},[475],[316,998,377],{"className":999,"style":392},[359,360],[316,1001,708],{"className":1002},[767],[316,1004],{"className":1005,"style":771},[662],[316,1007,408],{"className":1008},[359,360],[316,1010,455],{"className":1011},[482]," — is different from the cross entropy of ",[316,1014,1016,1029],{"className":1015},[319],[316,1017,1019],{"className":1018},[323],[325,1020,1021],{"xmlns":327},[329,1022,1023,1027],{},[332,1024,1025],{},[335,1026,377],{},[339,1028,377],{"encoding":341},[316,1030,1032],{"className":1031,"ariaHidden":346},[345],[316,1033,1035,1038],{"className":1034},[350],[316,1036],{"className":1037,"style":355},[354],[316,1039,377],{"className":1040,"style":392},[359,360],[316,1042,1044,1057],{"className":1043},[319],[316,1045,1047],{"className":1046},[323],[325,1048,1049],{"xmlns":327},[329,1050,1051,1055],{},[332,1052,1053],{},[335,1054,408],{},[339,1056,408],{"encoding":341},[316,1058,1060],{"className":1059,"ariaHidden":346},[345],[316,1061,1063,1066],{"className":1062},[350],[316,1064],{"className":1065,"style":420},[354],[316,1067,408],{"className":1068},[359,360],[316,1070,1072,1096],{"className":1071},[319],[316,1073,1075],{"className":1074},[323],[325,1076,1077],{"xmlns":327},[329,1078,1079,1093],{},[332,1080,1081,1083,1085,1087,1089,1091],{},[335,1082,337],{},[447,1084,450],{"stretchy":449},[335,1086,408],{},[447,1088,708],{"separator":346},[335,1090,377],{},[447,1092,455],{"stretchy":449},[339,1094,1095],{"encoding":341},"H(Q, P)",[316,1097,1099],{"className":1098,"ariaHidden":346},[345],[316,1100,1102,1105,1108,1111,1114,1117,1120,1123],{"className":1101},[350],[316,1103],{"className":1104,"style":468},[354],[316,1106,337],{"className":1107,"style":361},[359,360],[316,1109,450],{"className":1110},[475],[316,1112,408],{"className":1113},[359,360],[316,1115,708],{"className":1116},[767],[316,1118],{"className":1119,"style":771},[662],[316,1121,377],{"className":1122,"style":392},[359,360],[316,1124,455],{"className":1125},[482],[152,1127,1128,1129,514,1157,1185,1186,1217],{},"A language model is trained to minimize its cross entropy with respect to the training data. If the language model learns perfectly from its training data, the model's cross entropy will be exactly the same as the entropy of the training data. The KL divergence of ",[316,1130,1132,1145],{"className":1131},[319],[316,1133,1135],{"className":1134},[323],[325,1136,1137],{"xmlns":327},[329,1138,1139,1143],{},[332,1140,1141],{},[335,1142,408],{},[339,1144,408],{"encoding":341},[316,1146,1148],{"className":1147,"ariaHidden":346},[345],[316,1149,1151,1154],{"className":1150},[350],[316,1152],{"className":1153,"style":420},[354],[316,1155,408],{"className":1156},[359,360],[316,1158,1160,1173],{"className":1159},[319],[316,1161,1163],{"className":1162},[323],[325,1164,1165],{"xmlns":327},[329,1166,1167,1171],{},[332,1168,1169],{},[335,1170,377],{},[339,1172,377],{"encoding":341},[316,1174,1176],{"className":1175,"ariaHidden":346},[345],[316,1177,1179,1182],{"className":1178},[350],[316,1180],{"className":1181,"style":355},[354],[316,1183,377],{"className":1184,"style":392},[359,360]," will then be ",[316,1187,1189,1204],{"className":1188},[319],[316,1190,1192],{"className":1191},[323],[325,1193,1194],{"xmlns":327},[329,1195,1196,1202],{},[332,1197,1198],{},[1199,1200,1201],"mn",{},"0",[339,1203,1201],{"encoding":341},[316,1205,1207],{"className":1206,"ariaHidden":346},[345],[316,1208,1210,1214],{"className":1209},[350],[316,1211],{"className":1212,"style":1213},[354],"height:0.6444em;",[316,1215,1201],{"className":1216},[359],". You can think of a model's cross entropy as its approximation of the entropy of its training data.",[147,1219,1221],{"id":1220},"bits-per-character-and-bits-per-byte","Bits-per-Character and Bits-per-Byte",[152,1223,1224,1225,1254,1255,1283],{},"One unit of entropy and cross entropy is bits. If the cross entropy of a language model is ",[316,1226,1228,1242],{"className":1227},[319],[316,1229,1231],{"className":1230},[323],[325,1232,1233],{"xmlns":327},[329,1234,1235,1240],{},[332,1236,1237],{},[1199,1238,1239],{},"6",[339,1241,1239],{"encoding":341},[316,1243,1245],{"className":1244,"ariaHidden":346},[345],[316,1246,1248,1251],{"className":1247},[350],[316,1249],{"className":1250,"style":1213},[354],[316,1252,1239],{"className":1253},[359]," bits, this language model needs ",[316,1256,1258,1271],{"className":1257},[319],[316,1259,1261],{"className":1260},[323],[325,1262,1263],{"xmlns":327},[329,1264,1265,1269],{},[332,1266,1267],{},[1199,1268,1239],{},[339,1270,1239],{"encoding":341},[316,1272,1274],{"className":1273,"ariaHidden":346},[345],[316,1275,1277,1280],{"className":1276},[350],[316,1278],{"className":1279,"style":1213},[354],[316,1281,1239],{"className":1282},[359]," bits to represent each token.",[152,1285,1286,1287,1290,1291,1319,1320,1348,1349,141],{},"Since different models have different tokenization methods — for example, one model uses words as tokens and another uses characters as tokens — the number of bits per token isn't comparable across models. Some use the number of ",[230,1288,1289],{},"bits-per-character"," (BPC) instead. If the number of bits per token is ",[316,1292,1294,1307],{"className":1293},[319],[316,1295,1297],{"className":1296},[323],[325,1298,1299],{"xmlns":327},[329,1300,1301,1305],{},[332,1302,1303],{},[1199,1304,1239],{},[339,1306,1239],{"encoding":341},[316,1308,1310],{"className":1309,"ariaHidden":346},[345],[316,1311,1313,1316],{"className":1312},[350],[316,1314],{"className":1315,"style":1213},[354],[316,1317,1239],{"className":1318},[359]," and on average, each token consists of ",[316,1321,1323,1336],{"className":1322},[319],[316,1324,1326],{"className":1325},[323],[325,1327,1328],{"xmlns":327},[329,1329,1330,1334],{},[332,1331,1332],{},[1199,1333,281],{},[339,1335,281],{"encoding":341},[316,1337,1339],{"className":1338,"ariaHidden":346},[345],[316,1340,1342,1345],{"className":1341},[350],[316,1343],{"className":1344,"style":1213},[354],[316,1346,281],{"className":1347},[359]," characters, the BPC is ",[316,1350,1352,1377],{"className":1351},[319],[316,1353,1355],{"className":1354},[323],[325,1356,1357],{"xmlns":327},[329,1358,1359,1374],{},[332,1360,1361,1363,1367,1369,1371],{},[1199,1362,1239],{},[335,1364,1366],{"mathvariant":1365},"normal","\u002F",[1199,1368,281],{},[447,1370,715],{},[1199,1372,1373],{},"3",[339,1375,1376],{"encoding":341},"6\u002F2 = 3",[316,1378,1380,1399],{"className":1379,"ariaHidden":346},[345],[316,1381,1383,1386,1390,1393,1396],{"className":1382},[350],[316,1384],{"className":1385,"style":468},[354],[316,1387,1389],{"className":1388},[359],"6\u002F2",[316,1391],{"className":1392,"style":663},[662],[316,1394,715],{"className":1395},[667],[316,1397],{"className":1398,"style":663},[662],[316,1400,1402,1405],{"className":1401},[350],[316,1403],{"className":1404,"style":1213},[354],[316,1406,1373],{"className":1407},[359],[152,1409,1410,1411,1440,1441,1470,1471,1500,1501,1504,1505,1533,1534,1562,1563,1596,1597,141],{},"One complication with BPC arises from different character encoding schemes. For example, with ASCII, each character is encoded using ",[316,1412,1414,1428],{"className":1413},[319],[316,1415,1417],{"className":1416},[323],[325,1418,1419],{"xmlns":327},[329,1420,1421,1426],{},[332,1422,1423],{},[1199,1424,1425],{},"7",[339,1427,1425],{"encoding":341},[316,1429,1431],{"className":1430,"ariaHidden":346},[345],[316,1432,1434,1437],{"className":1433},[350],[316,1435],{"className":1436,"style":1213},[354],[316,1438,1425],{"className":1439},[359]," bits, but with UTF-8, a character can be encoded using anywhere between ",[316,1442,1444,1458],{"className":1443},[319],[316,1445,1447],{"className":1446},[323],[325,1448,1449],{"xmlns":327},[329,1450,1451,1456],{},[332,1452,1453],{},[1199,1454,1455],{},"8",[339,1457,1455],{"encoding":341},[316,1459,1461],{"className":1460,"ariaHidden":346},[345],[316,1462,1464,1467],{"className":1463},[350],[316,1465],{"className":1466,"style":1213},[354],[316,1468,1455],{"className":1469},[359]," and ",[316,1472,1474,1488],{"className":1473},[319],[316,1475,1477],{"className":1476},[323],[325,1478,1479],{"xmlns":327},[329,1480,1481,1486],{},[332,1482,1483],{},[1199,1484,1485],{},"32",[339,1487,1485],{"encoding":341},[316,1489,1491],{"className":1490,"ariaHidden":346},[345],[316,1492,1494,1497],{"className":1493},[350],[316,1495],{"className":1496,"style":1213},[354],[316,1498,1485],{"className":1499},[359]," bits. A more standardized metric would be ",[230,1502,1503],{},"bits-per-byte"," (BPB), the number of bits a language model needs to represent one byte of the original training data. If the BPC is ",[316,1506,1508,1521],{"className":1507},[319],[316,1509,1511],{"className":1510},[323],[325,1512,1513],{"xmlns":327},[329,1514,1515,1519],{},[332,1516,1517],{},[1199,1518,1373],{},[339,1520,1373],{"encoding":341},[316,1522,1524],{"className":1523,"ariaHidden":346},[345],[316,1525,1527,1530],{"className":1526},[350],[316,1528],{"className":1529,"style":1213},[354],[316,1531,1373],{"className":1532},[359]," and each character is ",[316,1535,1537,1550],{"className":1536},[319],[316,1538,1540],{"className":1539},[323],[325,1541,1542],{"xmlns":327},[329,1543,1544,1548],{},[332,1545,1546],{},[1199,1547,1425],{},[339,1549,1425],{"encoding":341},[316,1551,1553],{"className":1552,"ariaHidden":346},[345],[316,1554,1556,1559],{"className":1555},[350],[316,1557],{"className":1558,"style":1213},[354],[316,1560,1425],{"className":1561},[359]," bits, or ",[316,1564,1566,1584],{"className":1565},[319],[316,1567,1569],{"className":1568},[323],[325,1570,1571],{"xmlns":327},[329,1572,1573,1581],{},[332,1574,1575,1577,1579],{},[1199,1576,1425],{},[335,1578,1366],{"mathvariant":1365},[1199,1580,1455],{},[339,1582,1583],{"encoding":341},"7\u002F8",[316,1585,1587],{"className":1586,"ariaHidden":346},[345],[316,1588,1590,1593],{"className":1589},[350],[316,1591],{"className":1592,"style":468},[354],[316,1594,1583],{"className":1595},[359]," of a byte, then the BPB is ",[316,1598,1600,1631],{"className":1599},[319],[316,1601,1603],{"className":1602},[323],[325,1604,1605],{"xmlns":327},[329,1606,1607,1628],{},[332,1608,1609,1611,1613,1615,1617,1619,1621,1623,1625],{},[1199,1610,1373],{},[335,1612,1366],{"mathvariant":1365},[447,1614,450],{"stretchy":449},[1199,1616,1425],{},[335,1618,1366],{"mathvariant":1365},[1199,1620,1455],{},[447,1622,455],{"stretchy":449},[447,1624,715],{},[1199,1626,1627],{},"3.43",[339,1629,1630],{"encoding":341},"3 \u002F (7\u002F8) = 3.43",[316,1632,1634,1662],{"className":1633,"ariaHidden":346},[345],[316,1635,1637,1640,1644,1647,1650,1653,1656,1659],{"className":1636},[350],[316,1638],{"className":1639,"style":468},[354],[316,1641,1643],{"className":1642},[359],"3\u002F",[316,1645,450],{"className":1646},[475],[316,1648,1583],{"className":1649},[359],[316,1651,455],{"className":1652},[482],[316,1654],{"className":1655,"style":663},[662],[316,1657,715],{"className":1658},[667],[316,1660],{"className":1661,"style":663},[662],[316,1663,1665,1668],{"className":1664},[350],[316,1666],{"className":1667,"style":1213},[354],[316,1669,1627],{"className":1670},[359],[152,1672,1673,1674,1702,1703,1731,1732,1760],{},"Cross entropy tells us how efficient a language model will be at compressing text. If the BPB of a language model is ",[316,1675,1677,1690],{"className":1676},[319],[316,1678,1680],{"className":1679},[323],[325,1681,1682],{"xmlns":327},[329,1683,1684,1688],{},[332,1685,1686],{},[1199,1687,1627],{},[339,1689,1627],{"encoding":341},[316,1691,1693],{"className":1692,"ariaHidden":346},[345],[316,1694,1696,1699],{"className":1695},[350],[316,1697],{"className":1698,"style":1213},[354],[316,1700,1627],{"className":1701},[359],", meaning it can represent each original byte (",[316,1704,1706,1719],{"className":1705},[319],[316,1707,1709],{"className":1708},[323],[325,1710,1711],{"xmlns":327},[329,1712,1713,1717],{},[332,1714,1715],{},[1199,1716,1455],{},[339,1718,1455],{"encoding":341},[316,1720,1722],{"className":1721,"ariaHidden":346},[345],[316,1723,1725,1728],{"className":1724},[350],[316,1726],{"className":1727,"style":1213},[354],[316,1729,1455],{"className":1730},[359]," bits) using ",[316,1733,1735,1748],{"className":1734},[319],[316,1736,1738],{"className":1737},[323],[325,1739,1740],{"xmlns":327},[329,1741,1742,1746],{},[332,1743,1744],{},[1199,1745,1627],{},[339,1747,1627],{"encoding":341},[316,1749,1751],{"className":1750,"ariaHidden":346},[345],[316,1752,1754,1757],{"className":1753},[350],[316,1755],{"className":1756,"style":1213},[354],[316,1758,1627],{"className":1759},[359]," bits, this language model can compress the original training text to less than half the text's original size.",[147,1762,172],{"id":1763},"perplexity",[152,1765,1766,1768,1769,1797],{},[230,1767,172],{}," is the exponential of entropy and cross entropy. Perplexity is often shortened to PPL. Given a dataset with the true distribution ",[316,1770,1772,1785],{"className":1771},[319],[316,1773,1775],{"className":1774},[323],[325,1776,1777],{"xmlns":327},[329,1778,1779,1783],{},[332,1780,1781],{},[335,1782,377],{},[339,1784,377],{"encoding":341},[316,1786,1788],{"className":1787,"ariaHidden":346},[345],[316,1789,1791,1794],{"className":1790},[350],[316,1792],{"className":1793,"style":355},[354],[316,1795,377],{"className":1796,"style":392},[359,360],", its perplexity is defined as:",[152,1799,1800],{},[316,1801,1803,1841],{"className":1802},[319],[316,1804,1806],{"className":1805},[323],[325,1807,1808],{"xmlns":327},[329,1809,1810,1838],{},[332,1811,1812,1815,1817,1819,1821,1823],{},[563,1813,1814],{},"PPL",[447,1816,450],{"stretchy":449},[335,1818,377],{},[447,1820,455],{"stretchy":449},[447,1822,715],{},[1824,1825,1826,1828],"msup",{},[1199,1827,281],{},[332,1829,1830,1832,1834,1836],{},[335,1831,337],{},[447,1833,450],{"stretchy":449},[335,1835,377],{},[447,1837,455],{"stretchy":449},[339,1839,1840],{"encoding":341},"\\text{PPL}(P) = 2^{H(P)}",[316,1842,1844,1874],{"className":1843,"ariaHidden":346},[345],[316,1845,1847,1850,1856,1859,1862,1865,1868,1871],{"className":1846},[350],[316,1848],{"className":1849,"style":468},[354],[316,1851,1853],{"className":1852},[359,635],[316,1854,1814],{"className":1855},[359],[316,1857,450],{"className":1858},[475],[316,1860,377],{"className":1861,"style":392},[359,360],[316,1863,455],{"className":1864},[482],[316,1866],{"className":1867,"style":663},[662],[316,1869,715],{"className":1870},[667],[316,1872],{"className":1873,"style":663},[662],[316,1875,1877,1881],{"className":1876},[350],[316,1878],{"className":1879,"style":1880},[354],"height:0.888em;",[316,1882,1884,1887],{"className":1883},[359],[316,1885,281],{"className":1886},[359],[316,1888,1890],{"className":1889},[599],[316,1891,1893],{"className":1892},[603],[316,1894,1896],{"className":1895},[608],[316,1897,1899],{"className":1898,"style":1880},[612],[316,1900,1902,1905],{"style":1901},"top:-3.063em;margin-right:0.05em;",[316,1903],{"className":1904,"style":621},[620],[316,1906,1908],{"className":1907},[625,626,627,628],[316,1909,1911,1914,1917,1920],{"className":1910},[359,628],[316,1912,337],{"className":1913,"style":361},[359,360,628],[316,1915,450],{"className":1916},[475,628],[316,1918,377],{"className":1919,"style":392},[359,360,628],[316,1921,455],{"className":1922},[482,628],[152,1924,1925,1926,1928],{},"The perplexity of a language model (with the learned distribution ",[230,1927,408],{},") on this dataset is defined as:",[152,1930,1931],{},[316,1932,1934,1978],{"className":1933},[319],[316,1935,1937],{"className":1936},[323],[325,1938,1939],{"xmlns":327},[329,1940,1941,1975],{},[332,1942,1943,1945,1947,1949,1951,1953,1955,1957],{},[563,1944,1814],{},[447,1946,450],{"stretchy":449},[335,1948,377],{},[447,1950,708],{"separator":346},[335,1952,408],{},[447,1954,455],{"stretchy":449},[447,1956,715],{},[1824,1958,1959,1961],{},[1199,1960,281],{},[332,1962,1963,1965,1967,1969,1971,1973],{},[335,1964,337],{},[447,1966,450],{"stretchy":449},[335,1968,377],{},[447,1970,708],{"separator":346},[335,1972,408],{},[447,1974,455],{"stretchy":449},[339,1976,1977],{"encoding":341},"\\text{PPL}(P, Q) = 2^{H(P, Q)}",[316,1979,1981,2020],{"className":1980,"ariaHidden":346},[345],[316,1982,1984,1987,1993,1996,1999,2002,2005,2008,2011,2014,2017],{"className":1983},[350],[316,1985],{"className":1986,"style":468},[354],[316,1988,1990],{"className":1989},[359,635],[316,1991,1814],{"className":1992},[359],[316,1994,450],{"className":1995},[475],[316,1997,377],{"className":1998,"style":392},[359,360],[316,2000,708],{"className":2001},[767],[316,2003],{"className":2004,"style":771},[662],[316,2006,408],{"className":2007},[359,360],[316,2009,455],{"className":2010},[482],[316,2012],{"className":2013,"style":663},[662],[316,2015,715],{"className":2016},[667],[316,2018],{"className":2019,"style":663},[662],[316,2021,2023,2026],{"className":2022},[350],[316,2024],{"className":2025,"style":1880},[354],[316,2027,2029,2032],{"className":2028},[359],[316,2030,281],{"className":2031},[359],[316,2033,2035],{"className":2034},[599],[316,2036,2038],{"className":2037},[603],[316,2039,2041],{"className":2040},[608],[316,2042,2044],{"className":2043,"style":1880},[612],[316,2045,2046,2049],{"style":1901},[316,2047],{"className":2048,"style":621},[620],[316,2050,2052],{"className":2051},[625,626,627,628],[316,2053,2055,2058,2061,2064,2067,2070],{"className":2054},[359,628],[316,2056,337],{"className":2057,"style":361},[359,360,628],[316,2059,450],{"className":2060},[475,628],[316,2062,377],{"className":2063,"style":392},[359,360,628],[316,2065,708],{"className":2066},[767,628],[316,2068,408],{"className":2069},[359,360,628],[316,2071,455],{"className":2072},[482,628],[152,2074,2075],{},"If cross entropy measures how difficult it is for a model to predict the next token, perplexity measures the amount of uncertainty it has when predicting the next token. Higher uncertainty means there are more possible options for the next token.",[152,2077,2078,2079,2162],{},"Consider a language model trained to encode the 4 position tokens, as in Figure 3-4 (b), perfectly. The cross entropy of this language model is 2 bits. If this language model tries to predict a position in the square, it has to choose among ",[316,2080,2082,2105],{"className":2081},[319],[316,2083,2085],{"className":2084},[323],[325,2086,2087],{"xmlns":327},[329,2088,2089,2102],{},[332,2090,2091,2097,2099],{},[1824,2092,2093,2095],{},[1199,2094,281],{},[1199,2096,281],{},[447,2098,715],{},[1199,2100,2101],{},"4",[339,2103,2104],{"encoding":341},"2^2 = 4",[316,2106,2108,2153],{"className":2107,"ariaHidden":346},[345],[316,2109,2111,2115,2144,2147,2150],{"className":2110},[350],[316,2112],{"className":2113,"style":2114},[354],"height:0.8141em;",[316,2116,2118,2121],{"className":2117},[359],[316,2119,281],{"className":2120},[359],[316,2122,2124],{"className":2123},[599],[316,2125,2127],{"className":2126},[603],[316,2128,2130],{"className":2129},[608],[316,2131,2133],{"className":2132,"style":2114},[612],[316,2134,2135,2138],{"style":1901},[316,2136],{"className":2137,"style":621},[620],[316,2139,2141],{"className":2140},[625,626,627,628],[316,2142,281],{"className":2143},[359,628],[316,2145],{"className":2146,"style":663},[662],[316,2148,715],{"className":2149},[667],[316,2151],{"className":2152,"style":663},[662],[316,2154,2156,2159],{"className":2155},[350],[316,2157],{"className":2158,"style":1213},[354],[316,2160,2101],{"className":2161},[359]," possible options. Thus, this language model has a perplexity of 4.",[152,2164,2165,2166,2169],{},"So far, I've been using ",[230,2167,2168],{},"bit"," as the unit for entropy and cross entropy. Each bit can represent 2 unique values, hence the base of 2 in the preceding perplexity equation.",[152,2171,2172,2173,2176,2177,2212],{},"Popular ML frameworks, including TensorFlow and PyTorch, use ",[230,2174,2175],{},"nat"," (natural log) as the unit for entropy and cross entropy. Nat uses the ",[130,2178,2181,2182],{"href":2179,"rel":2180},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FE_(mathematical_constant)",[134],"base of ",[316,2183,2185,2199],{"className":2184},[319],[316,2186,2188],{"className":2187},[323],[325,2189,2190],{"xmlns":327},[329,2191,2192,2197],{},[332,2193,2194],{},[335,2195,2196],{},"e",[339,2198,2196],{"encoding":341},[316,2200,2202],{"className":2201,"ariaHidden":346},[345],[316,2203,2205,2209],{"className":2204},[350],[316,2206],{"className":2207,"style":2208},[354],"height:0.4306em;",[316,2210,2196],{"className":2211},[359,360],", the base of natural logarithm.",[143,2214,2215,2216,2244,2245,2295,2296,141],{},"One reason many people might prefer natural log over log base ",[316,2217,2219,2232],{"className":2218},[319],[316,2220,2222],{"className":2221},[323],[325,2223,2224],{"xmlns":327},[329,2225,2226,2230],{},[332,2227,2228],{},[1199,2229,281],{},[339,2231,281],{"encoding":341},[316,2233,2235],{"className":2234,"ariaHidden":346},[345],[316,2236,2238,2241],{"className":2237},[350],[316,2239],{"className":2240,"style":1213},[354],[316,2242,281],{"className":2243},[359]," is because natural log has certain properties that makes its math easier. For example, the derivative of natural log ",[316,2246,2248,2273],{"className":2247},[319],[316,2249,2251],{"className":2250},[323],[325,2252,2253],{"xmlns":327},[329,2254,2255,2270],{},[332,2256,2257,2260,2263,2265,2268],{},[335,2258,2259],{},"ln",[447,2261,2262],{},"⁡",[447,2264,450],{"stretchy":449},[335,2266,2267],{},"x",[447,2269,455],{"stretchy":449},[339,2271,2272],{"encoding":341},"\\ln(x)",[316,2274,2276],{"className":2275,"ariaHidden":346},[345],[316,2277,2279,2282,2286,2289,2292],{"className":2278},[350],[316,2280],{"className":2281,"style":468},[354],[316,2283,2259],{"className":2284},[2285],"mop",[316,2287,450],{"className":2288},[475],[316,2290,2267],{"className":2291},[359,360],[316,2293,455],{"className":2294},[482]," is ",[316,2297,2299,2317],{"className":2298},[319],[316,2300,2302],{"className":2301},[323],[325,2303,2304],{"xmlns":327},[329,2305,2306,2314],{},[332,2307,2308,2310,2312],{},[1199,2309,273],{},[335,2311,1366],{"mathvariant":1365},[335,2313,2267],{},[339,2315,2316],{"encoding":341},"1\u002Fx",[316,2318,2320],{"className":2319,"ariaHidden":346},[345],[316,2321,2323,2326,2330],{"className":2322},[350],[316,2324],{"className":2325,"style":468},[354],[316,2327,2329],{"className":2328},[359],"1\u002F",[316,2331,2267],{"className":2332},[359,360],[152,2334,2335,2336,2338,2339,243],{},"If you use ",[230,2337,2175],{}," as the unit, perplexity is the exponential of ",[316,2340,2342,2355],{"className":2341},[319],[316,2343,2345],{"className":2344},[323],[325,2346,2347],{"xmlns":327},[329,2348,2349,2353],{},[332,2350,2351],{},[335,2352,2196],{},[339,2354,2196],{"encoding":341},[316,2356,2358],{"className":2357,"ariaHidden":346},[345],[316,2359,2361,2364],{"className":2360},[350],[316,2362],{"className":2363,"style":2208},[354],[316,2365,2196],{"className":2366},[359,360],[152,2368,2369],{},[316,2370,2372,2416],{"className":2371},[319],[316,2373,2375],{"className":2374},[323],[325,2376,2377],{"xmlns":327},[329,2378,2379,2413],{},[332,2380,2381,2383,2385,2387,2389,2391,2393,2395],{},[563,2382,1814],{},[447,2384,450],{"stretchy":449},[335,2386,377],{},[447,2388,708],{"separator":346},[335,2390,408],{},[447,2392,455],{"stretchy":449},[447,2394,715],{},[1824,2396,2397,2399],{},[335,2398,2196],{},[332,2400,2401,2403,2405,2407,2409,2411],{},[335,2402,337],{},[447,2404,450],{"stretchy":449},[335,2406,377],{},[447,2408,708],{"separator":346},[335,2410,408],{},[447,2412,455],{"stretchy":449},[339,2414,2415],{"encoding":341},"\\text{PPL}(P, Q) = e^{H(P, Q)}",[316,2417,2419,2458],{"className":2418,"ariaHidden":346},[345],[316,2420,2422,2425,2431,2434,2437,2440,2443,2446,2449,2452,2455],{"className":2421},[350],[316,2423],{"className":2424,"style":468},[354],[316,2426,2428],{"className":2427},[359,635],[316,2429,1814],{"className":2430},[359],[316,2432,450],{"className":2433},[475],[316,2435,377],{"className":2436,"style":392},[359,360],[316,2438,708],{"className":2439},[767],[316,2441],{"className":2442,"style":771},[662],[316,2444,408],{"className":2445},[359,360],[316,2447,455],{"className":2448},[482],[316,2450],{"className":2451,"style":663},[662],[316,2453,715],{"className":2454},[667],[316,2456],{"className":2457,"style":663},[662],[316,2459,2461,2464],{"className":2460},[350],[316,2462],{"className":2463,"style":1880},[354],[316,2465,2467,2470],{"className":2466},[359],[316,2468,2196],{"className":2469},[359,360],[316,2471,2473],{"className":2472},[599],[316,2474,2476],{"className":2475},[603],[316,2477,2479],{"className":2478},[608],[316,2480,2482],{"className":2481,"style":1880},[612],[316,2483,2484,2487],{"style":1901},[316,2485],{"className":2486,"style":621},[620],[316,2488,2490],{"className":2489},[625,626,627,628],[316,2491,2493,2496,2499,2502,2505,2508],{"className":2492},[359,628],[316,2494,337],{"className":2495,"style":361},[359,360,628],[316,2497,450],{"className":2498},[475,628],[316,2500,377],{"className":2501,"style":392},[359,360,628],[316,2503,708],{"className":2504},[767,628],[316,2506,408],{"className":2507},[359,360,628],[316,2509,455],{"className":2510},[482,628],[220,2512,2513,2514,1470,2516,2518],{},"Due to the confusion around ",[230,2515,2168],{},[230,2517,2175],{},", many people report perplexity, instead of cross entropy, when reporting their language models' performance.",[147,2520,2522],{"id":2521},"perplexity-interpretation-and-use-cases","Perplexity Interpretation and Use Cases",[152,2524,2525],{},"As discussed, cross entropy, perplexity, BPC, and BPB are variations of language models' predictive accuracy measurements. The more accurately a model can predict a text, the lower these metrics are. In this book, I'll use perplexity as the default language modeling metric. Remember that the more uncertainty the model has in predicting what comes next in a given dataset, the higher the perplexity.",[152,2527,2528],{},"What's considered a good value for perplexity depends on the data itself and how exactly perplexity is computed, such as how many previous tokens a model has access to. Here are some general rules:",[159,2530,2531,2544,2553],{},[162,2532,2535,2536,2539,2540,2543],{"icon":2533,"title":2534},"i-lucide-code","More Structured Data Gives Lower Expected Perplexity","More structured data is more predictable. For example, HTML code is more predictable than everyday text. If you see an opening HTML tag like ",[200,2537,2538],{},"\u003Chead>",", you can predict that there should be a closing tag, ",[200,2541,2542],{},"\u003C\u002Fhead>",", nearby. Therefore, the expected perplexity of a model on HTML code should be lower than the expected perplexity of a model on everyday text.",[162,2545,2548,2549,2552],{"icon":2546,"title":2547},"i-lucide-book-text","The Bigger the Vocabulary, the Higher the Perplexity","Intuitively, the more possible tokens there are, the harder it is for the model to predict the next token. For example, a model's perplexity on a children's book will likely be lower than the same model's perplexity on ",[230,2550,2551],{},"War and Peace",". For the same dataset, say in English, character-based perplexity (predicting the next character) will be lower than word-based perplexity (predicting the next word), because the number of possible characters is smaller than the number of possible words.",[162,2554,2557,2558,2561,2562,2565],{"icon":2555,"title":2556},"i-lucide-text-select","The Longer the Context Length, the Lower the Perplexity","The more context a model has, the less uncertainty it will have in predicting the next token. In 1951, Claude Shannon evaluated his model's cross entropy by using it to predict the next token conditioned on up to ",[138,2559,2560],{},"10"," previous tokens. As of this writing, a model's perplexity can typically be computed and conditioned on between ",[138,2563,2564],{},"500 and 10,000"," previous tokens, and possibly more, upperbounded by the model's maximum context length.",[143,2567,2568,2569,2571],{},"For reference, it's not uncommon to see perplexity values as low as ",[138,2570,1373],{}," or even lower. If all tokens in a hypothetical language have an equal chance of happening, a perplexity of 3 means that this model has a 1 in 3 chance of predicting the next token correctly. Given that a model's vocabulary is in the order of 10,000s and 100,000s, these odds are incredible.",[152,2573,2574],{},"Other than guiding the training of language models, perplexity is useful in many parts of an AI engineering workflow. First, perplexity is a good proxy for a model's capabilities. If a model's bad at predicting the next token, its performance on downstream tasks will also likely be bad. OpenAI's GPT-2 report shows that larger models, which are also more powerful models, consistently give lower perplexity on a range of datasets, as shown in Table 3-1. Sadly, following the trend of companies being increasingly more secretive about their models, many have stopped reporting their models' perplexity.",[245,2576,2577],{},[152,2578,2579,2580],{},"Table 3-1. Larger GPT-2 models consistently give lower perplexity on different datasets. ",[130,2581,2584],{"href":2582,"rel":2583},"https:\u002F\u002Fcdn.openai.com\u002Fbetter-language-models\u002Flanguage_models_are_unsupervised_multitask_learners.pdf",[134],"Source: OpenAI, 2018.",[2586,2587,2588,2629],"table",{},[2589,2590,2591],"thead",{},[2592,2593,2594,2599,2602,2605,2608,2611,2614,2617,2620,2623,2626],"tr",{},[2595,2596,2598],"th",{"align":2597},"left","Model",[2595,2600,2601],{"align":2597},"LAMBADA (PPL)",[2595,2603,2604],{"align":2597},"LAMBADA (ACC)",[2595,2606,2607],{"align":2597},"CBT-CN (ACC)",[2595,2609,2610],{"align":2597},"CBT-NE (ACC)",[2595,2612,2613],{"align":2597},"WikiText2 (PPL)",[2595,2615,2616],{"align":2597},"PTB (PPL)",[2595,2618,2619],{"align":2597},"enwiki8 (BPB)",[2595,2621,2622],{"align":2597},"text8 (BPC)",[2595,2624,2625],{"align":2597},"WikiText103 (PPL)",[2595,2627,2628],{"align":2597},"1BW (PPL)",[2630,2631,2632,2670,2707,2744,2781],"tbody",{},[2592,2633,2634,2640,2643,2646,2649,2652,2655,2658,2661,2664,2667],{},[2635,2636,2637],"td",{"align":2597},[138,2638,2639],{},"SOTA",[2635,2641,2642],{"align":2597},"99.8",[2635,2644,2645],{"align":2597},"59.23",[2635,2647,2648],{"align":2597},"85.7",[2635,2650,2651],{"align":2597},"82.3",[2635,2653,2654],{"align":2597},"39.14",[2635,2656,2657],{"align":2597},"46.54",[2635,2659,2660],{"align":2597},"0.99",[2635,2662,2663],{"align":2597},"1.08",[2635,2665,2666],{"align":2597},"18.3",[2635,2668,2669],{"align":2597},"21.8",[2592,2671,2672,2677,2680,2683,2686,2689,2692,2695,2698,2701,2704],{},[2635,2673,2674],{"align":2597},[138,2675,2676],{},"117M",[2635,2678,2679],{"align":2597},"35.13",[2635,2681,2682],{"align":2597},"45.99",[2635,2684,2685],{"align":2597},"87.65",[2635,2687,2688],{"align":2597},"83.4",[2635,2690,2691],{"align":2597},"29.41",[2635,2693,2694],{"align":2597},"65.85",[2635,2696,2697],{"align":2597},"1.16",[2635,2699,2700],{"align":2597},"1.17",[2635,2702,2703],{"align":2597},"37.50",[2635,2705,2706],{"align":2597},"75.20",[2592,2708,2709,2714,2717,2720,2723,2726,2729,2732,2735,2738,2741],{},[2635,2710,2711],{"align":2597},[138,2712,2713],{},"345M",[2635,2715,2716],{"align":2597},"15.60",[2635,2718,2719],{"align":2597},"55.48",[2635,2721,2722],{"align":2597},"92.35",[2635,2724,2725],{"align":2597},"87.1",[2635,2727,2728],{"align":2597},"22.76",[2635,2730,2731],{"align":2597},"47.33",[2635,2733,2734],{"align":2597},"1.01",[2635,2736,2737],{"align":2597},"1.06",[2635,2739,2740],{"align":2597},"26.37",[2635,2742,2743],{"align":2597},"55.72",[2592,2745,2746,2751,2754,2757,2760,2763,2766,2769,2772,2775,2778],{},[2635,2747,2748],{"align":2597},[138,2749,2750],{},"762M",[2635,2752,2753],{"align":2597},"10.87",[2635,2755,2756],{"align":2597},"60.12",[2635,2758,2759],{"align":2597},"93.45",[2635,2761,2762],{"align":2597},"88.0",[2635,2764,2765],{"align":2597},"19.93",[2635,2767,2768],{"align":2597},"40.31",[2635,2770,2771],{"align":2597},"0.97",[2635,2773,2774],{"align":2597},"1.02",[2635,2776,2777],{"align":2597},"22.05",[2635,2779,2780],{"align":2597},"44.575",[2592,2782,2783,2788,2791,2794,2797,2800,2803,2806,2809,2812,2815],{},[2635,2784,2785],{"align":2597},[138,2786,2787],{},"1542M",[2635,2789,2790],{"align":2597},"8.63",[2635,2792,2793],{"align":2597},"63.24",[2635,2795,2796],{"align":2597},"93.30",[2635,2798,2799],{"align":2597},"89.05",[2635,2801,2802],{"align":2597},"18.34",[2635,2804,2805],{"align":2597},"35.76",[2635,2807,2808],{"align":2597},"0.93",[2635,2810,2811],{"align":2597},"0.98",[2635,2813,2814],{"align":2597},"17.48",[2635,2816,2817],{"align":2597},"42.16",[143,2819,2820,2827],{},[152,2821,2822,2823,2826],{},"Perplexity might not be a great proxy to evaluate models that have been post-trained using techniques like SFT and RLHF. Post-training is about teaching models how to complete tasks. As a model gets better at completing tasks, it might get worse at predicting the next tokens. A language model's perplexity typically increases after post-training. Some people say that post-training ",[230,2824,2825],{},"collapses"," entropy. Similarly, quantization — a technique that reduces a model's numerical precision and, with it, its memory footprint — can also change a model's perplexity in unexpected ways.",[152,2828,2829],{},"If you're unsure what SFT (supervised finetuning) and RLHF (reinforcement learning from human feedback) mean, revisit Chapter 2. Quantization is discussed in Chapter 7.",[152,2831,2832],{},"Recall that the perplexity of a model with respect to a text measures how difficult it is for this model to predict this text. For a given model, perplexity is the lowest for texts that the model has seen and memorized during training. Therefore, perplexity can be used to detect whether a text was in a model's training data.",[159,2834,2835,2839,2847],{},[162,2836,2838],{"icon":90,"title":2837},"Data Contamination","If a model's perplexity on a benchmark's data is low, this benchmark was likely included in the model's training data, making the model's performance on this benchmark less trustworthy.",[162,2840,2843,2844,141],{"icon":2841,"title":2842},"i-lucide-files","Training Data Deduplication","Perplexity can also be used for deduplication of training data: e.g., add new data to the existing training dataset only if the perplexity of the new data is ",[138,2845,2846],{},"high",[162,2848,2851],{"icon":2849,"title":2850},"i-lucide-scan-search","Abnormal Text Detection","Perplexity is the highest for unpredictable texts, such as texts expressing unusual ideas (like \"my dog teaches quantum physics in his free time\") or gibberish (like \"home cat go eye\"). Therefore, perplexity can be used to detect abnormal texts.",[152,2853,2854],{},"Perplexity and its related metrics help us understand the performance of the underlying language model, which is a proxy for understanding the model's performance on downstream tasks. The rest of the chapter discusses how to measure a model's performance on downstream tasks directly.",[147,2856,2858],{"id":2857},"how-to-use-a-language-model-to-compute-a-texts-perplexity","How to Use a Language Model to Compute a Text's Perplexity",[152,2860,2861,2862,2892,2893,3105,3106,3134],{},"A model's perplexity with respect to a text measures how difficult it is for the model to predict that text. Given a language model ",[316,2863,2865,2879],{"className":2864},[319],[316,2866,2868],{"className":2867},[323],[325,2869,2870],{"xmlns":327},[329,2871,2872,2877],{},[332,2873,2874],{},[335,2875,2876],{},"X",[339,2878,2876],{"encoding":341},[316,2880,2882],{"className":2881,"ariaHidden":346},[345],[316,2883,2885,2888],{"className":2884},[350],[316,2886],{"className":2887,"style":355},[354],[316,2889,2876],{"className":2890,"style":2891},[359,360],"margin-right:0.0785em;",", and a sequence of tokens ",[316,2894,2896,2942],{"className":2895},[319],[316,2897,2899],{"className":2898},[323],[325,2900,2901],{"xmlns":327},[329,2902,2903,2939],{},[332,2904,2905,2908,2914,2916,2922,2924,2927,2929,2936],{},[447,2906,2907],{"stretchy":449},"[",[557,2909,2910,2912],{},[335,2911,2267],{},[1199,2913,273],{},[447,2915,708],{"separator":346},[557,2917,2918,2920],{},[335,2919,2267],{},[1199,2921,281],{},[447,2923,708],{"separator":346},[447,2925,2926],{},"…",[447,2928,708],{"separator":346},[557,2930,2931,2933],{},[335,2932,2267],{},[335,2934,2935],{},"n",[447,2937,2938],{"stretchy":449},"]",[339,2940,2941],{"encoding":341},"[x_1, x_2, \\dots, x_n]",[316,2943,2945],{"className":2944,"ariaHidden":346},[345],[316,2946,2948,2951,2954,2996,2999,3002,3042,3045,3048,3052,3055,3058,3061,3102],{"className":2947},[350],[316,2949],{"className":2950,"style":468},[354],[316,2952,2907],{"className":2953},[475],[316,2955,2957,2960],{"className":2956},[359],[316,2958,2267],{"className":2959},[359,360],[316,2961,2963],{"className":2962},[599],[316,2964,2966,2988],{"className":2965},[603,604],[316,2967,2969,2985],{"className":2968},[608],[316,2970,2973],{"className":2971,"style":2972},[612],"height:0.3011em;",[316,2974,2976,2979],{"style":2975},"top:-2.55em;margin-left:0em;margin-right:0.05em;",[316,2977],{"className":2978,"style":621},[620],[316,2980,2982],{"className":2981},[625,626,627,628],[316,2983,273],{"className":2984},[359,628],[316,2986,643],{"className":2987},[642],[316,2989,2991],{"className":2990},[608],[316,2992,2994],{"className":2993,"style":650},[612],[316,2995],{},[316,2997,708],{"className":2998},[767],[316,3000],{"className":3001,"style":771},[662],[316,3003,3005,3008],{"className":3004},[359],[316,3006,2267],{"className":3007},[359,360],[316,3009,3011],{"className":3010},[599],[316,3012,3014,3034],{"className":3013},[603,604],[316,3015,3017,3031],{"className":3016},[608],[316,3018,3020],{"className":3019,"style":2972},[612],[316,3021,3022,3025],{"style":2975},[316,3023],{"className":3024,"style":621},[620],[316,3026,3028],{"className":3027},[625,626,627,628],[316,3029,281],{"className":3030},[359,628],[316,3032,643],{"className":3033},[642],[316,3035,3037],{"className":3036},[608],[316,3038,3040],{"className":3039,"style":650},[612],[316,3041],{},[316,3043,708],{"className":3044},[767],[316,3046],{"className":3047,"style":771},[662],[316,3049,2926],{"className":3050},[3051],"minner",[316,3053],{"className":3054,"style":771},[662],[316,3056,708],{"className":3057},[767],[316,3059],{"className":3060,"style":771},[662],[316,3062,3064,3067],{"className":3063},[359],[316,3065,2267],{"className":3066},[359,360],[316,3068,3070],{"className":3069},[599],[316,3071,3073,3094],{"className":3072},[603,604],[316,3074,3076,3091],{"className":3075},[608],[316,3077,3080],{"className":3078,"style":3079},[612],"height:0.1514em;",[316,3081,3082,3085],{"style":2975},[316,3083],{"className":3084,"style":621},[620],[316,3086,3088],{"className":3087},[625,626,627,628],[316,3089,2935],{"className":3090},[359,360,628],[316,3092,643],{"className":3093},[642],[316,3095,3097],{"className":3096},[608],[316,3098,3100],{"className":3099,"style":650},[612],[316,3101],{},[316,3103,2938],{"className":3104},[482],", ",[316,3107,3109,3122],{"className":3108},[319],[316,3110,3112],{"className":3111},[323],[325,3113,3114],{"xmlns":327},[329,3115,3116,3120],{},[332,3117,3118],{},[335,3119,2876],{},[339,3121,2876],{"encoding":341},[316,3123,3125],{"className":3124,"ariaHidden":346},[345],[316,3126,3128,3131],{"className":3127},[350],[316,3129],{"className":3130,"style":355},[354],[316,3132,2876],{"className":3133,"style":2891},[359,360],"'s perplexity for this sequence is:",[152,3136,3137],{},[316,3138,3140,3330],{"className":3139},[319],[316,3141,3143],{"className":3142},[323],[325,3144,3145],{"xmlns":327},[329,3146,3147,3327],{},[332,3148,3149,3151,3153,3159,3161,3167,3169,3171,3173,3179,3195,3197,3249,3251],{},[335,3150,377],{},[447,3152,450],{"stretchy":449},[557,3154,3155,3157],{},[335,3156,2267],{},[1199,3158,273],{},[447,3160,708],{"separator":346},[557,3162,3163,3165],{},[335,3164,2267],{},[1199,3166,281],{},[447,3168,708],{"separator":346},[447,3170,2926],{},[447,3172,708],{"separator":346},[557,3174,3175,3177],{},[335,3176,2267],{},[335,3178,2935],{},[1824,3180,3181,3183],{},[447,3182,455],{"stretchy":449},[332,3184,3185,3188],{},[447,3186,3187],{},"−",[3189,3190,3191,3193],"mfrac",{},[1199,3192,273],{},[335,3194,2935],{},[447,3196,715],{},[1824,3198,3199,3243],{},[332,3200,3201,3203,3241],{},[447,3202,450],{"fence":346},[3189,3204,3205,3207],{},[1199,3206,273],{},[332,3208,3209,3211,3213,3219,3221,3227,3229,3231,3233,3239],{},[335,3210,377],{},[447,3212,450],{"stretchy":449},[557,3214,3215,3217],{},[335,3216,2267],{},[1199,3218,273],{},[447,3220,708],{"separator":346},[557,3222,3223,3225],{},[335,3224,2267],{},[1199,3226,281],{},[447,3228,708],{"separator":346},[447,3230,2926],{},[447,3232,708],{"separator":346},[557,3234,3235,3237],{},[335,3236,2267],{},[335,3238,2935],{},[447,3240,455],{"stretchy":449},[447,3242,455],{"fence":346},[3189,3244,3245,3247],{},[1199,3246,273],{},[335,3248,2935],{},[447,3250,715],{},[1824,3252,3253,3321],{},[332,3254,3255,3257,3274,3319],{},[447,3256,450],{"fence":346},[3258,3259,3260,3263,3272],"msubsup",{},[447,3261,3262],{},"∏",[332,3264,3265,3268,3270],{},[335,3266,3267],{},"i",[447,3269,715],{},[1199,3271,273],{},[335,3273,2935],{},[3189,3275,3276,3278],{},[1199,3277,273],{},[332,3279,3280,3282,3284,3290,3293,3299,3301,3303,3305,3317],{},[335,3281,377],{},[447,3283,450],{"stretchy":449},[557,3285,3286,3288],{},[335,3287,2267],{},[335,3289,3267],{},[447,3291,3292],{},"∣",[557,3294,3295,3297],{},[335,3296,2267],{},[1199,3298,273],{},[447,3300,708],{"separator":346},[447,3302,2926],{},[447,3304,708],{"separator":346},[557,3306,3307,3309],{},[335,3308,2267],{},[332,3310,3311,3313,3315],{},[335,3312,3267],{},[447,3314,3187],{},[1199,3316,273],{},[447,3318,455],{"stretchy":449},[447,3320,455],{"fence":346},[3189,3322,3323,3325],{},[1199,3324,273],{},[335,3326,2935],{},[339,3328,3329],{"encoding":341},"P(x_1, x_2, \\dots, x_n)^{-\\frac{1}{n}} = \\left( \\frac{1}{P(x_1, x_2, \\dots, x_n)} \\right)^{\\frac{1}{n}} = \\left( \\prod_{i=1}^n \\frac{1}{P(x_i \\mid x_1, \\dots, x_{i-1})} \\right)^{\\frac{1}{n}}",[316,3331,3333,3613,3961],{"className":3332,"ariaHidden":346},[345],[316,3334,3336,3340,3343,3346,3386,3389,3392,3432,3435,3438,3441,3444,3447,3450,3490,3604,3607,3610],{"className":3335},[350],[316,3337],{"className":3338,"style":3339},[354],"height:1.204em;vertical-align:-0.25em;",[316,3341,377],{"className":3342,"style":392},[359,360],[316,3344,450],{"className":3345},[475],[316,3347,3349,3352],{"className":3348},[359],[316,3350,2267],{"className":3351},[359,360],[316,3353,3355],{"className":3354},[599],[316,3356,3358,3378],{"className":3357},[603,604],[316,3359,3361,3375],{"className":3360},[608],[316,3362,3364],{"className":3363,"style":2972},[612],[316,3365,3366,3369],{"style":2975},[316,3367],{"className":3368,"style":621},[620],[316,3370,3372],{"className":3371},[625,626,627,628],[316,3373,273],{"className":3374},[359,628],[316,3376,643],{"className":3377},[642],[316,3379,3381],{"className":3380},[608],[316,3382,3384],{"className":3383,"style":650},[612],[316,3385],{},[316,3387,708],{"className":3388},[767],[316,3390],{"className":3391,"style":771},[662],[316,3393,3395,3398],{"className":3394},[359],[316,3396,2267],{"className":3397},[359,360],[316,3399,3401],{"className":3400},[599],[316,3402,3404,3424],{"className":3403},[603,604],[316,3405,3407,3421],{"className":3406},[608],[316,3408,3410],{"className":3409,"style":2972},[612],[316,3411,3412,3415],{"style":2975},[316,3413],{"className":3414,"style":621},[620],[316,3416,3418],{"className":3417},[625,626,627,628],[316,3419,281],{"className":3420},[359,628],[316,3422,643],{"className":3423},[642],[316,3425,3427],{"className":3426},[608],[316,3428,3430],{"className":3429,"style":650},[612],[316,3431],{},[316,3433,708],{"className":3434},[767],[316,3436],{"className":3437,"style":771},[662],[316,3439,2926],{"className":3440},[3051],[316,3442],{"className":3443,"style":771},[662],[316,3445,708],{"className":3446},[767],[316,3448],{"className":3449,"style":771},[662],[316,3451,3453,3456],{"className":3452},[359],[316,3454,2267],{"className":3455},[359,360],[316,3457,3459],{"className":3458},[599],[316,3460,3462,3482],{"className":3461},[603,604],[316,3463,3465,3479],{"className":3464},[608],[316,3466,3468],{"className":3467,"style":3079},[612],[316,3469,3470,3473],{"style":2975},[316,3471],{"className":3472,"style":621},[620],[316,3474,3476],{"className":3475},[625,626,627,628],[316,3477,2935],{"className":3478},[359,360,628],[316,3480,643],{"className":3481},[642],[316,3483,3485],{"className":3484},[608],[316,3486,3488],{"className":3487,"style":650},[612],[316,3489],{},[316,3491,3493,3496],{"className":3492},[482],[316,3494,455],{"className":3495},[482],[316,3497,3499],{"className":3498},[599],[316,3500,3502],{"className":3501},[603],[316,3503,3505],{"className":3504},[608],[316,3506,3509],{"className":3507,"style":3508},[612],"height:0.954em;",[316,3510,3512,3516],{"style":3511},"top:-3.363em;margin-right:0.05em;",[316,3513],{"className":3514,"style":3515},[620],"height:3em;",[316,3517,3519],{"className":3518},[625,626,627,628],[316,3520,3522,3525],{"className":3521},[359,628],[316,3523,3187],{"className":3524},[359,628],[316,3526,3528,3534,3601],{"className":3527},[359,628],[316,3529],{"className":3530},[475,3531,625,3532,3533],"nulldelimiter","reset-size3","size6",[316,3535,3537],{"className":3536},[3189],[316,3538,3540,3592],{"className":3539},[603,604],[316,3541,3543,3589],{"className":3542},[608],[316,3544,3547,3563,3574],{"className":3545,"style":3546},[612],"height:0.8443em;",[316,3548,3550,3553],{"style":3549},"top:-2.656em;",[316,3551],{"className":3552,"style":3515},[620],[316,3554,3557],{"className":3555},[625,3532,3556,628],"size1",[316,3558,3560],{"className":3559},[359,628],[316,3561,2935],{"className":3562},[359,360,628],[316,3564,3566,3569],{"style":3565},"top:-3.2255em;",[316,3567],{"className":3568,"style":3515},[620],[316,3570],{"className":3571,"style":3573},[3572,628],"frac-line","border-bottom-width:0.049em;",[316,3575,3577,3580],{"style":3576},"top:-3.384em;",[316,3578],{"className":3579,"style":3515},[620],[316,3581,3583],{"className":3582},[625,3532,3556,628],[316,3584,3586],{"className":3585},[359,628],[316,3587,273],{"className":3588},[359,628],[316,3590,643],{"className":3591},[642],[316,3593,3595],{"className":3594},[608],[316,3596,3599],{"className":3597,"style":3598},[612],"height:0.344em;",[316,3600],{},[316,3602],{"className":3603},[482,3531,625,3532,3533],[316,3605],{"className":3606,"style":663},[662],[316,3608,715],{"className":3609},[667],[316,3611],{"className":3612,"style":663},[662],[316,3614,3616,3620,3952,3955,3958],{"className":3615},[350],[316,3617],{"className":3618,"style":3619},[354],"height:2.1439em;vertical-align:-0.65em;",[316,3621,3623,3859],{"className":3622},[3051],[316,3624,3626,3636,3853],{"className":3625},[3051],[316,3627,3631],{"className":3628,"style":3630},[475,3629],"delimcenter","top:0em;",[316,3632,450],{"className":3633},[3634,3635],"delimsizing","size2",[316,3637,3639,3642,3850],{"className":3638},[359],[316,3640],{"className":3641},[475,3531],[316,3643,3645],{"className":3644},[3189],[316,3646,3648,3841],{"className":3647},[603,604],[316,3649,3651,3838],{"className":3650},[608],[316,3652,3655,3813,3823],{"className":3653,"style":3654},[612],"height:0.8451em;",[316,3656,3658,3661],{"style":3657},"top:-2.655em;",[316,3659],{"className":3660,"style":3515},[620],[316,3662,3664],{"className":3663},[625,626,627,628],[316,3665,3667,3670,3673,3717,3720,3760,3763,3766,3769,3810],{"className":3666},[359,628],[316,3668,377],{"className":3669,"style":392},[359,360,628],[316,3671,450],{"className":3672},[475,628],[316,3674,3676,3679],{"className":3675},[359,628],[316,3677,2267],{"className":3678},[359,360,628],[316,3680,3682],{"className":3681},[599],[316,3683,3685,3708],{"className":3684},[603,604],[316,3686,3688,3705],{"className":3687},[608],[316,3689,3692],{"className":3690,"style":3691},[612],"height:0.3173em;",[316,3693,3695,3699],{"style":3694},"top:-2.357em;margin-left:0em;margin-right:0.0714em;",[316,3696],{"className":3697,"style":3698},[620],"height:2.5em;",[316,3700,3702],{"className":3701},[625,3532,3556,628],[316,3703,273],{"className":3704},[359,628],[316,3706,643],{"className":3707},[642],[316,3709,3711],{"className":3710},[608],[316,3712,3715],{"className":3713,"style":3714},[612],"height:0.143em;",[316,3716],{},[316,3718,708],{"className":3719},[767,628],[316,3721,3723,3726],{"className":3722},[359,628],[316,3724,2267],{"className":3725},[359,360,628],[316,3727,3729],{"className":3728},[599],[316,3730,3732,3752],{"className":3731},[603,604],[316,3733,3735,3749],{"className":3734},[608],[316,3736,3738],{"className":3737,"style":3691},[612],[316,3739,3740,3743],{"style":3694},[316,3741],{"className":3742,"style":3698},[620],[316,3744,3746],{"className":3745},[625,3532,3556,628],[316,3747,281],{"className":3748},[359,628],[316,3750,643],{"className":3751},[642],[316,3753,3755],{"className":3754},[608],[316,3756,3758],{"className":3757,"style":3714},[612],[316,3759],{},[316,3761,708],{"className":3762},[767,628],[316,3764,2926],{"className":3765},[3051,628],[316,3767,708],{"className":3768},[767,628],[316,3770,3772,3775],{"className":3771},[359,628],[316,3773,2267],{"className":3774},[359,360,628],[316,3776,3778],{"className":3777},[599],[316,3779,3781,3802],{"className":3780},[603,604],[316,3782,3784,3799],{"className":3783},[608],[316,3785,3788],{"className":3786,"style":3787},[612],"height:0.1645em;",[316,3789,3790,3793],{"style":3694},[316,3791],{"className":3792,"style":3698},[620],[316,3794,3796],{"className":3795},[625,3532,3556,628],[316,3797,2935],{"className":3798},[359,360,628],[316,3800,643],{"className":3801},[642],[316,3803,3805],{"className":3804},[608],[316,3806,3808],{"className":3807,"style":3714},[612],[316,3809],{},[316,3811,455],{"className":3812},[482,628],[316,3814,3816,3819],{"style":3815},"top:-3.23em;",[316,3817],{"className":3818,"style":3515},[620],[316,3820],{"className":3821,"style":3822},[3572],"border-bottom-width:0.04em;",[316,3824,3826,3829],{"style":3825},"top:-3.394em;",[316,3827],{"className":3828,"style":3515},[620],[316,3830,3832],{"className":3831},[625,626,627,628],[316,3833,3835],{"className":3834},[359,628],[316,3836,273],{"className":3837},[359,628],[316,3839,643],{"className":3840},[642],[316,3842,3844],{"className":3843},[608],[316,3845,3848],{"className":3846,"style":3847},[612],"height:0.52em;",[316,3849],{},[316,3851],{"className":3852},[482,3531],[316,3854,3856],{"className":3855,"style":3630},[482,3629],[316,3857,455],{"className":3858},[3634,3635],[316,3860,3862],{"className":3861},[599],[316,3863,3865],{"className":3864},[603],[316,3866,3868],{"className":3867},[608],[316,3869,3872],{"className":3870,"style":3871},[612],"height:1.4939em;",[316,3873,3875,3878],{"style":3874},"top:-3.9029em;margin-right:0.05em;",[316,3876],{"className":3877,"style":3515},[620],[316,3879,3881],{"className":3880},[625,626,627,628],[316,3882,3884],{"className":3883},[359,628],[316,3885,3887,3890,3949],{"className":3886},[359,628],[316,3888],{"className":3889},[475,3531,625,3532,3533],[316,3891,3893],{"className":3892},[3189],[316,3894,3896,3941],{"className":3895},[603,604],[316,3897,3899,3938],{"className":3898},[608],[316,3900,3902,3916,3924],{"className":3901,"style":3546},[612],[316,3903,3904,3907],{"style":3549},[316,3905],{"className":3906,"style":3515},[620],[316,3908,3910],{"className":3909},[625,3532,3556,628],[316,3911,3913],{"className":3912},[359,628],[316,3914,2935],{"className":3915},[359,360,628],[316,3917,3918,3921],{"style":3565},[316,3919],{"className":3920,"style":3515},[620],[316,3922],{"className":3923,"style":3573},[3572,628],[316,3925,3926,3929],{"style":3576},[316,3927],{"className":3928,"style":3515},[620],[316,3930,3932],{"className":3931},[625,3532,3556,628],[316,3933,3935],{"className":3934},[359,628],[316,3936,273],{"className":3937},[359,628],[316,3939,643],{"className":3940},[642],[316,3942,3944],{"className":3943},[608],[316,3945,3947],{"className":3946,"style":3598},[612],[316,3948],{},[316,3950],{"className":3951},[482,3531,625,3532,3533],[316,3953],{"className":3954,"style":663},[662],[316,3956,715],{"className":3957},[667],[316,3959],{"className":3960,"style":663},[662],[316,3962,3964,3967],{"className":3963},[350],[316,3965],{"className":3966,"style":3619},[354],[316,3968,3970,4272],{"className":3969},[3051],[316,3971,3973,3979,4046,4049,4266],{"className":3972},[3051],[316,3974,3976],{"className":3975,"style":3630},[475,3629],[316,3977,450],{"className":3978},[3634,3635],[316,3980,3982,3988],{"className":3981},[2285],[316,3983,3262],{"className":3984,"style":3987},[2285,3985,3986],"op-symbol","small-op","position:relative;top:0em;",[316,3989,3991],{"className":3990},[599],[316,3992,3994,4037],{"className":3993},[603,604],[316,3995,3997,4034],{"className":3996},[608],[316,3998,4001,4022],{"className":3999,"style":4000},[612],"height:0.8043em;",[316,4002,4004,4007],{"style":4003},"top:-2.4003em;margin-left:0em;margin-right:0.05em;",[316,4005],{"className":4006,"style":621},[620],[316,4008,4010],{"className":4009},[625,626,627,628],[316,4011,4013,4016,4019],{"className":4012},[359,628],[316,4014,3267],{"className":4015},[359,360,628],[316,4017,715],{"className":4018},[667,628],[316,4020,273],{"className":4021},[359,628],[316,4023,4025,4028],{"style":4024},"top:-3.2029em;margin-right:0.05em;",[316,4026],{"className":4027,"style":621},[620],[316,4029,4031],{"className":4030},[625,626,627,628],[316,4032,2935],{"className":4033},[359,360,628],[316,4035,643],{"className":4036},[642],[316,4038,4040],{"className":4039},[608],[316,4041,4044],{"className":4042,"style":4043},[612],"height:0.2997em;",[316,4045],{},[316,4047],{"className":4048,"style":771},[662],[316,4050,4052,4055,4263],{"className":4051},[359],[316,4053],{"className":4054},[475,3531],[316,4056,4058],{"className":4057},[3189],[316,4059,4061,4255],{"className":4060},[603,604],[316,4062,4064,4252],{"className":4063},[608],[316,4065,4067,4230,4238],{"className":4066,"style":3654},[612],[316,4068,4069,4072],{"style":3657},[316,4070],{"className":4071,"style":3515},[620],[316,4073,4075],{"className":4074},[625,626,627,628],[316,4076,4078,4081,4084,4125,4128,4168,4171,4174,4177,4227],{"className":4077},[359,628],[316,4079,377],{"className":4080,"style":392},[359,360,628],[316,4082,450],{"className":4083},[475,628],[316,4085,4087,4090],{"className":4086},[359,628],[316,4088,2267],{"className":4089},[359,360,628],[316,4091,4093],{"className":4092},[599],[316,4094,4096,4117],{"className":4095},[603,604],[316,4097,4099,4114],{"className":4098},[608],[316,4100,4103],{"className":4101,"style":4102},[612],"height:0.3281em;",[316,4104,4105,4108],{"style":3694},[316,4106],{"className":4107,"style":3698},[620],[316,4109,4111],{"className":4110},[625,3532,3556,628],[316,4112,3267],{"className":4113},[359,360,628],[316,4115,643],{"className":4116},[642],[316,4118,4120],{"className":4119},[608],[316,4121,4123],{"className":4122,"style":3714},[612],[316,4124],{},[316,4126,3292],{"className":4127},[667,628],[316,4129,4131,4134],{"className":4130},[359,628],[316,4132,2267],{"className":4133},[359,360,628],[316,4135,4137],{"className":4136},[599],[316,4138,4140,4160],{"className":4139},[603,604],[316,4141,4143,4157],{"className":4142},[608],[316,4144,4146],{"className":4145,"style":3691},[612],[316,4147,4148,4151],{"style":3694},[316,4149],{"className":4150,"style":3698},[620],[316,4152,4154],{"className":4153},[625,3532,3556,628],[316,4155,273],{"className":4156},[359,628],[316,4158,643],{"className":4159},[642],[316,4161,4163],{"className":4162},[608],[316,4164,4166],{"className":4165,"style":3714},[612],[316,4167],{},[316,4169,708],{"className":4170},[767,628],[316,4172,2926],{"className":4173},[3051,628],[316,4175,708],{"className":4176},[767,628],[316,4178,4180,4183],{"className":4179},[359,628],[316,4181,2267],{"className":4182},[359,360,628],[316,4184,4186],{"className":4185},[599],[316,4187,4189,4218],{"className":4188},[603,604],[316,4190,4192,4215],{"className":4191},[608],[316,4193,4195],{"className":4194,"style":4102},[612],[316,4196,4197,4200],{"style":3694},[316,4198],{"className":4199,"style":3698},[620],[316,4201,4203],{"className":4202},[625,3532,3556,628],[316,4204,4206,4209,4212],{"className":4205},[359,628],[316,4207,3267],{"className":4208},[359,360,628],[316,4210,3187],{"className":4211},[812,628],[316,4213,273],{"className":4214},[359,628],[316,4216,643],{"className":4217},[642],[316,4219,4221],{"className":4220},[608],[316,4222,4225],{"className":4223,"style":4224},[612],"height:0.2025em;",[316,4226],{},[316,4228,455],{"className":4229},[482,628],[316,4231,4232,4235],{"style":3815},[316,4233],{"className":4234,"style":3515},[620],[316,4236],{"className":4237,"style":3822},[3572],[316,4239,4240,4243],{"style":3825},[316,4241],{"className":4242,"style":3515},[620],[316,4244,4246],{"className":4245},[625,626,627,628],[316,4247,4249],{"className":4248},[359,628],[316,4250,273],{"className":4251},[359,628],[316,4253,643],{"className":4254},[642],[316,4256,4258],{"className":4257},[608],[316,4259,4261],{"className":4260,"style":3847},[612],[316,4262],{},[316,4264],{"className":4265},[482,3531],[316,4267,4269],{"className":4268,"style":3630},[482,3629],[316,4270,455],{"className":4271},[3634,3635],[316,4273,4275],{"className":4274},[599],[316,4276,4278],{"className":4277},[603],[316,4279,4281],{"className":4280},[608],[316,4282,4284],{"className":4283,"style":3871},[612],[316,4285,4286,4289],{"style":3874},[316,4287],{"className":4288,"style":3515},[620],[316,4290,4292],{"className":4291},[625,626,627,628],[316,4293,4295],{"className":4294},[359,628],[316,4296,4298,4301,4360],{"className":4297},[359,628],[316,4299],{"className":4300},[475,3531,625,3532,3533],[316,4302,4304],{"className":4303},[3189],[316,4305,4307,4352],{"className":4306},[603,604],[316,4308,4310,4349],{"className":4309},[608],[316,4311,4313,4327,4335],{"className":4312,"style":3546},[612],[316,4314,4315,4318],{"style":3549},[316,4316],{"className":4317,"style":3515},[620],[316,4319,4321],{"className":4320},[625,3532,3556,628],[316,4322,4324],{"className":4323},[359,628],[316,4325,2935],{"className":4326},[359,360,628],[316,4328,4329,4332],{"style":3565},[316,4330],{"className":4331,"style":3515},[620],[316,4333],{"className":4334,"style":3573},[3572,628],[316,4336,4337,4340],{"style":3576},[316,4338],{"className":4339,"style":3515},[620],[316,4341,4343],{"className":4342},[625,3532,3556,628],[316,4344,4346],{"className":4345},[359,628],[316,4347,273],{"className":4348},[359,628],[316,4350,643],{"className":4351},[642],[316,4353,4355],{"className":4354},[608],[316,4356,4358],{"className":4357,"style":3598},[612],[316,4359],{},[316,4361],{"className":4362},[482,3531,625,3532,3533],[152,4364,4365,4366,4601,4602,4630,4631,4702,4703,141],{},"where ",[316,4367,4369,4419],{"className":4368},[319],[316,4370,4372],{"className":4371},[323],[325,4373,4374],{"xmlns":327},[329,4375,4376,4416],{},[332,4377,4378,4380,4382,4388,4390,4396,4398,4400,4402,4414],{},[335,4379,377],{},[447,4381,450],{"stretchy":449},[557,4383,4384,4386],{},[335,4385,2267],{},[335,4387,3267],{},[447,4389,3292],{},[557,4391,4392,4394],{},[335,4393,2267],{},[1199,4395,273],{},[447,4397,708],{"separator":346},[447,4399,2926],{},[447,4401,708],{"separator":346},[557,4403,4404,4406],{},[335,4405,2267],{},[332,4407,4408,4410,4412],{},[335,4409,3267],{},[447,4411,3187],{},[1199,4413,273],{},[447,4415,455],{"stretchy":449},[339,4417,4418],{"encoding":341},"P(x_i \\mid x_1, \\dots, x_{i-1})",[316,4420,4422,4484],{"className":4421,"ariaHidden":346},[345],[316,4423,4425,4428,4431,4434,4475,4478,4481],{"className":4424},[350],[316,4426],{"className":4427,"style":468},[354],[316,4429,377],{"className":4430,"style":392},[359,360],[316,4432,450],{"className":4433},[475],[316,4435,4437,4440],{"className":4436},[359],[316,4438,2267],{"className":4439},[359,360],[316,4441,4443],{"className":4442},[599],[316,4444,4446,4467],{"className":4445},[603,604],[316,4447,4449,4464],{"className":4448},[608],[316,4450,4453],{"className":4451,"style":4452},[612],"height:0.3117em;",[316,4454,4455,4458],{"style":2975},[316,4456],{"className":4457,"style":621},[620],[316,4459,4461],{"className":4460},[625,626,627,628],[316,4462,3267],{"className":4463},[359,360,628],[316,4465,643],{"className":4466},[642],[316,4468,4470],{"className":4469},[608],[316,4471,4473],{"className":4472,"style":650},[612],[316,4474],{},[316,4476],{"className":4477,"style":663},[662],[316,4479,3292],{"className":4480},[667],[316,4482],{"className":4483,"style":663},[662],[316,4485,4487,4490,4530,4533,4536,4539,4542,4545,4548,4598],{"className":4486},[350],[316,4488],{"className":4489,"style":468},[354],[316,4491,4493,4496],{"className":4492},[359],[316,4494,2267],{"className":4495},[359,360],[316,4497,4499],{"className":4498},[599],[316,4500,4502,4522],{"className":4501},[603,604],[316,4503,4505,4519],{"className":4504},[608],[316,4506,4508],{"className":4507,"style":2972},[612],[316,4509,4510,4513],{"style":2975},[316,4511],{"className":4512,"style":621},[620],[316,4514,4516],{"className":4515},[625,626,627,628],[316,4517,273],{"className":4518},[359,628],[316,4520,643],{"className":4521},[642],[316,4523,4525],{"className":4524},[608],[316,4526,4528],{"className":4527,"style":650},[612],[316,4529],{},[316,4531,708],{"className":4532},[767],[316,4534],{"className":4535,"style":771},[662],[316,4537,2926],{"className":4538},[3051],[316,4540],{"className":4541,"style":771},[662],[316,4543,708],{"className":4544},[767],[316,4546],{"className":4547,"style":771},[662],[316,4549,4551,4554],{"className":4550},[359],[316,4552,2267],{"className":4553},[359,360],[316,4555,4557],{"className":4556},[599],[316,4558,4560,4589],{"className":4559},[603,604],[316,4561,4563,4586],{"className":4562},[608],[316,4564,4566],{"className":4565,"style":4452},[612],[316,4567,4568,4571],{"style":2975},[316,4569],{"className":4570,"style":621},[620],[316,4572,4574],{"className":4573},[625,626,627,628],[316,4575,4577,4580,4583],{"className":4576},[359,628],[316,4578,3267],{"className":4579},[359,360,628],[316,4581,3187],{"className":4582},[812,628],[316,4584,273],{"className":4585},[359,628],[316,4587,643],{"className":4588},[642],[316,4590,4592],{"className":4591},[608],[316,4593,4596],{"className":4594,"style":4595},[612],"height:0.2083em;",[316,4597],{},[316,4599,455],{"className":4600},[482]," denotes the probability that ",[316,4603,4605,4618],{"className":4604},[319],[316,4606,4608],{"className":4607},[323],[325,4609,4610],{"xmlns":327},[329,4611,4612,4616],{},[332,4613,4614],{},[335,4615,2876],{},[339,4617,2876],{"encoding":341},[316,4619,4621],{"className":4620,"ariaHidden":346},[345],[316,4622,4624,4627],{"className":4623},[350],[316,4625],{"className":4626,"style":355},[354],[316,4628,2876],{"className":4629,"style":2891},[359,360]," assigns to the token ",[316,4632,4634,4652],{"className":4633},[319],[316,4635,4637],{"className":4636},[323],[325,4638,4639],{"xmlns":327},[329,4640,4641,4649],{},[332,4642,4643],{},[557,4644,4645,4647],{},[335,4646,2267],{},[335,4648,3267],{},[339,4650,4651],{"encoding":341},"x_i",[316,4653,4655],{"className":4654,"ariaHidden":346},[345],[316,4656,4658,4662],{"className":4657},[350],[316,4659],{"className":4660,"style":4661},[354],"height:0.5806em;vertical-align:-0.15em;",[316,4663,4665,4668],{"className":4664},[359],[316,4666,2267],{"className":4667},[359,360],[316,4669,4671],{"className":4670},[599],[316,4672,4674,4694],{"className":4673},[603,604],[316,4675,4677,4691],{"className":4676},[608],[316,4678,4680],{"className":4679,"style":4452},[612],[316,4681,4682,4685],{"style":2975},[316,4683],{"className":4684,"style":621},[620],[316,4686,4688],{"className":4687},[625,626,627,628],[316,4689,3267],{"className":4690},[359,360,628],[316,4692,643],{"className":4693},[642],[316,4695,4697],{"className":4696},[608],[316,4698,4700],{"className":4699,"style":650},[612],[316,4701],{}," given the previous tokens ",[316,4704,4706,4742],{"className":4705},[319],[316,4707,4709],{"className":4708},[323],[325,4710,4711],{"xmlns":327},[329,4712,4713,4739],{},[332,4714,4715,4721,4723,4725,4727],{},[557,4716,4717,4719],{},[335,4718,2267],{},[1199,4720,273],{},[447,4722,708],{"separator":346},[447,4724,2926],{},[447,4726,708],{"separator":346},[557,4728,4729,4731],{},[335,4730,2267],{},[332,4732,4733,4735,4737],{},[335,4734,3267],{},[447,4736,3187],{},[1199,4738,273],{},[339,4740,4741],{"encoding":341},"x_1, \\dots, x_{i-1}",[316,4743,4745],{"className":4744,"ariaHidden":346},[345],[316,4746,4748,4752,4792,4795,4798,4801,4804,4807,4810],{"className":4747},[350],[316,4749],{"className":4750,"style":4751},[354],"height:0.6389em;vertical-align:-0.2083em;",[316,4753,4755,4758],{"className":4754},[359],[316,4756,2267],{"className":4757},[359,360],[316,4759,4761],{"className":4760},[599],[316,4762,4764,4784],{"className":4763},[603,604],[316,4765,4767,4781],{"className":4766},[608],[316,4768,4770],{"className":4769,"style":2972},[612],[316,4771,4772,4775],{"style":2975},[316,4773],{"className":4774,"style":621},[620],[316,4776,4778],{"className":4777},[625,626,627,628],[316,4779,273],{"className":4780},[359,628],[316,4782,643],{"className":4783},[642],[316,4785,4787],{"className":4786},[608],[316,4788,4790],{"className":4789,"style":650},[612],[316,4791],{},[316,4793,708],{"className":4794},[767],[316,4796],{"className":4797,"style":771},[662],[316,4799,2926],{"className":4800},[3051],[316,4802],{"className":4803,"style":771},[662],[316,4805,708],{"className":4806},[767],[316,4808],{"className":4809,"style":771},[662],[316,4811,4813,4816],{"className":4812},[359],[316,4814,2267],{"className":4815},[359,360],[316,4817,4819],{"className":4818},[599],[316,4820,4822,4851],{"className":4821},[603,604],[316,4823,4825,4848],{"className":4824},[608],[316,4826,4828],{"className":4827,"style":4452},[612],[316,4829,4830,4833],{"style":2975},[316,4831],{"className":4832,"style":621},[620],[316,4834,4836],{"className":4835},[625,626,627,628],[316,4837,4839,4842,4845],{"className":4838},[359,628],[316,4840,3267],{"className":4841},[359,360,628],[316,4843,3187],{"className":4844},[812,628],[316,4846,273],{"className":4847},[359,628],[316,4849,643],{"className":4850},[642],[316,4852,4854],{"className":4853},[608],[316,4855,4857],{"className":4856,"style":4595},[612],[316,4858],{},[216,4860,4861],{},"To compute perplexity, you need access to the probabilities (or logprobs) the language model assigns to each next token. Unfortunately, not all commercial models expose their models' logprobs, as discussed in Chapter 2.",{"title":125,"searchDepth":4863,"depth":4863,"links":4864},2,[4865,4866,4867,4868,4869,4870,4871],{"id":149,"depth":4863,"text":150},{"id":225,"depth":4863,"text":226},{"id":288,"depth":4863,"text":164},{"id":1220,"depth":4863,"text":1221},{"id":1763,"depth":4863,"text":172},{"id":2521,"depth":4863,"text":2522},{"id":2857,"depth":4863,"text":2858},"How entropy, cross entropy, perplexity, and BPC\u002FBPB measure a language model's predictive accuracy and how they relate.","md",{},{"icon":95},{"title":92,"description":4872},"LqWTUk174e3mAYMDcB716csert8BanOAcGKExrk4pLA",[4879,4881],{"title":87,"path":88,"stem":89,"description":4880,"icon":90,"children":-1},"Why evaluating foundation models is harder than traditional ML — intelligence, open-ended outputs, black boxes, saturating benchmarks, and expanding scope.",{"title":97,"path":98,"stem":99,"description":4882,"icon":100,"children":-1},"How functional correctness, similarity against reference data, and embeddings produce exact scores for open-ended model outputs.",1788871975540]