[{"data":1,"prerenderedAt":2383},["ShallowReactive",2],{"navigation_docs_en":3,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-3-exact-evaluation":114,"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-3-exact-evaluation-surround":2378},[4],{"title":5,"icon":6,"path":7,"stem":8,"children":9,"page":45},"AI Engineering",null,"\u002Fen\u002Fai-engineering","en\u002F1.ai-engineering",[10,46,77],{"title":11,"icon":12,"path":13,"stem":14,"children":15,"page":45},"Introduction to Building AI Applications with Foundation Models","i-lucide-brain-circuit","\u002Fen\u002Fai-engineering\u002Fintro","en\u002F1.ai-engineering\u002F1.intro",[16,20,25,30,35,40],{"title":11,"path":17,"stem":18,"icon":19},"\u002Fen\u002Fai-engineering\u002Fintro\u002Fch01","en\u002F1.ai-engineering\u002F1.intro\u002Fch01","i-lucide-sparkles",{"title":21,"path":22,"stem":23,"icon":24},"The Rise of AI Engineering","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch011-the-rise-of-ai-engineering","en\u002F1.ai-engineering\u002F1.intro\u002Fch011-the-rise-of-ai-engineering","i-lucide-history",{"title":26,"path":27,"stem":28,"icon":29},"Foundation Model Use Cases","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch012-foundation-model-use-cases","en\u002F1.ai-engineering\u002F1.intro\u002Fch012-foundation-model-use-cases","i-lucide-layout-grid",{"title":31,"path":32,"stem":33,"icon":34},"Planning AI Applications","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch013-planning-ai-applications","en\u002F1.ai-engineering\u002F1.intro\u002Fch013-planning-ai-applications","i-lucide-clipboard-list",{"title":36,"path":37,"stem":38,"icon":39},"The AI Engineering Stack","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch014-the-ai-engineering-stack","en\u002F1.ai-engineering\u002F1.intro\u002Fch014-the-ai-engineering-stack","i-lucide-layers",{"title":41,"path":42,"stem":43,"icon":44},"Summary","\u002Fen\u002Fai-engineering\u002Fintro\u002Fch015-summary","en\u002F1.ai-engineering\u002F1.intro\u002Fch015-summary","i-lucide-flag",false,{"title":47,"icon":6,"path":48,"stem":49,"children":50,"page":45},"Understanding Foundation Models","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models","en\u002F1.ai-engineering\u002F2.understanding-foundation-models",[51,54,59,64,69,74],{"title":47,"path":52,"stem":53,"icon":12},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02",{"title":55,"path":56,"stem":57,"icon":58},"Training Data","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-1-training-data","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-1-training-data","i-lucide-database",{"title":60,"path":61,"stem":62,"icon":63},"Modeling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-2-modeling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-2-modeling","i-lucide-network",{"title":65,"path":66,"stem":67,"icon":68},"Post-Training","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-3-post-training","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-3-post-training","i-lucide-sliders-horizontal",{"title":70,"path":71,"stem":72,"icon":73},"Sampling","\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-4-sampling","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-4-sampling","i-lucide-dices",{"title":41,"path":75,"stem":76,"icon":44},"\u002Fen\u002Fai-engineering\u002Funderstanding-foundation-models\u002Fch02-5-summary","en\u002F1.ai-engineering\u002F2.understanding-foundation-models\u002Fch02-5-summary",{"title":78,"path":79,"stem":80,"children":81,"page":45},"Evaluation Methodology","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology","en\u002F1.ai-engineering\u002F3.evaluation-methodology",[82,86,91,96,101,106,111],{"title":78,"path":83,"stem":84,"icon":85},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03","i-lucide-clipboard-check",{"title":87,"path":88,"stem":89,"icon":90},"Challenges of Evaluating Foundation Models","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-1-challenges-of-evaluating-foundation-models","i-lucide-shield-alert",{"title":92,"path":93,"stem":94,"icon":95},"Understanding Language Modeling Metrics","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-2-understanding-language-modeling-metrics","i-lucide-sigma",{"title":97,"path":98,"stem":99,"icon":100},"Exact Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-3-exact-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-3-exact-evaluation","i-lucide-check-check",{"title":102,"path":103,"stem":104,"icon":105},"AI as a Judge","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-4-ai-as-a-judge","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-4-ai-as-a-judge","i-lucide-scale",{"title":107,"path":108,"stem":109,"icon":110},"Ranking Models with Comparative Evaluation","\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-5-ranking-models-with-comparative-evaluation","i-lucide-trophy",{"title":41,"path":112,"stem":113,"icon":44},"\u002Fen\u002Fai-engineering\u002Fevaluation-methodology\u002Fch03-6-summary","en\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-6-summary",{"id":115,"title":97,"body":116,"description":2372,"extension":2373,"links":6,"meta":2374,"navigation":2375,"path":98,"seo":2376,"stem":99,"__hash__":2377},"docs_en\u002Fen\u002F1.ai-engineering\u002F3.evaluation-methodology\u002Fch03-3-exact-evaluation.md",{"type":117,"value":118,"toc":2357},"minimark",[119,138,143,147,171,174,182,186,196,216,226,229,232,244,252,259,264,271,307,318,350,354,357,485,568,649,653,660,663,667,670,673,697,700,716,719,752,755,758,765,768,795,798,802,805,829,848,864,886,893,897,900,907,940,943,958,989,999,1020,1047,1063,1081,1090,1097,1102,1110,1122,1126,1142,1162,1175,1182,1185,1469,2004,2019,2022,2029,2032,2036,2042,2057,2063,2072,2090,2095,2225,2232,2240,2253,2261,2281,2302,2305,2328,2334,2338,2353],[120,121,122,126],"u-page-hero",{},[123,124,97],"template",{"v-slot:title":125},"",[123,127,128,129,133,134,137],{"v-slot:description":125},"When evaluating models' performance, it's important to differentiate between ",[130,131,132],"strong",{},"exact"," and ",[130,135,136],{},"subjective"," evaluation. Exact evaluation produces judgment without ambiguity.",[139,140,142],"h2",{"id":141},"exact-versus-subjective","Exact Versus Subjective",[144,145,146],"p",{},"For example, if the answer to a multiple-choice question is A and you pick B, your answer is wrong. There's no ambiguity around that. On the other hand, essay grading is subjective. An essay's score depends on who grades the essay. The same person, if asked twice some time apart, can give the same essay different scores.",[148,149,150,163],"card-group",{},[151,152,154,155,158,159,162],"card",{"icon":153,"title":97},"i-lucide-check","If the answer is ",[130,156,157],{},"A"," and you pick ",[130,160,161],{},"B",", you're wrong. The judgment has no ambiguity.",[151,164,166,167,170],{"icon":105,"title":165},"Subjective Evaluation","Essay grading depends on ",[130,168,169],{},"who grades"," the essay. The same person, asked twice some time apart, can give the same essay different scores.",[144,172,173],{},"Essay grading can become more exact with clear grading guidelines.",[175,176,177,178,181],"note",{},"As you'll see in the next section, ",[130,179,180],{},"AI as a judge"," is subjective. The evaluation result can change based on the judge model and the prompt.",[139,183,185],{"id":184},"two-approaches-that-produce-exact-scores","Two Approaches That Produce Exact Scores",[144,187,188,189,133,192,195],{},"I'll cover two evaluation approaches that produce exact scores: ",[130,190,191],{},"functional correctness",[130,193,194],{},"similarity measurements against reference data",".",[148,197,198,207],{},[151,199,202,203,206],{"icon":200,"title":201},"i-lucide-play","Functional Correctness","Does the system ",[130,204,205],{},"perform the intended functionality","? If you ask a model to create a website, does the generated website meet your requirements?",[151,208,211,212,215],{"icon":209,"title":210},"i-lucide-git-compare","Similarity Against Reference Data","Compare generated outputs to ",[130,213,214],{},"reference responses",", such as evaluating an English translation against the correct English translation.",[175,217,218,219,222,223,195],{},"This section focuses on evaluating ",[130,220,221],{},"open-ended responses"," (arbitrary text generation) as opposed to close-ended responses (such as classification). This is not because foundation models aren't being used for close-ended tasks. In fact, many foundation model systems have at least a classification component, typically for intent classification or scoring. This section focuses on open-ended evaluation because ",[130,224,225],{},"close-ended evaluation is already well understood",[139,227,201],{"id":228},"functional-correctness",[144,230,231],{},"Functional correctness evaluation means evaluating a system based on whether it performs the intended functionality.",[148,233,234,239],{},[151,235,238],{"icon":236,"title":237},"i-lucide-globe","Generate a Website","If you ask a model to create a website, does the generated website meet your requirements?",[151,240,243],{"icon":241,"title":242},"i-lucide-calendar-check","Make a Reservation","If you ask a model to make a reservation at a certain restaurant, does the model succeed?",[245,246,247,248,251],"tip",{},"Functional correctness is the ",[130,249,250],{},"ultimate metric"," for evaluating the performance of any application, as it measures whether your application does what it's intended to do.",[253,254,255,256,195],"warning",{},"However, functional correctness isn't always straightforward to measure, and its measurement ",[130,257,258],{},"can't be easily automated",[260,261,263],"h3",{"id":262},"code-generation-and-execution-accuracy","Code Generation and Execution Accuracy",[144,265,266,267,195],{},"Code generation is an example of a task where functional correctness measurement can be automated. Functional correctness in coding is sometimes ",[268,269,270],"em",{},"execution accuracy",[144,272,273,274,278,279,133,282,285,286,288,289,291,292,288,295,298,299,302,303,306],{},"Say you ask the model to write a Python function, ",[275,276,277],"code",{},"gcd(num1, num2)",", to find the greatest common denominator (gcd) of two numbers, ",[275,280,281],{},"num1",[275,283,284],{},"num2",". The generated code can then be input into a Python interpreter to check whether the code is valid and if it is, whether it outputs the correct result of a given pair (",[275,287,281],{},", ",[275,290,284],{},"). For example, given the pair (",[275,293,294],{},"num1=15",[275,296,297],{},"num2=20","), if the function ",[275,300,301],{},"gcd(15, 20)"," doesn't return ",[275,304,305],{},"5",", the correct answer, you know that the function is wrong.",[144,308,309,310,317],{},"Long before AI was used for writing code, automatically verifying code's functional correctness was standard practice in software engineering. Code is typically validated with ",[311,312,316],"a",{"href":313,"rel":314},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FUnit_testing",[315],"nofollow","unit tests"," where code is executed in different scenarios to ensure that it generates the expected outputs. Functional correctness evaluation is how coding platforms like LeetCode and HackerRank validate the submitted solutions.",[144,319,320,321,133,326,331,332,337,338,343,344,349],{},"Popular benchmarks for evaluating AI's code generation capabilities, such as ",[311,322,325],{"href":323,"rel":324},"https:\u002F\u002Fhuggingface.co\u002Fdatasets\u002Fopenai\u002Fopenai_humaneval",[315],"OpenAI's HumanEval",[311,327,330],{"href":328,"rel":329},"https:\u002F\u002Fgithub.com\u002Fgoogle-research\u002Fgoogle-research\u002Ftree\u002Fmaster\u002Fmbpp",[315],"Google's MBPP"," (Mostly Basic Python Problems Dataset) use functional correctness as their metrics. Benchmarks for text-to-SQL (generating SQL queries from natural languages) like Spider (",[311,333,336],{"href":334,"rel":335},"https:\u002F\u002Fyale-lily.github.io\u002Fspider",[315],"Yu et al., 2018","), BIRD-SQL (Big Bench for Large-scale Database Grounded Text-to-SQL Evaluation) (",[311,339,342],{"href":340,"rel":341},"https:\u002F\u002Fbird-bench.github.io\u002F",[315],"Li et al., 2023","), and WikiSQL (",[311,345,348],{"href":346,"rel":347},"https:\u002F\u002Farxiv.org\u002Fabs\u002F1709.00103",[315],"Zhong, et al., 2017",") also rely on functional correctness.",[260,351,353],{"id":352},"humaneval-problems-test-cases-and-passk","HumanEval: Problems, Test Cases, and pass@k",[144,355,356],{},"A benchmark problem comes with a set of test cases. Each test case consists of a scenario the code should run and the expected output for that scenario. Here's an example of a problem and its test cases in HumanEval:",[358,359,363],"pre",{"className":360,"code":361,"language":362,"meta":125,"style":125},"language-python shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","# Problem\n\nfrom typing import List\n\ndef has_close_elements(numbers: List[float], threshold: float) -> bool:\n    \"\"\" Check if in given list of numbers, are any two numbers closer to each other than given threshold.\n    >>> has_close_elements([1.0, 2.0, 3.0], 0.5) False\n    >>> has_close_elements([1.0, 2.8, 3.0, 4.0, 5.0, 2.0], 0.3) True\n    \"\"\"\n\n# Test cases (each assert statement represents a test case)\n\ndef check(candidate):\n    assert candidate([1.0, 2.0, 3.9, 4.0, 5.0, 2.2], 0.3) == True\n    assert candidate([1.0, 2.0, 3.9, 4.0, 5.0, 2.2], 0.05) == False\n    assert candidate([1.0, 2.0, 5.9, 4.0, 5.0], 0.95) == True\n    assert candidate([1.0, 2.0, 5.9, 4.0, 5.0], 0.8) == False\n    assert candidate([1.0, 2.0, 3.0, 4.0, 5.0, 2.0], 0.1) == True\n    assert candidate([1.1, 2.2, 3.1, 4.1, 5.1], 1.0) == True\n    assert candidate([1.1, 2.2, 3.1, 4.1, 5.1], 0.5) == False\n","python",[275,364,365,373,380,386,391,397,403,409,415,421,426,432,437,443,449,455,461,467,473,479],{"__ignoreMap":125},[366,367,370],"span",{"class":368,"line":369},"line",1,[366,371,372],{},"# Problem\n",[366,374,376],{"class":368,"line":375},2,[366,377,379],{"emptyLinePlaceholder":378},true,"\n",[366,381,383],{"class":368,"line":382},3,[366,384,385],{},"from typing import List\n",[366,387,389],{"class":368,"line":388},4,[366,390,379],{"emptyLinePlaceholder":378},[366,392,394],{"class":368,"line":393},5,[366,395,396],{},"def has_close_elements(numbers: List[float], threshold: float) -> bool:\n",[366,398,400],{"class":368,"line":399},6,[366,401,402],{},"    \"\"\" Check if in given list of numbers, are any two numbers closer to each other than given threshold.\n",[366,404,406],{"class":368,"line":405},7,[366,407,408],{},"    >>> has_close_elements([1.0, 2.0, 3.0], 0.5) False\n",[366,410,412],{"class":368,"line":411},8,[366,413,414],{},"    >>> has_close_elements([1.0, 2.8, 3.0, 4.0, 5.0, 2.0], 0.3) True\n",[366,416,418],{"class":368,"line":417},9,[366,419,420],{},"    \"\"\"\n",[366,422,424],{"class":368,"line":423},10,[366,425,379],{"emptyLinePlaceholder":378},[366,427,429],{"class":368,"line":428},11,[366,430,431],{},"# Test cases (each assert statement represents a test case)\n",[366,433,435],{"class":368,"line":434},12,[366,436,379],{"emptyLinePlaceholder":378},[366,438,440],{"class":368,"line":439},13,[366,441,442],{},"def check(candidate):\n",[366,444,446],{"class":368,"line":445},14,[366,447,448],{},"    assert candidate([1.0, 2.0, 3.9, 4.0, 5.0, 2.2], 0.3) == True\n",[366,450,452],{"class":368,"line":451},15,[366,453,454],{},"    assert candidate([1.0, 2.0, 3.9, 4.0, 5.0, 2.2], 0.05) == False\n",[366,456,458],{"class":368,"line":457},16,[366,459,460],{},"    assert candidate([1.0, 2.0, 5.9, 4.0, 5.0], 0.95) == True\n",[366,462,464],{"class":368,"line":463},17,[366,465,466],{},"    assert candidate([1.0, 2.0, 5.9, 4.0, 5.0], 0.8) == False\n",[366,468,470],{"class":368,"line":469},18,[366,471,472],{},"    assert candidate([1.0, 2.0, 3.0, 4.0, 5.0, 2.0], 0.1) == True\n",[366,474,476],{"class":368,"line":475},19,[366,477,478],{},"    assert candidate([1.1, 2.2, 3.1, 4.1, 5.1], 1.0) == True\n",[366,480,482],{"class":368,"line":481},20,[366,483,484],{},"    assert candidate([1.1, 2.2, 3.1, 4.1, 5.1], 0.5) == False\n",[144,486,487,488,534,535,563,564,567],{},"When evaluating a model, for each problem a number of code samples, denoted as ",[366,489,492,514],{"className":490},[491],"katex",[366,493,496],{"className":494},[495],"katex-mathml",[497,498,500],"math",{"xmlns":499},"http:\u002F\u002Fwww.w3.org\u002F1998\u002FMath\u002FMathML",[501,502,503,510],"semantics",{},[504,505,506],"mrow",{},[507,508,509],"mi",{},"k",[511,512,509],"annotation",{"encoding":513},"application\u002Fx-tex",[366,515,519],{"className":516,"ariaHidden":518},[517],"katex-html","true",[366,520,523,528],{"className":521},[522],"base",[366,524],{"className":525,"style":527},[526],"strut","height:0.6944em;",[366,529,509],{"className":530,"style":533},[531,532],"mord","mathnormal","margin-right:0.0315em;",", are generated. A model solves a problem if any of the ",[366,536,538,551],{"className":537},[491],[366,539,541],{"className":540},[495],[497,542,543],{"xmlns":499},[501,544,545,549],{},[504,546,547],{},[507,548,509],{},[511,550,509],{"encoding":513},[366,552,554],{"className":553,"ariaHidden":518},[517],[366,555,557,560],{"className":556},[522],[366,558],{"className":559,"style":527},[526],[366,561,509],{"className":562,"style":533},[531,532]," code samples it generated pass all of that problem's test cases. The final score, called ",[275,565,566],{},"pass@k",", is the fraction of the solved problems out of all problems.",[175,569,570,571,630,631,634,635,638,639,642,643,645,646,195],{},"If there are 10 problems and a model solves 5 with ",[366,572,574,596],{"className":573},[491],[366,575,577],{"className":576},[495],[497,578,579],{"xmlns":499},[501,580,581,593],{},[504,582,583,585,589],{},[507,584,509],{},[586,587,588],"mo",{},"=",[590,591,592],"mn",{},"3",[511,594,595],{"encoding":513},"k = 3",[366,597,599,620],{"className":598,"ariaHidden":518},[517],[366,600,602,605,608,613,617],{"className":601},[522],[366,603],{"className":604,"style":527},[526],[366,606,509],{"className":607,"style":533},[531,532],[366,609],{"className":610,"style":612},[611],"mspace","margin-right:0.2778em;",[366,614,588],{"className":615},[616],"mrel",[366,618],{"className":619,"style":612},[611],[366,621,623,627],{"className":622},[522],[366,624],{"className":625,"style":626},[526],"height:0.6444em;",[366,628,592],{"className":629},[531],", then that model's ",[275,632,633],{},"pass@3"," score is ",[130,636,637],{},"50%",". The more code samples a model generates, the more chance the model has at solving each problem, hence the greater the final score. This means that in expectation, ",[275,640,641],{},"pass@1"," score should be lower than ",[275,644,633],{},", which, in turn, should be lower than ",[275,647,648],{},"pass@10",[260,650,652],{"id":651},"game-bots-and-measurable-objectives","Game Bots and Measurable Objectives",[144,654,655,656,659],{},"Another category of tasks whose functional correctness can be automatically evaluated is game bots. If you create a bot to play ",[268,657,658],{},"Tetris",", you can tell how good the bot is by the score it gets. Tasks with measurable objectives can typically be evaluated using functional correctness. For example, if you ask AI to schedule your workloads to optimize energy consumption, the AI's performance can be measured by how much energy it saves.",[175,661,662],{},"The challenge is that while many complex tasks have measurable objectives, AI isn't quite good enough to perform complex tasks end-to-end, so AI might be used to do part of the solution. Sometimes, evaluating a part of a solution is harder than evaluating the end outcome. Imagine you want to evaluate someone's ability to play chess. It's easier to evaluate the end game outcome (win\u002Flose\u002Fdraw) than to evaluate just one move.",[139,664,666],{"id":665},"similarity-measurements-against-reference-data","Similarity Measurements Against Reference Data",[144,668,669],{},"If the task you care about can't be automatically evaluated using functional correctness, one common approach is to evaluate AI's outputs against reference data. For example, if you ask a model to translate a sentence from French to English, you can evaluate the generated English translation against the correct English translation.",[144,671,672],{},"Each example in the reference data follows the format (input, reference responses). An input can have multiple reference responses, such as multiple possible English translations of a French sentence.",[674,675,676],"accordion",{},[677,678,681,682,685,686,689,690,693,694,195],"accordion-item",{"icon":679,"label":680},"i-lucide-book-marked","Ground truths, canonical responses, and reference-free metrics","Reference responses are also called ",[268,683,684],{},"ground truths"," or ",[268,687,688],{},"canonical responses",". Metrics that require references are ",[268,691,692],{},"reference-based",", and metrics that don't are ",[268,695,696],{},"reference-free",[144,698,699],{},"Since this evaluation approach requires reference data, it's bottlenecked by how much and how fast reference data can be generated. Reference data is generated typically by humans and increasingly by AIs.",[148,701,702,711],{},[151,703,706,707,710],{"icon":704,"title":705},"i-lucide-users","Human-Generated References","Using human-generated data as the reference means that we treat ",[130,708,709],{},"human performance as the gold standard",", and AI's performance is measured against human performance. Human-generated data can be expensive and time-consuming to generate.",[151,712,715],{"icon":713,"title":714},"i-lucide-bot","AI-Generated References","This cost leads many to use AI to generate reference data instead. AI-generated data might still need human reviews, but the labor needed to review it is much less than the labor needed to generate reference data from scratch.",[144,717,718],{},"Generated responses that are more similar to the reference responses are considered better. There are four ways to measure the similarity between two open-ended texts:",[720,721,723,728,731,735,738,742,745,749],"steps",{"level":722},"4",[724,725,727],"h4",{"id":726},"asking-an-evaluator","Asking an evaluator",[144,729,730],{},"Asking an evaluator to make the judgment whether two texts are the same.",[724,732,734],{"id":733},"exact-match","Exact match",[144,736,737],{},"Whether the generated response matches one of the reference responses exactly.",[724,739,741],{"id":740},"lexical-similarity","Lexical similarity",[144,743,744],{},"How similar the generated response looks to the reference responses.",[724,746,748],{"id":747},"semantic-similarity","Semantic similarity",[144,750,751],{},"How close the generated response is to the reference responses in meaning (semantics).",[144,753,754],{},"Two responses can be compared by human evaluators or AI evaluators. AI evaluators are increasingly common and will be the focus of the next section.",[144,756,757],{},"This section focuses on hand-designed metrics: exact match, lexical similarity, and semantic similarity. Scores by exact matching are binary (match or not), whereas the other two scores are on a sliding scale (such as between 0 and 1 or between –1 and 1).",[245,759,760,761,764],{},"Despite the ease of use and flexibility of the AI as a judge approach, ",[130,762,763],{},"hand-designed similarity measurements are still widely used"," in the industry for their exact nature.",[144,766,767],{},"This section discusses how you can use similarity measurements to evaluate the quality of a generated output. However, you can also use similarity measurements for many other use cases, including but not limited to the following:",[148,769,770,775,780,785,790],{},[151,771,774],{"icon":772,"title":773},"i-lucide-search","Retrieval and Search","Find items similar to a query.",[151,776,779],{"icon":777,"title":778},"i-lucide-list-ordered","Ranking","Rank items based on how similar they are to a query.",[151,781,784],{"icon":782,"title":783},"i-lucide-group","Clustering","Cluster items based on how similar they are to each other.",[151,786,789],{"icon":787,"title":788},"i-lucide-scan-search","Anomaly Detection","Detect items that are the least similar to the rest.",[151,791,794],{"icon":792,"title":793},"i-lucide-copy-minus","Data Deduplication","Remove items that are too similar to other items.",[245,796,797],{},"Techniques discussed in this section will come up again throughout the book.",[260,799,801],{"id":800},"exact-match-1","Exact Match",[144,803,804],{},"It's considered an exact match if the generated response matches one of the reference responses exactly. Exact matching works for tasks that expect short, exact responses such as simple math problems, common knowledge queries, and trivia-style questions. Here are examples of inputs that have short, exact responses:",[806,807,808,814,819,824],"ul",{},[809,810,811],"li",{},[275,812,813],{},"\"What's 2 + 3?\"",[809,815,816],{},[275,817,818],{},"\"Who was the first woman to win a Nobel Prize?\"",[809,820,821],{},[275,822,823],{},"\"What's my current account balance?\"",[809,825,826],{},[275,827,828],{},"\"Fill in the blank: Paris to France is like ___ to England.\"",[144,830,831,832,834,835,838,839,841,842,133,845,195],{},"There are variations to matching that take into account formatting issues. One variation is to accept any output that contains the reference response as a match. Consider the question ",[275,833,813],{}," The reference response is ",[275,836,837],{},"\"5\"",". This variation accepts all outputs that contain ",[275,840,837],{},", including ",[275,843,844],{},"\"The answer is 5\"",[275,846,847],{},"\"2 + 3 is 5\"",[849,850,851,852,855,856,859,860,863],"caution",{},"However, this variation can sometimes lead to the ",[130,853,854],{},"wrong solution being accepted",". Consider the question ",[275,857,858],{},"\"What year was Anne Frank born?\""," Anne Frank was born on June 12, 1929, so the correct response is 1929. If the model outputs ",[275,861,862],{},"\"September 12, 1929\"",", the correct year is included in the output, but the output is factually wrong.",[144,865,866,867,870,871,288,874,877,878,881,882,885],{},"Beyond simple tasks, exact match rarely works. Given the original French sentence ",[275,868,869],{},"\"Comment ça va?\"",", there are multiple possible English translations, such as ",[275,872,873],{},"\"How are you?\"",[275,875,876],{},"\"How is everything?\"",", and ",[275,879,880],{},"\"How are you doing?\""," If the reference data contains only these three translations and a model generates ",[275,883,884],{},"\"How is it going?\"",", the model's response will be marked as wrong. The longer and more complex the original text, the more possible translations there are. It's impossible to create an exhaustive set of possible responses for an input.",[253,887,888,889,892],{},"For complex tasks, ",[130,890,891],{},"lexical similarity and semantic similarity"," work better.",[260,894,896],{"id":895},"lexical-similarity-1","Lexical Similarity",[144,898,899],{},"Lexical similarity measures how much two texts overlap. You can do this by first breaking each text into smaller tokens.",[144,901,902,903,906],{},"In its simplest form, lexical similarity can be measured by counting how many tokens two texts have in common. As an example, consider the reference response ",[275,904,905],{},"\"My cats scare the mice\""," and two generated responses:",[148,908,909,925],{},[151,910,913,916,917,920,921,924],{"icon":911,"title":912},"i-lucide-cat","Response A",[275,914,915],{},"\"My cats eat the mice\""," — assume that each token is a word. If you count overlapping of individual words only, response A contains ",[130,918,919],{},"4 out of 5"," words in the reference response (the similarity score is ",[130,922,923],{},"80%",").",[151,926,929,932,933,936,937,924],{"icon":927,"title":928},"i-lucide-swords","Response B",[275,930,931],{},"\"Cats and mice fight all the time\""," — response B contains only ",[130,934,935],{},"3 out of 5"," (the similarity score is ",[130,938,939],{},"60%",[144,941,942],{},"Response A is, therefore, considered more similar to the reference response.",[144,944,945,946,949,950,953,954,957],{},"One way to measure lexical similarity is ",[268,947,948],{},"approximate string matching",", known colloquially as ",[268,951,952],{},"fuzzy matching",". It measures the similarity between two texts by counting how many edits it'd need to convert from one text to another, a number called ",[268,955,956],{},"edit distance",". The usual three edit operations are:",[148,959,960,971,980],{},[151,961,964,967,968],{"icon":962,"title":963},"i-lucide-minus","Deletion",[275,965,966],{},"\"brad\""," → ",[275,969,970],{},"\"bad\"",[151,972,975,967,977],{"icon":973,"title":974},"i-lucide-plus","Insertion",[275,976,970],{},[275,978,979],{},"\"bard\"",[151,981,984,967,986],{"icon":982,"title":983},"i-lucide-replace","Substitution",[275,985,970],{},[275,987,988],{},"\"bed\"",[144,990,991,992,967,995,998],{},"Some fuzzy matchers also treat transposition, swapping two letters (e.g., ",[275,993,994],{},"\"mats\"",[275,996,997],{},"\"mast\"","), to be an edit. However, some fuzzy matchers treat each transposition as two edit operations: one deletion and one insertion.",[144,1000,1001,1002,1004,1005,1007,1008,1011,1012,1014,1015,1017,1018,195],{},"For example, ",[275,1003,970],{}," is one edit to ",[275,1006,979],{}," and three edits to ",[275,1009,1010],{},"\"cash\"",", so ",[275,1013,970],{}," is considered more similar to ",[275,1016,979],{}," than to ",[275,1019,1010],{},[144,1021,1022,1023,1026,1027,1030,1031,1033,1034,288,1037,288,1040,877,1043,1046],{},"Another way to measure lexical similarity is ",[268,1024,1025],{},"n-gram similarity",", measured based on the overlapping of sequences of tokens, ",[268,1028,1029],{},"n-grams",", instead of single tokens. A 1-gram (unigram) is a token. A 2-gram (bigram) is a set of two tokens. ",[275,1032,905],{}," consists of four bigrams: ",[275,1035,1036],{},"\"my cats\"",[275,1038,1039],{},"\"cats scare\"",[275,1041,1042],{},"\"scare the\"",[275,1044,1045],{},"\"the mice\"",". You measure what percentage of n-grams in reference responses is also in the generated response.",[175,1048,1049,1050,133,1053,685,1056,133,1059,1062],{},"You might also want to do some processing depending on whether you want ",[275,1051,1052],{},"\"cats\"",[275,1054,1055],{},"\"cat\"",[275,1057,1058],{},"\"will not\"",[275,1060,1061],{},"\"won't\""," to be considered two separate tokens.",[144,1064,1065,1066,288,1071,877,1076,195],{},"Common metrics for lexical similarity are BLEU, ROUGE, METEOR++, TER, and CIDEr. They differ in exactly how the overlapping is calculated. Before foundation models, BLEU, ROUGE, and their relatives were common, especially for translation tasks. Since the rise of foundation models, fewer benchmarks use lexical similarity. Examples of benchmarks that use these metrics are ",[311,1067,1070],{"href":1068,"rel":1069},"https:\u002F\u002Fhuggingface.co\u002Fpapers\u002Ftrending?q=wmt",[315],"WMT",[311,1072,1075],{"href":1073,"rel":1074},"https:\u002F\u002Fhuggingface.co\u002Fpapers\u002F1504.00325",[315],"COCO Captions",[311,1077,1080],{"href":1078,"rel":1079},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2206.11249",[315],"GEMv2",[144,1082,1083,1084,1089],{},"A drawback of this method is that it requires curating a comprehensive set of reference responses. A good response can get a low similarity score if the reference set doesn't contain any response that looks like it. On some benchmark examples, ",[311,1085,1088],{"href":1086,"rel":1087},"https:\u002F\u002Fwww.adept.ai\u002Fblog\u002Ffuyu-8b\u002F",[315],"Adept"," found that its model Fuyu performed poorly not because the model's outputs were wrong, but because some correct answers were missing in the reference data. Figure 3-5 shows an example of an image-captioning task in which Fuyu generated a correct caption but was given a low score.",[144,1091,1092],{},[1093,1094],"img",{"alt":1095,"src":1096},"Figure 3-5. An example where Fuyu generated a correct option but was given a low score because of the limitation of reference captions.",".\u002Fmedia\u002Ffig-3-5.png",[1098,1099,1100],"blockquote",{},[144,1101,1095],{},[144,1103,1104,1105,924],{},"Not only that, but references can be wrong. For example, the organizers of the WMT 2023 Metrics shared task, which focuses on examining evaluation metrics for machine translation, reported that they found many bad reference translations in their data. Low-quality reference data is one of the reasons that reference-free metrics were strong contenders for reference-based metrics in terms of correlation to human judgment (",[311,1106,1109],{"href":1107,"rel":1108},"https:\u002F\u002Faclanthology.org\u002F2023.wmt-1.51\u002F",[315],"Freitag et al., 2023",[253,1111,1112,1113,1116,1117,924],{},"Another drawback of this measurement is that ",[130,1114,1115],{},"higher lexical similarity scores don't always mean better responses",". For example, on HumanEval, a code generation benchmark, OpenAI found that BLEU scores for incorrect and correct solutions were similar. This indicates that optimizing for BLEU scores isn't the same as optimizing for functional correctness (",[311,1118,1121],{"href":1119,"rel":1120},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2107.03374",[315],"Chen et al., 2021",[260,1123,1125],{"id":1124},"semantic-similarity-1","Semantic Similarity",[144,1127,1128,1129,133,1132,1134,1135,133,1138,1141],{},"Lexical similarity measures whether two texts look similar, not whether they have the same meaning. Consider the two sentences ",[275,1130,1131],{},"\"What's up?\"",[275,1133,873],{}," Lexically, they are different—there's little overlapping in the words and letters they use. However, semantically, they are close. Conversely, similar-looking texts can mean very different things. ",[275,1136,1137],{},"\"Let's eat, grandma\"",[275,1139,1140],{},"\"Let's eat grandma\""," mean two completely different things.",[144,1143,1144,1146,1147,1150,1151,1154,1155,1158,1159,195],{},[268,1145,748],{}," aims to compute the similarity in semantics. This first requires transforming a text into a numerical representation, which is called an ",[268,1148,1149],{},"embedding",". For example, the sentence ",[275,1152,1153],{},"\"the cat sits on a mat\""," might be represented using an embedding that looks like this: ",[275,1156,1157],{},"[0.11, 0.02, 0.54]",". Semantic similarity is, therefore, also called ",[268,1160,1161],{},"embedding similarity",[144,1163,1164,1167,1168,1171,1172,195],{},[268,1165,1166],{},"\"Introduction to Embedding\""," discusses how embeddings work. For now, let's assume that you have a way to transform texts into embeddings. The similarity between two embeddings can be computed using metrics such as cosine similarity. Two embeddings that are exactly the same have a similarity score of ",[275,1169,1170],{},"1",". Two opposite embeddings have a similarity score of ",[275,1173,1174],{},"–1",[175,1176,1177,1178,1181],{},"I'm using text examples, but semantic similarity can be computed for embeddings of ",[130,1179,1180],{},"any data modality",", including images and audio. Semantic similarity for text is sometimes called semantic textual similarity.",[253,1183,1184],{},"While I put semantic similarity in the exact evaluation category, it can be considered subjective, as different embedding algorithms can produce different embeddings. However, given two embeddings, the similarity score between them is computed exactly.",[144,1186,1187,1188,1217,1218,1247,1248,133,1276,1304,1305,1468],{},"Mathematically, let ",[366,1189,1191,1204],{"className":1190},[491],[366,1192,1194],{"className":1193},[495],[497,1195,1196],{"xmlns":499},[501,1197,1198,1202],{},[504,1199,1200],{},[507,1201,157],{},[511,1203,157],{"encoding":513},[366,1205,1207],{"className":1206,"ariaHidden":518},[517],[366,1208,1210,1214],{"className":1209},[522],[366,1211],{"className":1212,"style":1213},[526],"height:0.6833em;",[366,1215,157],{"className":1216},[531,532]," be an embedding of the generated response, and ",[366,1219,1221,1234],{"className":1220},[491],[366,1222,1224],{"className":1223},[495],[497,1225,1226],{"xmlns":499},[501,1227,1228,1232],{},[504,1229,1230],{},[507,1231,161],{},[511,1233,161],{"encoding":513},[366,1235,1237],{"className":1236,"ariaHidden":518},[517],[366,1238,1240,1243],{"className":1239},[522],[366,1241],{"className":1242,"style":1213},[526],[366,1244,161],{"className":1245,"style":1246},[531,532],"margin-right:0.0502em;"," be an embedding of a reference response. The cosine similarity between ",[366,1249,1251,1264],{"className":1250},[491],[366,1252,1254],{"className":1253},[495],[497,1255,1256],{"xmlns":499},[501,1257,1258,1262],{},[504,1259,1260],{},[507,1261,157],{},[511,1263,157],{"encoding":513},[366,1265,1267],{"className":1266,"ariaHidden":518},[517],[366,1268,1270,1273],{"className":1269},[522],[366,1271],{"className":1272,"style":1213},[526],[366,1274,157],{"className":1275},[531,532],[366,1277,1279,1292],{"className":1278},[491],[366,1280,1282],{"className":1281},[495],[497,1283,1284],{"xmlns":499},[501,1285,1286,1290],{},[504,1287,1288],{},[507,1289,161],{},[511,1291,161],{"encoding":513},[366,1293,1295],{"className":1294,"ariaHidden":518},[517],[366,1296,1298,1301],{"className":1297},[522],[366,1299],{"className":1300,"style":1213},[526],[366,1302,161],{"className":1303,"style":1246},[531,532]," is computed as ",[366,1306,1308,1348],{"className":1307},[491],[366,1309,1311],{"className":1310},[495],[497,1312,1313],{"xmlns":499},[501,1314,1315,1345],{},[504,1316,1317],{},[1318,1319,1320,1329],"mfrac",{},[504,1321,1322,1324,1327],{},[507,1323,157],{},[586,1325,1326],{},"⋅",[507,1328,161],{},[504,1330,1331,1335,1337,1339,1341,1343],{},[507,1332,1334],{"mathvariant":1333},"normal","∥",[507,1336,157],{},[507,1338,1334],{"mathvariant":1333},[507,1340,1334],{"mathvariant":1333},[507,1342,161],{},[507,1344,1334],{"mathvariant":1333},[511,1346,1347],{"encoding":513},"\\frac{A \\cdot B}{\\|A\\| \\|B\\|}",[366,1349,1351],{"className":1350,"ariaHidden":518},[517],[366,1352,1354,1358],{"className":1353},[522],[366,1355],{"className":1356,"style":1357},[526],"height:1.3923em;vertical-align:-0.52em;",[366,1359,1361,1366,1464],{"className":1360},[531],[366,1362],{"className":1363},[1364,1365],"mopen","nulldelimiter",[366,1367,1369],{"className":1368},[1318],[366,1370,1374,1455],{"className":1371},[1372,1373],"vlist-t","vlist-t2",[366,1375,1378,1450],{"className":1376},[1377],"vlist-r",[366,1379,1383,1417,1428],{"className":1380,"style":1382},[1381],"vlist","height:0.8723em;",[366,1384,1386,1391],{"style":1385},"top:-2.655em;",[366,1387],{"className":1388,"style":1390},[1389],"pstrut","height:3em;",[366,1392,1398],{"className":1393},[1394,1395,1396,1397],"sizing","reset-size6","size3","mtight",[366,1399,1401,1404,1407,1411,1414],{"className":1400},[531,1397],[366,1402,1334],{"className":1403},[531,1397],[366,1405,157],{"className":1406},[531,532,1397],[366,1408,1410],{"className":1409},[531,1397],"∥∥",[366,1412,161],{"className":1413,"style":1246},[531,532,1397],[366,1415,1334],{"className":1416},[531,1397],[366,1418,1420,1423],{"style":1419},"top:-3.23em;",[366,1421],{"className":1422,"style":1390},[1389],[366,1424],{"className":1425,"style":1427},[1426],"frac-line","border-bottom-width:0.04em;",[366,1429,1431,1434],{"style":1430},"top:-3.394em;",[366,1432],{"className":1433,"style":1390},[1389],[366,1435,1437],{"className":1436},[1394,1395,1396,1397],[366,1438,1440,1443,1447],{"className":1439},[531,1397],[366,1441,157],{"className":1442},[531,532,1397],[366,1444,1326],{"className":1445},[1446,1397],"mbin",[366,1448,161],{"className":1449,"style":1246},[531,532,1397],[366,1451,1454],{"className":1452},[1453],"vlist-s","​",[366,1456,1458],{"className":1457},[1377],[366,1459,1462],{"className":1460,"style":1461},[1381],"height:0.52em;",[366,1463],{},[366,1465],{"className":1466},[1467,1365],"mclose",", with:",[1470,1471,1472,1534],"field-group",{},[1473,1474,1477,1478,133,1506,195],"field",{"name":1475,"type":1476},"A · B","dot product","The dot product of ",[366,1479,1481,1494],{"className":1480},[491],[366,1482,1484],{"className":1483},[495],[497,1485,1486],{"xmlns":499},[501,1487,1488,1492],{},[504,1489,1490],{},[507,1491,157],{},[511,1493,157],{"encoding":513},[366,1495,1497],{"className":1496,"ariaHidden":518},[517],[366,1498,1500,1503],{"className":1499},[522],[366,1501],{"className":1502,"style":1213},[526],[366,1504,157],{"className":1505},[531,532],[366,1507,1509,1522],{"className":1508},[491],[366,1510,1512],{"className":1511},[495],[497,1513,1514],{"xmlns":499},[501,1515,1516,1520],{},[504,1517,1518],{},[507,1519,161],{},[511,1521,161],{"encoding":513},[366,1523,1525],{"className":1524,"ariaHidden":518},[517],[366,1526,1528,1531],{"className":1527},[522],[366,1529],{"className":1530,"style":1213},[526],[366,1532,161],{"className":1533,"style":1246},[531,532],[1473,1535,1538,1539,1605,1606,1634,1635,1663,1664,288,1738,195],{"name":1536,"type":1537},"||A||","L² norm","The Euclidean norm (also known as ",[366,1540,1542,1563],{"className":1541},[491],[366,1543,1545],{"className":1544},[495],[497,1546,1547],{"xmlns":499},[501,1548,1549,1560],{},[504,1550,1551],{},[1552,1553,1554,1557],"msup",{},[507,1555,1556],{},"L",[590,1558,1559],{},"2",[511,1561,1562],{"encoding":513},"L^2",[366,1564,1566],{"className":1565,"ariaHidden":518},[517],[366,1567,1569,1573],{"className":1568},[522],[366,1570],{"className":1571,"style":1572},[526],"height:0.8141em;",[366,1574,1576,1579],{"className":1575},[531],[366,1577,1556],{"className":1578},[531,532],[366,1580,1583],{"className":1581},[1582],"msupsub",[366,1584,1586],{"className":1585},[1372],[366,1587,1589],{"className":1588},[1377],[366,1590,1592],{"className":1591,"style":1572},[1381],[366,1593,1595,1599],{"style":1594},"top:-3.063em;margin-right:0.05em;",[366,1596],{"className":1597,"style":1598},[1389],"height:2.7em;",[366,1600,1602],{"className":1601},[1394,1395,1396,1397],[366,1603,1559],{"className":1604},[531,1397]," norm) of ",[366,1607,1609,1622],{"className":1608},[491],[366,1610,1612],{"className":1611},[495],[497,1613,1614],{"xmlns":499},[501,1615,1616,1620],{},[504,1617,1618],{},[507,1619,157],{},[511,1621,157],{"encoding":513},[366,1623,1625],{"className":1624,"ariaHidden":518},[517],[366,1626,1628,1631],{"className":1627},[522],[366,1629],{"className":1630,"style":1213},[526],[366,1632,157],{"className":1633},[531,532],". If ",[366,1636,1638,1651],{"className":1637},[491],[366,1639,1641],{"className":1640},[495],[497,1642,1643],{"xmlns":499},[501,1644,1645,1649],{},[504,1646,1647],{},[507,1648,157],{},[511,1650,157],{"encoding":513},[366,1652,1654],{"className":1653,"ariaHidden":518},[517],[366,1655,1657,1660],{"className":1656},[522],[366,1658],{"className":1659,"style":1213},[526],[366,1661,157],{"className":1662},[531,532]," is ",[366,1665,1667,1699],{"className":1666},[491],[366,1668,1670],{"className":1669},[495],[497,1671,1672],{"xmlns":499},[501,1673,1674,1697],{},[504,1675,1676,1680,1683,1686,1689,1691,1694],{},[586,1677,1679],{"stretchy":1678},"false","[",[590,1681,1682],{},"0.11",[586,1684,1685],{"separator":518},",",[590,1687,1688],{},"0.02",[586,1690,1685],{"separator":518},[590,1692,1693],{},"0.54",[586,1695,1696],{"stretchy":1678},"]",[511,1698,1157],{"encoding":513},[366,1700,1702],{"className":1701,"ariaHidden":518},[517],[366,1703,1705,1709,1712,1715,1719,1723,1726,1729,1732,1735],{"className":1704},[522],[366,1706],{"className":1707,"style":1708},[526],"height:1em;vertical-align:-0.25em;",[366,1710,1679],{"className":1711},[1364],[366,1713,1682],{"className":1714},[531],[366,1716,1685],{"className":1717},[1718],"mpunct",[366,1720],{"className":1721,"style":1722},[611],"margin-right:0.1667em;",[366,1724,1688],{"className":1725},[531],[366,1727,1685],{"className":1728},[1718],[366,1730],{"className":1731,"style":1722},[611],[366,1733,1693],{"className":1734},[531],[366,1736,1696],{"className":1737},[1467],[366,1739,1741,1789],{"className":1740},[491],[366,1742,1744],{"className":1743},[495],[497,1745,1746],{"xmlns":499},[501,1747,1748,1786],{},[504,1749,1750,1752,1754,1756,1758],{},[507,1751,1334],{"mathvariant":1333},[507,1753,157],{},[507,1755,1334],{"mathvariant":1333},[586,1757,588],{},[1759,1760,1761],"msqrt",{},[504,1762,1763,1769,1772,1778,1780],{},[1552,1764,1765,1767],{},[590,1766,1682],{},[590,1768,1559],{},[586,1770,1771],{},"+",[1552,1773,1774,1776],{},[590,1775,1688],{},[590,1777,1559],{},[586,1779,1771],{},[1552,1781,1782,1784],{},[590,1783,1693],{},[590,1785,1559],{},[511,1787,1788],{"encoding":513},"\\|A\\| = \\sqrt{0.11^2 + 0.02^2 + 0.54^2}",[366,1790,1792,1816],{"className":1791,"ariaHidden":518},[517],[366,1793,1795,1798,1801,1804,1807,1810,1813],{"className":1794},[522],[366,1796],{"className":1797,"style":1708},[526],[366,1799,1334],{"className":1800},[531],[366,1802,157],{"className":1803},[531,532],[366,1805,1334],{"className":1806},[531],[366,1808],{"className":1809,"style":612},[611],[366,1811,588],{"className":1812},[616],[366,1814],{"className":1815,"style":612},[611],[366,1817,1819,1823],{"className":1818},[522],[366,1820],{"className":1821,"style":1822},[526],"height:1.04em;vertical-align:-0.1266em;",[366,1824,1827],{"className":1825},[531,1826],"sqrt",[366,1828,1830,1995],{"className":1829},[1372,1373],[366,1831,1833,1992],{"className":1832},[1377],[366,1834,1837,1969],{"className":1835,"style":1836},[1381],"height:0.9134em;",[366,1838,1842,1845],{"className":1839,"style":1841},[1840],"svg-align","top:-3em;",[366,1843],{"className":1844,"style":1390},[1389],[366,1846,1849,1853,1884,1888,1891,1894,1898,1927,1930,1933,1936,1940],{"className":1847,"style":1848},[531],"padding-left:0.833em;",[366,1850,1852],{"className":1851},[531],"0.1",[366,1854,1856,1859],{"className":1855},[531],[366,1857,1170],{"className":1858},[531],[366,1860,1862],{"className":1861},[1582],[366,1863,1865],{"className":1864},[1372],[366,1866,1868],{"className":1867},[1377],[366,1869,1872],{"className":1870,"style":1871},[1381],"height:0.7401em;",[366,1873,1875,1878],{"style":1874},"top:-2.989em;margin-right:0.05em;",[366,1876],{"className":1877,"style":1598},[1389],[366,1879,1881],{"className":1880},[1394,1395,1396,1397],[366,1882,1559],{"className":1883},[531,1397],[366,1885],{"className":1886,"style":1887},[611],"margin-right:0.2222em;",[366,1889,1771],{"className":1890},[1446],[366,1892],{"className":1893,"style":1887},[611],[366,1895,1897],{"className":1896},[531],"0.0",[366,1899,1901,1904],{"className":1900},[531],[366,1902,1559],{"className":1903},[531],[366,1905,1907],{"className":1906},[1582],[366,1908,1910],{"className":1909},[1372],[366,1911,1913],{"className":1912},[1377],[366,1914,1916],{"className":1915,"style":1871},[1381],[366,1917,1918,1921],{"style":1874},[366,1919],{"className":1920,"style":1598},[1389],[366,1922,1924],{"className":1923},[1394,1395,1396,1397],[366,1925,1559],{"className":1926},[531,1397],[366,1928],{"className":1929,"style":1887},[611],[366,1931,1771],{"className":1932},[1446],[366,1934],{"className":1935,"style":1887},[611],[366,1937,1939],{"className":1938},[531],"0.5",[366,1941,1943,1946],{"className":1942},[531],[366,1944,722],{"className":1945},[531],[366,1947,1949],{"className":1948},[1582],[366,1950,1952],{"className":1951},[1372],[366,1953,1955],{"className":1954},[1377],[366,1956,1958],{"className":1957,"style":1871},[1381],[366,1959,1960,1963],{"style":1874},[366,1961],{"className":1962,"style":1598},[1389],[366,1964,1966],{"className":1965},[1394,1395,1396,1397],[366,1967,1559],{"className":1968},[531,1397],[366,1970,1972,1975],{"style":1971},"top:-2.8734em;",[366,1973],{"className":1974,"style":1390},[1389],[366,1976,1980],{"className":1977,"style":1979},[1978],"hide-tail","min-width:0.853em;height:1.08em;",[1981,1982,1988],"svg",{"xmlns":1983,"width":1984,"height":1985,"viewBox":1986,"preserveAspectRatio":1987},"http:\u002F\u002Fwww.w3.org\u002F2000\u002Fsvg","400em","1.08em","0 0 400000 1080","xMinYMin slice",[1989,1990],"path",{"d":1991},"M95,702\nc-2.7,0,-7.17,-2.7,-13.5,-8c-5.8,-5.3,-9.5,-10,-9.5,-14\nc0,-2,0.3,-3.3,1,-4c1.3,-2.7,23.83,-20.7,67.5,-54\nc44.2,-33.3,65.8,-50.3,66.5,-51c1.3,-1.3,3,-2,5,-2c4.7,0,8.7,3.3,12,10\ns173,378,173,378c0.7,0,35.3,-71,104,-213c68.7,-142,137.5,-285,206.5,-429\nc69,-144,104.5,-217.7,106.5,-221\nl0 -0\nc5.3,-9.3,12,-14,20,-14\nH400000v40H845.2724\ns-225.272,467,-225.272,467s-235,486,-235,486c-2.7,4.7,-9,7,-19,7\nc-6,0,-10,-1,-12,-3s-194,-422,-194,-422s-65,47,-65,47z\nM834 80h400000v40h-400000z",[366,1993,1454],{"className":1994},[1453],[366,1996,1998],{"className":1997},[1377],[366,1999,2002],{"className":2000,"style":2001},[1381],"height:0.1266em;",[366,2003],{},[144,2005,2006,2007,2012,2013,2018],{},"Metrics for semantic textual similarity include ",[311,2008,2011],{"href":2009,"rel":2010},"https:\u002F\u002Farxiv.org\u002Fabs\u002F1904.09675",[315],"BERTScore"," (embeddings are generated by BERT) and ",[311,2014,2017],{"href":2015,"rel":2016},"https:\u002F\u002Faclanthology.org\u002FD19-1053\u002F",[315],"MoverScore"," (embeddings are generated by a mixture of algorithms).",[144,2020,2021],{},"Semantic textual similarity doesn't require a set of reference responses as comprehensive as lexical similarity does. However, the reliability of semantic similarity depends on the quality of the underlying embedding algorithm. Two texts with the same meaning can still have a low semantic similarity score if their embeddings are bad.",[253,2023,2024,2025,2028],{},"Another drawback of this measurement is that the underlying embedding algorithm might require ",[130,2026,2027],{},"nontrivial compute and time"," to run.",[144,2030,2031],{},"Before we move on to discuss AI as a judge, let's go over a quick introduction to embedding. The concept of embedding lies at the heart of semantic similarity, and is the backbone of many topics we explore throughout the book, including vector search in Chapter 6 and data deduplication in Chapter 8.",[139,2033,2035],{"id":2034},"introduction-to-embedding","Introduction to Embedding",[144,2037,2038,2039],{},"Since computers work with numbers, a model needs to convert its input into numerical representations that computers can process. ",[268,2040,2041],{},"An embedding is a numerical representation that aims to capture the meaning of the original data.",[144,2043,2044,2045,2047,2048,2050,2051,133,2054,195],{},"An embedding is a vector. For example, the sentence ",[275,2046,1153],{}," might be represented using an embedding vector that looks like this: ",[275,2049,1157],{},". Here, I use a small vector as an example. In reality, the size of an embedding vector (the number of elements in the embedding vector) is typically between ",[275,2052,2053],{},"100",[275,2055,2056],{},"10,000",[175,2058,2059,2060,195],{},"While a 10,000-element vector space seems high-dimensional, it's much lower than the dimensionality of the raw data. An embedding is, therefore, considered a representation of complex data in a ",[130,2061,2062],{},"lower-dimensional space",[144,2064,2065,2066,2071],{},"Models trained especially to produce embeddings include the open source models BERT, CLIP (Contrastive Language–Image Pre-training), and ",[311,2067,2070],{"href":2068,"rel":2069},"https:\u002F\u002Fgithub.com\u002FUKPLab\u002Fsentence-transformers",[315],"Sentence Transformers",". There are also proprietary embedding models provided as APIs.",[175,2073,2074,2075,288,2080,2083,2084,2089],{},"There are also models that generate word embeddings, as opposed to documentation embeddings, such as word2vec (Mikolov et al., ",[311,2076,2079],{"href":2077,"rel":2078},"https:\u002F\u002Farxiv.org\u002Fabs\u002F1301.3781",[315],"\"Efficient Estimation of Word Representations in Vector Space\"",[268,2081,2082],{},"arXiv",", v3, September 7, 2013) and GloVe (Pennington et al., ",[311,2085,2088],{"href":2086,"rel":2087},"https:\u002F\u002Fnlp.stanford.edu\u002Fprojects\u002Fglove\u002F",[315],"\"GloVe: Global Vectors for Word Representation\"",", the Stanford University Natural Language Processing Group (blog), 2014).",[1098,2091,2092],{},[144,2093,2094],{},"Table 3-2. Embedding sizes used by common models.",[2096,2097,2098,2115],"table",{},[2099,2100,2101],"thead",{},[2102,2103,2104,2109,2112],"tr",{},[2105,2106,2108],"th",{"align":2107},"left","Provider \u002F Model",[2105,2110,2111],{"align":2107},"Variant",[2105,2113,2114],{"align":2107},"Embedding Size",[2116,2117,2118,2136,2146,2163,2172,2189,2199,2215],"tbody",{},[2102,2119,2120,2130,2133],{},[2121,2122,2123],"td",{"align":2107},[311,2124,2127],{"href":2125,"rel":2126},"https:\u002F\u002Farxiv.org\u002Fabs\u002F1810.04805",[315],[130,2128,2129],{},"Google's BERT",[2121,2131,2132],{"align":2107},"BERT base",[2121,2134,2135],{"align":2107},"768",[2102,2137,2138,2140,2143],{},[2121,2139],{"align":2107},[2121,2141,2142],{"align":2107},"BERT large",[2121,2144,2145],{"align":2107},"1024",[2102,2147,2148,2157,2160],{},[2121,2149,2150],{"align":2107},[311,2151,2154],{"href":2152,"rel":2153},"https:\u002F\u002Fopenai.com\u002Findex\u002Fclip\u002F",[315],[130,2155,2156],{},"OpenAI's CLIP",[2121,2158,2159],{"align":2107},"Image",[2121,2161,2162],{"align":2107},"512",[2102,2164,2165,2167,2170],{},[2121,2166],{"align":2107},[2121,2168,2169],{"align":2107},"Text",[2121,2171,2162],{"align":2107},[2102,2173,2174,2183,2186],{},[2121,2175,2176],{"align":2107},[311,2177,2180],{"href":2178,"rel":2179},"https:\u002F\u002Fdevelopers.openai.com\u002Fapi\u002Fdocs\u002Fguides\u002Fembeddings",[315],[130,2181,2182],{},"OpenAI Embeddings API",[2121,2184,2185],{"align":2107},"text-embedding-3-small",[2121,2187,2188],{"align":2107},"1536",[2102,2190,2191,2193,2196],{},[2121,2192],{"align":2107},[2121,2194,2195],{"align":2107},"text-embedding-3-large",[2121,2197,2198],{"align":2107},"3072",[2102,2200,2201,2210,2213],{},[2121,2202,2203],{"align":2107},[311,2204,2207],{"href":2205,"rel":2206},"https:\u002F\u002Fcohere.com\u002Fblog\u002Fintroducing-embed-v3",[315],[130,2208,2209],{},"Cohere's Embed v3",[2121,2211,2212],{"align":2107},"embed-english-v3.0",[2121,2214,2145],{"align":2107},[2102,2216,2217,2219,2222],{},[2121,2218],{"align":2107},[2121,2220,2221],{"align":2107},"embed-english-light-3.0",[2121,2223,2224],{"align":2107},"384",[144,2226,2227,2228,2231],{},"Because models typically require their inputs to first be transformed into vector representations, many ML models, including GPTs and Llamas, also involve a step to generate embeddings. ",[268,2229,2230],{},"\"Transformer architecture\""," visualizes the embedding layer in a transformer model. If you have access to the intermediate layers of these models, you can use them to extract embeddings. However, the quality of these embeddings might not be as good as the embeddings generated by specialized embedding models.",[144,2233,2234,2235,2237,2238,195],{},"The goal of the embedding algorithm is to produce embeddings that capture the essence of the original data. How do we verify that? The embedding vector ",[275,2236,1157],{}," looks nothing like the original text ",[275,2239,1153],{},[144,2241,2242,2243,2245,2246,2249,2250,195],{},"At a high level, an embedding algorithm is considered good if more-similar texts have closer embeddings, measured by cosine similarity or related metrics. The embedding of the sentence ",[275,2244,1153],{}," should be closer to the embedding of ",[275,2247,2248],{},"\"the dog plays on the grass\""," than the embedding of ",[275,2251,2252],{},"\"AI research is super fun\"",[144,2254,2255,2256,924],{},"You can also evaluate the quality of embeddings based on their utility for your task. Embeddings are used in many tasks, including classification, topic modeling, recommender systems, and RAG. An example of benchmarks that measure embedding quality on multiple tasks is MTEB, Massive Text Embedding Benchmark (",[311,2257,2260],{"href":2258,"rel":2259},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2210.07316",[315],"Muennighoff et al., 2023",[144,2262,2263,2264,133,2269,2274,2275,2280],{},"I use texts as examples, but any data can have embedding representations. For example, ecommerce solutions like ",[311,2265,2268],{"href":2266,"rel":2267},"https:\u002F\u002Farxiv.org\u002Fabs\u002F1607.07326",[315],"Criteo",[311,2270,2273],{"href":2271,"rel":2272},"https:\u002F\u002Fdocs.coveo.com\u002Fen\u002Fl9gg3565\u002Fcoveo-for-commerce\u002Fproduct-embeddings-and-vectors",[315],"Coveo"," have embeddings for products. ",[311,2276,2279],{"href":2277,"rel":2278},"https:\u002F\u002Foars-workshop.github.io\u002F2021\u002Fandrew.pdf",[315],"Pinterest"," has embeddings for images, graphs, queries, and even users.",[144,2282,2283,2284,2289,2290,2295,2296,2301],{},"A new frontier is to create joint embeddings for data of different modalities. CLIP (",[311,2285,2288],{"href":2286,"rel":2287},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2103.00020",[315],"Radford et al., 2021",") was one of the first major models that could map data of different modalities, text and images, into a joint embedding space. ULIP (unified representation of language, images, and point clouds) (",[311,2291,2294],{"href":2292,"rel":2293},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2212.05171",[315],"Xue et al., 2022",") aims to create unified representations of text, images, and 3D point clouds. ImageBind (",[311,2297,2300],{"href":2298,"rel":2299},"https:\u002F\u002Farxiv.org\u002Fabs\u002F2305.05665",[315],"Girdhar et al., 2023",") learns a joint embedding across six different modalities, including text, images, and audio.",[144,2303,2304],{},"Figure 3-6 visualizes CLIP's architecture. CLIP is trained using (image, text) pairs. The text corresponding to an image can be the caption or a comment associated with this image.",[720,2306,2307,2311,2314,2318,2321,2325],{"level":722},[724,2308,2310],{"id":2309},"encode-each-modality","Encode each modality",[144,2312,2313],{},"For each (image, text) pair, CLIP uses a text encoder to convert the text to a text embedding, and an image encoder to convert the image to an image embedding.",[724,2315,2317],{"id":2316},"project-into-a-joint-space","Project into a joint space",[144,2319,2320],{},"It then projects both these embeddings into a joint embedding space.",[724,2322,2324],{"id":2323},"pull-matching-pairs-together","Pull matching pairs together",[144,2326,2327],{},"The training goal is to get the embedding of an image close to the embedding of the corresponding text in this joint space.",[144,2329,2330],{},[1093,2331],{"alt":2332,"src":2333},"Figure 3-6. CLIP's architecture (Radford et al., 2021).",".\u002Fmedia\u002Ffig-3-6.png",[1098,2335,2336],{},[144,2337,2332],{},[144,2339,2340,2341,2344,2345,2348,2349,2352],{},"A joint embedding space that can represent data of different modalities is a ",[268,2342,2343],{},"multimodal embedding space",". In a text–image joint embedding space, the embedding of an image of a man fishing should be closer to the embedding of the text ",[275,2346,2347],{},"\"a fisherman\""," than the embedding of the text ",[275,2350,2351],{},"\"fashion show\"",". This joint embedding space allows embeddings of different modalities to be compared and combined. For example, this enables text-based image search. Given a text, it helps you find images closest to this text.",[2354,2355,2356],"style",{},"html .light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html.light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}",{"title":125,"searchDepth":375,"depth":375,"links":2358},[2359,2360,2361,2366,2371],{"id":141,"depth":375,"text":142},{"id":184,"depth":375,"text":185},{"id":228,"depth":375,"text":201,"children":2362},[2363,2364,2365],{"id":262,"depth":382,"text":263},{"id":352,"depth":382,"text":353},{"id":651,"depth":382,"text":652},{"id":665,"depth":375,"text":666,"children":2367},[2368,2369,2370],{"id":800,"depth":382,"text":801},{"id":895,"depth":382,"text":896},{"id":1124,"depth":382,"text":1125},{"id":2034,"depth":375,"text":2035},"How functional correctness, similarity against reference data, and embeddings produce exact scores for open-ended model outputs.","md",{},{"icon":100},{"title":97,"description":2372},"MZRKcVv70GlXQCQfo5rjLdYuUypuYZANTclPRufR6To",[2379,2381],{"title":92,"path":93,"stem":94,"description":2380,"icon":95,"children":-1},"How entropy, cross entropy, perplexity, and BPC\u002FBPB measure a language model's predictive accuracy and how they relate.",{"title":102,"path":103,"stem":104,"description":2382,"icon":105,"children":-1},"Why AI judges took off, how to prompt them, their limits (inconsistency, cost, bias), and which models can judge.",1788871975940]