[{"data":1,"prerenderedAt":4413},["ShallowReactive",2],{"navigation_docs":3,"-data-ops-llm-ops-evaluations":369,"-data-ops-llm-ops-evaluations-surround":4409},[4,8,72,102,220,266,280,301,365],{"title":5,"path":6,"stem":7},"Introduction","\u002Fintroduction","0.introduction",{"title":9,"icon":10,"path":11,"stem":12,"children":13,"page":67},"Company","i-lucide-building-2","\u002Fcompany","1.company",[14,18,22,26,30,34,38,42,46,50,54,68],{"title":15,"path":16,"stem":17},"About","\u002Fcompany\u002Fabout","1.company\u002F0.about",{"title":19,"path":20,"stem":21},"Values","\u002Fcompany\u002Fvalues","1.company\u002F1.values",{"title":23,"path":24,"stem":25},"Communication","\u002Fcompany\u002Fcommunication","1.company\u002Fcommunication",{"title":27,"path":28,"stem":29},"Competition","\u002Fcompany\u002Fcompetition","1.company\u002Fcompetition",{"title":31,"path":32,"stem":33},"Hybrid Working","\u002Fcompany\u002Fhybrid-working","1.company\u002Fhybrid-working",{"title":35,"path":36,"stem":37},"Manchester Office","\u002Fcompany\u002Foffice","1.company\u002Foffice",{"title":39,"path":40,"stem":41},"Operations","\u002Fcompany\u002Foperations","1.company\u002Foperations",{"title":43,"path":44,"stem":45},"Policies","\u002Fcompany\u002Fpolicies","1.company\u002Fpolicies",{"title":47,"path":48,"stem":49},"Problem Statement","\u002Fcompany\u002Fproblem-statement","1.company\u002Fproblem-statement",{"title":51,"path":52,"stem":53},"Product Strategy","\u002Fcompany\u002Fproduct-strategy","1.company\u002Fproduct-strategy",{"title":55,"path":56,"stem":57,"children":58,"page":67},"Products","\u002Fcompany\u002Fproducts","1.company\u002Fproducts",[59,63],{"title":60,"path":61,"stem":62},"Capability Exchange","\u002Fcompany\u002Fproducts\u002Fcapability-exchange","1.company\u002Fproducts\u002Fcapability-exchange",{"title":64,"path":65,"stem":66},"ESProfiler Platform","\u002Fcompany\u002Fproducts\u002Fesprofiler","1.company\u002Fproducts\u002Fesprofiler",false,{"title":69,"path":70,"stem":71},"Security","\u002Fcompany\u002Fsecurity","1.company\u002Fsecurity",{"title":73,"icon":74,"path":75,"stem":76,"children":77,"page":67},"People Ops","i-lucide-users","\u002Fpeople-ops","2.people-ops",[78,82,86,90,94,98],{"title":79,"path":80,"stem":81},"Compensation","\u002Fpeople-ops\u002Fcompensation","2.people-ops\u002Fcompensation",{"title":83,"path":84,"stem":85},"Education","\u002Fpeople-ops\u002Feducation","2.people-ops\u002Feducation",{"title":87,"path":88,"stem":89},"Expenses","\u002Fpeople-ops\u002Fexpenses","2.people-ops\u002Fexpenses",{"title":91,"path":92,"stem":93},"Holiday & Leave","\u002Fpeople-ops\u002Fleave","2.people-ops\u002Fleave",{"title":95,"path":96,"stem":97},"Onboarding","\u002Fpeople-ops\u002Fonboarding","2.people-ops\u002Fonboarding",{"title":99,"path":100,"stem":101},"Recruitment","\u002Fpeople-ops\u002Frecruitment","2.people-ops\u002Frecruitment",{"title":103,"icon":104,"path":105,"stem":106,"children":107,"page":67},"Engineering","i-lucide-rocket","\u002Fengineering","3.engineering",[108,151,155,175,196,200,208,212,216],{"title":109,"path":110,"stem":111,"children":112,"page":67},"Contributing","\u002Fengineering\u002Fcontributing","3.engineering\u002Fcontributing",[113,117,121,125,129,142],{"title":114,"path":115,"stem":116},"Development Setup","\u002Fengineering\u002Fcontributing\u002Fdevelopment-setup","3.engineering\u002Fcontributing\u002F1.development-setup",{"title":118,"path":119,"stem":120},"Engineering Operations","\u002Fengineering\u002Fcontributing\u002Fengineering-operations","3.engineering\u002Fcontributing\u002F2.engineering-operations",{"title":122,"path":123,"stem":124},"Documentation","\u002Fengineering\u002Fcontributing\u002Fdocumentation","3.engineering\u002Fcontributing\u002F3.documentation",{"title":126,"path":127,"stem":128},"Agentic Coding","\u002Fengineering\u002Fcontributing\u002Fagentic-coding","3.engineering\u002Fcontributing\u002Fagentic-coding",{"title":130,"path":131,"stem":132,"children":133,"page":67},"Back End","\u002Fengineering\u002Fcontributing\u002Fback-end","3.engineering\u002Fcontributing\u002Fback-end",[134,138],{"title":135,"path":136,"stem":137},"API Guidelines","\u002Fengineering\u002Fcontributing\u002Fback-end\u002Fapi-guidelines","3.engineering\u002Fcontributing\u002Fback-end\u002Fapi-guidelines",{"title":139,"path":140,"stem":141},"LLM Prompts & Langfuse Integration","\u002Fengineering\u002Fcontributing\u002Fback-end\u002Fllm-prompts","3.engineering\u002Fcontributing\u002Fback-end\u002Fllm-prompts",{"title":143,"path":144,"stem":145,"children":146,"page":67},"Front End","\u002Fengineering\u002Fcontributing\u002Ffront-end","3.engineering\u002Fcontributing\u002Ffront-end",[147],{"title":148,"path":149,"stem":150},"Testing","\u002Fengineering\u002Fcontributing\u002Ffront-end\u002Ftesting","3.engineering\u002Fcontributing\u002Ffront-end\u002Ftesting",{"title":152,"path":153,"stem":154},"Production Database","\u002Fengineering\u002Fdatabase-connection","3.engineering\u002Fdatabase-connection",{"title":156,"path":157,"stem":158,"children":159},"Deployment","\u002Fengineering\u002Fdeployment","3.engineering\u002Fdeployment",[160,163,167,171],{"title":60,"path":161,"stem":162},"\u002Fengineering\u002Fdeployment\u002Fcapability-exchange","3.engineering\u002Fdeployment\u002Fcapability-exchange",{"title":164,"path":165,"stem":166},"Langfuse Deployment","\u002Fengineering\u002Fdeployment\u002Fecs-langfuse-deployment","3.engineering\u002Fdeployment\u002Fecs-langfuse-deployment",{"title":168,"path":169,"stem":170},"ESP Platform Configuration","\u002Fengineering\u002Fdeployment\u002Fesp-platform-configuration","3.engineering\u002Fdeployment\u002Fesp-platform-configuration",{"title":172,"path":173,"stem":174},"Platform","\u002Fengineering\u002Fdeployment\u002Fplatform","3.engineering\u002Fdeployment\u002Fplatform",{"title":176,"path":177,"stem":178,"children":179,"page":67},"Github","\u002Fengineering\u002Fgithub","3.engineering\u002Fgithub",[180,184,188,192],{"title":181,"path":182,"stem":183},"Packages","\u002Fengineering\u002Fgithub\u002Fpackages","3.engineering\u002Fgithub\u002Fpackages",{"title":185,"path":186,"stem":187},"Personal Access Token","\u002Fengineering\u002Fgithub\u002Fpersonal-access-token","3.engineering\u002Fgithub\u002Fpersonal-access-token",{"title":189,"path":190,"stem":191},"Troubleshooting","\u002Fengineering\u002Fgithub\u002Ftroubleshooting","3.engineering\u002Fgithub\u002Ftroubleshooting",{"title":193,"path":194,"stem":195},"Workflows","\u002Fengineering\u002Fgithub\u002Fworkflows","3.engineering\u002Fgithub\u002Fworkflows",{"title":197,"path":198,"stem":199},"Platform Ops","\u002Fengineering\u002Fplatform-ops","3.engineering\u002Fplatform-ops",{"title":172,"path":201,"stem":202,"children":203,"page":67},"\u002Fengineering\u002Fplatform","3.engineering\u002Fplatform",[204],{"title":205,"path":206,"stem":207},"useAPI","\u002Fengineering\u002Fplatform\u002Fuse-api","3.engineering\u002Fplatform\u002Fuse-api",{"title":209,"path":210,"stem":211},"Project Management","\u002Fengineering\u002Fproject-management","3.engineering\u002Fproject-management",{"title":213,"path":214,"stem":215},"Releases","\u002Fengineering\u002Frelease","3.engineering\u002Frelease",{"title":217,"path":218,"stem":219},"Tools","\u002Fengineering\u002Ftools","3.engineering\u002Ftools",{"title":221,"icon":222,"path":223,"stem":224,"children":225,"page":67},"Design","i-lucide-palette","\u002Fdesign","4.design",[226,247,251,255,259,262],{"title":227,"path":228,"stem":229,"children":230,"page":67},"Personas","\u002Fdesign\u002Fpersonas","4.design\u002F0.personas",[231,235,239,243],{"title":232,"path":233,"stem":234},"CISO - Simon","\u002Fdesign\u002Fpersonas\u002F01-ciso-simon","4.design\u002F0.personas\u002F01-ciso-simon",{"title":236,"path":237,"stem":238},"Head of Cyber - Harry","\u002Fdesign\u002Fpersonas\u002F02-head-of-cyber-harry","4.design\u002F0.personas\u002F02-head-of-cyber-harry",{"title":240,"path":241,"stem":242},"Security Architect - Sasha","\u002Fdesign\u002Fpersonas\u002F03-security-architect-sasha","4.design\u002F0.personas\u002F03-security-architect-sasha",{"title":244,"path":245,"stem":246},"Procurement Officer - Paige","\u002Fdesign\u002Fpersonas\u002F04-procurement-officer-paige","4.design\u002F0.personas\u002F04-procurement-officer-paige",{"title":248,"path":249,"stem":250},"Design Thinking","\u002Fdesign\u002Fdesign-thinking","4.design\u002F1.design-thinking",{"title":252,"path":253,"stem":254},"Design & Development","\u002Fdesign\u002Fdesign-and-development","4.design\u002F3.design-and-development",{"title":256,"path":257,"stem":258},"Branding","\u002Fdesign\u002Fbranding","4.design\u002F4.branding",{"title":217,"path":260,"stem":261},"\u002Fdesign\u002Ftools","4.design\u002F5.tools",{"title":263,"path":264,"stem":265},"Customer Success","\u002Fdesign\u002Fworking-with-customers","4.design\u002F6.working-with-customers",{"title":267,"icon":268,"path":269,"stem":270,"children":271,"page":67},"Sales","i-lucide-dollar-sign","\u002Fsales","4.sales",[272,276],{"title":273,"path":274,"stem":275},"Customer Onboarding","\u002Fsales\u002Fonboarding","4.sales\u002Fonboarding",{"title":277,"path":278,"stem":279},"Sales Tools","\u002Fsales\u002Ftools","4.sales\u002Ftools",{"title":281,"icon":282,"path":283,"stem":284,"children":285,"page":67},"Marketing","i-lucide-book-image","\u002Fmarketing","5.marketing",[286,290,294,297],{"title":287,"path":288,"stem":289},"Content","\u002Fmarketing\u002Fcontent","5.marketing\u002Fcontent",{"title":291,"path":292,"stem":293},"Messaging","\u002Fmarketing\u002Fmessaging","5.marketing\u002Fmessaging",{"title":217,"path":295,"stem":296},"\u002Fmarketing\u002Ftools","5.marketing\u002Ftools",{"title":298,"path":299,"stem":300},"Website","\u002Fmarketing\u002Fwebsite","5.marketing\u002Fwebsite",{"title":302,"icon":303,"path":304,"stem":305,"children":306,"page":67},"AI & Data Ops","i-lucide-database","\u002Fdata-ops","6.data-ops",[307,315,319,344,361],{"title":60,"path":308,"stem":309,"children":310,"page":67},"\u002Fdata-ops\u002Fcapability-exchange","6.data-ops\u002FCapability Exchange",[311],{"title":312,"path":313,"stem":314},"Leaderboard Calculation","\u002Fdata-ops\u002Fcapability-exchange\u002Fleaderboard-calculation","6.data-ops\u002FCapability Exchange\u002Fleaderboard-calculation",{"title":316,"path":317,"stem":318},"Account Portal (CAS)","\u002Fdata-ops\u002Faccount-portal","6.data-ops\u002Faccount-portal",{"title":320,"path":321,"stem":322,"children":323,"page":67},"Data Management","\u002Fdata-ops\u002Fdata-management","6.data-ops\u002Fdata-management",[324,328,332,336,340],{"title":325,"path":326,"stem":327},"Adding Products","\u002Fdata-ops\u002Fdata-management\u002Fadding-products","6.data-ops\u002Fdata-management\u002Fadding-products",{"title":329,"path":330,"stem":331},"Adding Vendors","\u002Fdata-ops\u002Fdata-management\u002Fadding-vendors","6.data-ops\u002Fdata-management\u002Fadding-vendors",{"title":333,"path":334,"stem":335},"Framework Mapping","\u002Fdata-ops\u002Fdata-management\u002Fframework-mapping","6.data-ops\u002Fdata-management\u002Fframework-mapping",{"title":337,"path":338,"stem":339},"Refreshing Vendors","\u002Fdata-ops\u002Fdata-management\u002Frefreshing-vendors","6.data-ops\u002Fdata-management\u002Frefreshing-vendors",{"title":341,"path":342,"stem":343},"Reviewing Draft Vendors","\u002Fdata-ops\u002Fdata-management\u002Freviewing-draft-vendors","6.data-ops\u002Fdata-management\u002Freviewing-draft-vendors",{"title":345,"path":346,"stem":347,"children":348,"page":67},"LLM Ops","\u002Fdata-ops\u002Fllm-ops","6.data-ops\u002Fllm-ops",[349,353,357],{"title":350,"path":351,"stem":352},"Agents","\u002Fdata-ops\u002Fllm-ops\u002Fagents","6.data-ops\u002Fllm-ops\u002F1.agents",{"title":354,"path":355,"stem":356},"ESPi Architecture & Query Flow","\u002Fdata-ops\u002Fllm-ops\u002Fespi-architecture","6.data-ops\u002Fllm-ops\u002F2.espi-architecture",{"title":358,"path":359,"stem":360},"Evaluating Agents","\u002Fdata-ops\u002Fllm-ops\u002Fevaluations","6.data-ops\u002Fllm-ops\u002F3.evaluations",{"title":362,"path":363,"stem":364},"Message Queues","\u002Fdata-ops\u002Fmessage-queues","6.data-ops\u002Fmessage-queues",{"title":366,"path":367,"stem":368},"Glossary","\u002Fglossary","glossary",{"id":370,"title":358,"body":371,"description":4403,"extension":4404,"links":4405,"meta":4406,"navigation":2429,"path":359,"seo":4407,"stem":360,"__hash__":4408},"docs\u002F6.data-ops\u002Fllm-ops\u002F3.evaluations.md",{"type":372,"value":373,"toc":4380},"minimark",[374,386,389,433,483,486,513,517,520,525,528,573,578,621,624,628,638,679,688,691,695,720,724,2642,2644,2648,2651,2658,2680,2683,2697,2701,2744,2751,2754,2777,2781,2787,2796,2800,2849,2857,2861,3352,3354,3358,3372,3381,3395,3399,3468,3474,3484,3488,3491,3531,3535,3611,3614,3618,3624,3627,3639,3654,3658,3661,3674,3677,3681,4328,4330,4334,4376],[375,376,377,378,385],"p",{},"We use ",[379,380,384],"a",{"href":381,"rel":382},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development",[383],"nofollow","Langfuse"," to evaluate LLM agents in the ESProfiler ecosystem.",[375,387,388],{},"Evaluations follow three steps:",[390,391,392,415,424],"ol",{},[393,394,395,402,403,407,408,407,411,414],"li",{},[396,397,398],"strong",{},[379,399,401],{"href":400},"#1-creating-datasets","Create datasets"," — test cases (",[404,405,406],"code",{},"input"," \u002F ",[404,409,410],{},"expectedOutput",[404,412,413],{},"metadata",")",[393,416,417,423],{},[396,418,419],{},[379,420,422],{"href":421},"#2-creating-evaluators","Create evaluators"," — scoring definitions (mostly LLM-as-judge)",[393,425,426,432],{},[396,427,428],{},[379,429,431],{"href":430},"#3-running-experiments","Run experiments"," — prompt + dataset + evaluators",[434,435,436,449],"table",{},[437,438,439],"thead",{},[440,441,442,446],"tr",{},[443,444,445],"th",{},"Piece",[443,447,448],{},"Role",[450,451,452,463,473],"tbody",{},[440,453,454,460],{},[455,456,457],"td",{},[396,458,459],{},"Dataset",[455,461,462],{},"Reusable test cases",[440,464,465,470],{},[455,466,467],{},[396,468,469],{},"Evaluator",[455,471,472],{},"Scores one quality dimension of an agent output",[440,474,475,480],{},[455,476,477],{},[396,478,479],{},"Experiment",[455,481,482],{},"Runs a prompt against a dataset and applies evaluators",[375,484,485],{},"Existing resources in ESP Development:",[487,488,489,496,503],"ul",{},[393,490,491],{},[379,492,495],{"href":493,"rel":494},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development\u002Fdatasets?pageIndex=0&pageSize=50",[383],"Datasets",[393,497,498],{},[379,499,502],{"href":500,"rel":501},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development\u002Fevals\u002Ftemplates",[383],"Evaluator templates",[393,504,505],{},[379,506,509,510],{"href":507,"rel":508},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development\u002Fdatasets\u002Fcmruuouiv0021mk07r7itcz46\u002Fexperiments",[383],"Example experiments: ",[404,511,512],{},"findings\u002Fgate_test",[514,515,516],"note",{},"A walkthrough video of the full Langfuse UI will be recorded soon and added to this page later.",[518,519],"hr",{},[521,522,524],"h2",{"id":523},"_1-creating-datasets","1. Creating Datasets",[375,526,527],{},"An evaluation dataset is a collection of test cases. Each item typically has:",[434,529,530,540],{},[437,531,532],{},[440,533,534,537],{},[443,535,536],{},"Field",[443,538,539],{},"Purpose",[450,541,542,551,564],{},[440,543,544,548],{},[455,545,546],{},[396,547,406],{},[455,549,550],{},"What the agent\u002Fprompt receives at runtime",[440,552,553,557],{},[455,554,555],{},[396,556,410],{},[455,558,559,560,563],{},"Golden answer ",[396,561,562],{},"or"," evaluation guidance",[440,565,566,570],{},[455,567,568],{},[396,569,413],{},[455,571,572],{},"Filtering, debugging, and audit context (not fed to the agent)",[574,575,577],"h3",{"id":576},"_11-create-your-first-dataset-5-steps","1.1 Create your first dataset (5 steps)",[390,579,580,589,595,609,615],{},[393,581,582,585,586,588],{},[396,583,584],{},"Pick one agent"," — see ",[379,587,350],{"href":351},".",[393,590,591,594],{},[396,592,593],{},"Decide expected-output style"," — golden answer for deterministic tasks; judge guidance for open-ended ones.",[393,596,597,600,601,603,604,606,607,588],{},[396,598,599],{},"Write 3–5 items"," — ",[404,602,406],{}," keys must match that agent’s prompt variables; add ",[404,605,410],{}," and optional ",[404,608,413],{},[393,610,611,614],{},[396,612,613],{},"Anonymize if data came from production"," — never upload raw tenant\u002FPII.",[393,616,617,620],{},[396,618,619],{},"Create and upload"," — UI for tiny flat cases; Python SDK for nested JSON.",[375,622,623],{},"Then spot-check 2–3 items in Langfuse before attaching evaluators.",[574,625,627],{"id":626},"_12-dataset-naming","1.2 Dataset naming",[629,630,636],"pre",{"className":631,"code":633,"language":634,"meta":635},[632],"language-text","{agentName}\u002F{datasetRole}\n","text","",[404,637,633],{"__ignoreMap":635},[434,639,640,651],{},[437,641,642],{},[440,643,644,646,648],{},[443,645,459],{},[443,647,448],{},[443,649,650],{},"Size guidance",[450,652,653,666],{},[440,654,655,660,663],{},[455,656,657],{},[404,658,659],{},"{agent}\u002Fsmoke_test",[455,661,662],{},"Quick sanity checks after prompt or wiring changes",[455,664,665],{},"~5–15 items",[440,667,668,673,676],{},[455,669,670],{},[404,671,672],{},"{agent}\u002Fgate_test",[455,674,675],{},"Broader frozen set for prompt comparison and release decisions",[455,677,678],{},"~20–50 items",[375,680,681,682,685,686,588],{},"Examples: ",[404,683,684],{},"conversation-namer\u002Fsmoke_test",", ",[404,687,512],{},[375,689,690],{},"Keep both sets reviewed and anonymized. Update them deliberately — do not treat one as a staging queue for the other.",[574,692,694],{"id":693},"_13-upload-overview","1.3 Upload overview",[487,696,697,703,709],{},[393,698,699,702],{},[396,700,701],{},"UI"," — Datasets → New dataset → add items (or CSV for flat strings). Best for small \u002F simple cases.",[393,704,705,708],{},[396,706,707],{},"Python SDK"," — preferred for nested JSON (Findings transcripts, structured guidance). Keys: Langfuse UI → Settings → API Keys.",[393,710,711,714,715],{},[396,712,713],{},"Reference:"," ",[379,716,719],{"href":717,"rel":718},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fexperiments\u002Fdatasets",[383],"Langfuse Datasets",[574,721,723],{"id":722},"_14-dataset-reference-expand-as-needed","1.4 Dataset reference (expand as needed)",[725,726,727,797,850,882,1070,1272,2363,2573],"accordion",{},[728,729,732,736,743,758,762,768,777,787,790],"accordion-item",{"icon":730,"label":731},"i-lucide-table","1.4.1 Field details (input \u002F expectedOutput \u002F metadata)",[733,734,735],"h4",{"id":406},"Input",[375,737,738,739,742],{},"What is injected into the prompt or application under test: a user message, nested ",[396,740,741],{},"prompt variables",", or a multi-turn conversation.",[375,744,745,714,748,750,751,754,755,588],{},[396,746,747],{},"Rule:",[404,749,406],{}," keys must match the prompt variables of the agent you are evaluating. Find those in the agent’s ",[404,752,753],{},".st"," prompt \u002F config in ",[404,756,757],{},"platform-api",[733,759,761],{"id":760},"expected-output","Expected output",[375,763,764,767],{},[396,765,766],{},"A) Golden answer (deterministic)"," — use when the correct answer is known and comparable (tags, short titles, exact fields). A code check is often enough. Curate with human review.",[375,769,770,773,774,776],{},[396,771,772],{},"B) Judge guidance (non-deterministic)"," — use when many good outputs exist (summaries, reports, findings). Put criteria in ",[404,775,410],{}," (what must be covered, constraints, grounding rules) — not a full golden dump.",[514,778,779,780,782,783,786],{},"For Findings-style agents: keep guidance in ",[404,781,410],{},"; put full historical report\u002Ffindings in ",[404,784,785],{},"metadata.reference_output"," for human review only.",[733,788,789],{"id":413},"Metadata",[375,791,792,793,796],{},"Anything that should ",[396,794,795],{},"not"," be fed as prompt input: tags, tool expectations, anonymization flags, debug goldens.",[728,798,801],{"icon":799,"label":800},"i-lucide-git-compare","1.4.2 Choose evaluation style",[434,802,803,815],{},[437,804,805],{},[440,806,807,810,813],{},[443,808,809],{},"Use case",[443,811,812],{},"Expected output style",[443,814,469],{},[450,816,817,828,839],{},[440,818,819,822,825],{},[455,820,821],{},"Categorization \u002F tagging \u002F short titles",[455,823,824],{},"Golden labels",[455,826,827],{},"Code check (app or unit tests)",[440,829,830,833,836],{},[455,831,832],{},"Tone, grounding, alignment, finding types",[455,834,835],{},"Guidance + rubric",[455,837,838],{},"LLM-as-judge",[440,840,841,844,847],{},[455,842,843],{},"Mixed",[455,845,846],{},"Guidance + optional reference in metadata",[455,848,849],{},"Code + LLM-as-judge",[728,851,854,857],{"icon":852,"label":853},"i-lucide-list-filter","1.4.3 How to select records",[375,855,856],{},"Start from production failure modes, not random sampling.",[390,858,859,862,865,868,871],{},[393,860,861],{},"Observe where the agent fails (wrong types, invented claims, missed coverage, etc.)",[393,863,864],{},"Pull representative completed cases for those modes",[393,866,867],{},"Prefer diverse scenarios over near-duplicates",[393,869,870],{},"Anonymize before upload",[393,872,873,874,877,878,881],{},"Put stable regression cases in ",[404,875,876],{},"gate_test","; keep a smaller ",[404,879,880],{},"smoke_test"," set for quick checks",[728,883,886,891,965,968],{"icon":884,"label":885},"i-lucide-shield","1.4.4 Data anonymization (mandatory for production data)",[375,887,888],{},[396,889,890],{},"Do not upload raw tenant data to Langfuse.",[434,892,893,903],{},[437,894,895],{},[440,896,897,900],{},[443,898,899],{},"Original",[443,901,902],{},"Anonymized form",[450,904,905,916,924,935,943,954],{},[440,906,907,910],{},[455,908,909],{},"Tenant \u002F org name",[455,911,912,913,414],{},"Synthetic org (e.g. ",[396,914,915],{},"Black Mesa",[440,917,918,921],{},[455,919,920],{},"Real people",[455,922,923],{},"Stable synthetic names",[440,925,926,929],{},[455,927,928],{},"Emails \u002F tenant domains",[455,930,931,934],{},[404,932,933],{},"@blackmesa.example"," placeholders",[440,936,937,940],{},[455,938,939],{},"Real task \u002F product \u002F vendor UUIDs",[455,941,942],{},"Synthetic UUIDs from hashes",[440,944,945,948],{},[455,946,947],{},"Original task id",[455,949,950,953],{},[404,951,952],{},"original_task_id_hash"," only",[440,955,956,959],{},[455,957,958],{},"Images \u002F screenshots",[455,960,961,962,414],{},"Stripped (",[404,963,964],{},"imageStr: null",[375,966,967],{},"Replace tenant strings everywhere, keep people mapping stable across items, keep market product names only when they are not the customer identity, then sanity-check that known customer tokens are gone.",[629,969,973],{"className":970,"code":971,"language":972,"meta":635,"style":635},"language-json shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","{\n  \"anonymization\": {\n    \"tenant_replaced_with\": \"Black Mesa\",\n    \"people_synthetic\": true,\n    \"images_stripped\": true\n  }\n}\n","json",[404,974,975,984,1003,1028,1043,1058,1064],{"__ignoreMap":635},[976,977,980],"span",{"class":978,"line":979},"line",1,[976,981,983],{"class":982},"sMK4o","{\n",[976,985,987,990,994,997,1000],{"class":978,"line":986},2,[976,988,989],{"class":982},"  \"",[976,991,993],{"class":992},"spNyl","anonymization",[976,995,996],{"class":982},"\"",[976,998,999],{"class":982},":",[976,1001,1002],{"class":982}," {\n",[976,1004,1006,1009,1013,1015,1017,1020,1023,1025],{"class":978,"line":1005},3,[976,1007,1008],{"class":982},"    \"",[976,1010,1012],{"class":1011},"sBMFI","tenant_replaced_with",[976,1014,996],{"class":982},[976,1016,999],{"class":982},[976,1018,1019],{"class":982}," \"",[976,1021,915],{"class":1022},"sfazB",[976,1024,996],{"class":982},[976,1026,1027],{"class":982},",\n",[976,1029,1031,1033,1036,1038,1040],{"class":978,"line":1030},4,[976,1032,1008],{"class":982},[976,1034,1035],{"class":1011},"people_synthetic",[976,1037,996],{"class":982},[976,1039,999],{"class":982},[976,1041,1042],{"class":982}," true,\n",[976,1044,1046,1048,1051,1053,1055],{"class":978,"line":1045},5,[976,1047,1008],{"class":982},[976,1049,1050],{"class":1011},"images_stripped",[976,1052,996],{"class":982},[976,1054,999],{"class":982},[976,1056,1057],{"class":982}," true\n",[976,1059,1061],{"class":978,"line":1060},6,[976,1062,1063],{"class":982},"  }\n",[976,1065,1067],{"class":978,"line":1066},7,[976,1068,1069],{"class":982},"}\n",[728,1071,1074,1084],{"icon":1072,"label":1073},"i-lucide-file-json","1.4.5 Example: Conversation Namer item",[375,1075,1076,1077,1079,1080,1083],{},"First user message in → short title out. Confirm the real prompt variable name in ",[404,1078,757],{}," before uploading (replace ",[404,1081,1082],{},"message"," if it differs). Synthetic cases need no anonymization.",[629,1085,1087],{"className":970,"code":1086,"language":972,"meta":635,"style":635},"{\n  \"input\": {\n    \"message\": \"Can you compare our CrowdStrike and Wiz renewals for next quarter?\"\n  },\n  \"expectedOutput\": {\n    \"title\": \"CrowdStrike vs Wiz renewals\"\n  },\n  \"metadata\": {\n    \"dataset\": \"conversation-namer\u002Fgate_test\",\n    \"tags\": [\"conversation_namer\", \"synthetic\"],\n    \"constraints\": { \"max_words\": 5 }\n  }\n}\n",[404,1088,1089,1093,1105,1123,1128,1140,1158,1162,1175,1196,1231,1262,1267],{"__ignoreMap":635},[976,1090,1091],{"class":978,"line":979},[976,1092,983],{"class":982},[976,1094,1095,1097,1099,1101,1103],{"class":978,"line":986},[976,1096,989],{"class":982},[976,1098,406],{"class":992},[976,1100,996],{"class":982},[976,1102,999],{"class":982},[976,1104,1002],{"class":982},[976,1106,1107,1109,1111,1113,1115,1117,1120],{"class":978,"line":1005},[976,1108,1008],{"class":982},[976,1110,1082],{"class":1011},[976,1112,996],{"class":982},[976,1114,999],{"class":982},[976,1116,1019],{"class":982},[976,1118,1119],{"class":1022},"Can you compare our CrowdStrike and Wiz renewals for next quarter?",[976,1121,1122],{"class":982},"\"\n",[976,1124,1125],{"class":978,"line":1030},[976,1126,1127],{"class":982},"  },\n",[976,1129,1130,1132,1134,1136,1138],{"class":978,"line":1045},[976,1131,989],{"class":982},[976,1133,410],{"class":992},[976,1135,996],{"class":982},[976,1137,999],{"class":982},[976,1139,1002],{"class":982},[976,1141,1142,1144,1147,1149,1151,1153,1156],{"class":978,"line":1060},[976,1143,1008],{"class":982},[976,1145,1146],{"class":1011},"title",[976,1148,996],{"class":982},[976,1150,999],{"class":982},[976,1152,1019],{"class":982},[976,1154,1155],{"class":1022},"CrowdStrike vs Wiz renewals",[976,1157,1122],{"class":982},[976,1159,1160],{"class":978,"line":1066},[976,1161,1127],{"class":982},[976,1163,1165,1167,1169,1171,1173],{"class":978,"line":1164},8,[976,1166,989],{"class":982},[976,1168,413],{"class":992},[976,1170,996],{"class":982},[976,1172,999],{"class":982},[976,1174,1002],{"class":982},[976,1176,1178,1180,1183,1185,1187,1189,1192,1194],{"class":978,"line":1177},9,[976,1179,1008],{"class":982},[976,1181,1182],{"class":1011},"dataset",[976,1184,996],{"class":982},[976,1186,999],{"class":982},[976,1188,1019],{"class":982},[976,1190,1191],{"class":1022},"conversation-namer\u002Fgate_test",[976,1193,996],{"class":982},[976,1195,1027],{"class":982},[976,1197,1199,1201,1204,1206,1208,1211,1213,1216,1218,1221,1223,1226,1228],{"class":978,"line":1198},10,[976,1200,1008],{"class":982},[976,1202,1203],{"class":1011},"tags",[976,1205,996],{"class":982},[976,1207,999],{"class":982},[976,1209,1210],{"class":982}," [",[976,1212,996],{"class":982},[976,1214,1215],{"class":1022},"conversation_namer",[976,1217,996],{"class":982},[976,1219,1220],{"class":982},",",[976,1222,1019],{"class":982},[976,1224,1225],{"class":1022},"synthetic",[976,1227,996],{"class":982},[976,1229,1230],{"class":982},"],\n",[976,1232,1234,1236,1239,1241,1243,1246,1248,1252,1254,1256,1259],{"class":978,"line":1233},11,[976,1235,1008],{"class":982},[976,1237,1238],{"class":1011},"constraints",[976,1240,996],{"class":982},[976,1242,999],{"class":982},[976,1244,1245],{"class":982}," {",[976,1247,1019],{"class":982},[976,1249,1251],{"class":1250},"sbssI","max_words",[976,1253,996],{"class":982},[976,1255,999],{"class":982},[976,1257,1258],{"class":1250}," 5",[976,1260,1261],{"class":982}," }\n",[976,1263,1265],{"class":978,"line":1264},12,[976,1266,1063],{"class":982},[976,1268,1270],{"class":978,"line":1269},13,[976,1271,1069],{"class":982},[728,1273,1275,1278,1297,1300,1307,1765,1772,2099,2106],{"icon":1072,"label":1274},"1.4.6 Example: Findings Agent shapes",[375,1276,1277],{},"Turns interview transcript + directive into summary, report, and structured findings.",[375,1279,1280,1283,1284,1286,1287,1290,1291,1293,1294,1296],{},[396,1281,1282],{},"Process:"," pull production cases → map prompt variables to ",[404,1285,406],{}," → anonymize → put ",[396,1288,1289],{},"judge guidance"," in ",[404,1292,410],{}," → put historical summary\u002Freport\u002Ffindings in ",[404,1295,785],{}," → upload via SDK.",[514,1298,1299],{},"Internal pull\u002Fbuild scripts live outside this handbook. Use the shapes below as the contract; ask Data Ops \u002F LLM Ops for the current scripts if you need a production refresh.",[375,1301,1302],{},[396,1303,1304,1306],{},[404,1305,406],{}," (prompt variables only)",[629,1308,1310],{"className":970,"code":1309,"language":972,"meta":635,"style":635},"{\n  \"user_name\": \"Marcus Silva\",\n  \"user_role\": \"Team Lead and L3 Engineer\",\n  \"organisation_context\": \"Black Mesa is a diversified technology and research organization...\",\n  \"source_type\": \"PROD\",\n  \"source_info\": {\n    \"id\": \"02e4e876-4c33-b27b-9c8e-82cd402f0f85\",\n    \"name\": \"Akamai App & API Protector\",\n    \"type\": \"PROD\",\n    \"vendor\": { \"id\": \"...\", \"name\": \"Akamai\" }\n  },\n  \"task_directive\": \"You are conducting a structured interview to gather information about a product...\",\n  \"interview_transcript\": [\n    {\n      \"id\": \"rufsaa\",\n      \"role\": \"assistant\",\n      \"text\": \"Hi Marcus! I'm ESPi...\",\n      \"isThoughts\": false,\n      \"imageStr\": null\n    },\n    {\n      \"id\": \"oywj8t\",\n      \"role\": \"user\",\n      \"text\": \"Yes\",\n      \"isThoughts\": false,\n      \"imageStr\": null\n    }\n  ]\n}\n",[404,1311,1312,1316,1336,1356,1376,1396,1409,1429,1449,1468,1515,1519,1539,1553,1559,1580,1601,1621,1636,1651,1657,1662,1682,1702,1722,1735,1748,1754,1760],{"__ignoreMap":635},[976,1313,1314],{"class":978,"line":979},[976,1315,983],{"class":982},[976,1317,1318,1320,1323,1325,1327,1329,1332,1334],{"class":978,"line":986},[976,1319,989],{"class":982},[976,1321,1322],{"class":992},"user_name",[976,1324,996],{"class":982},[976,1326,999],{"class":982},[976,1328,1019],{"class":982},[976,1330,1331],{"class":1022},"Marcus Silva",[976,1333,996],{"class":982},[976,1335,1027],{"class":982},[976,1337,1338,1340,1343,1345,1347,1349,1352,1354],{"class":978,"line":1005},[976,1339,989],{"class":982},[976,1341,1342],{"class":992},"user_role",[976,1344,996],{"class":982},[976,1346,999],{"class":982},[976,1348,1019],{"class":982},[976,1350,1351],{"class":1022},"Team Lead and L3 Engineer",[976,1353,996],{"class":982},[976,1355,1027],{"class":982},[976,1357,1358,1360,1363,1365,1367,1369,1372,1374],{"class":978,"line":1030},[976,1359,989],{"class":982},[976,1361,1362],{"class":992},"organisation_context",[976,1364,996],{"class":982},[976,1366,999],{"class":982},[976,1368,1019],{"class":982},[976,1370,1371],{"class":1022},"Black Mesa is a diversified technology and research organization...",[976,1373,996],{"class":982},[976,1375,1027],{"class":982},[976,1377,1378,1380,1383,1385,1387,1389,1392,1394],{"class":978,"line":1045},[976,1379,989],{"class":982},[976,1381,1382],{"class":992},"source_type",[976,1384,996],{"class":982},[976,1386,999],{"class":982},[976,1388,1019],{"class":982},[976,1390,1391],{"class":1022},"PROD",[976,1393,996],{"class":982},[976,1395,1027],{"class":982},[976,1397,1398,1400,1403,1405,1407],{"class":978,"line":1060},[976,1399,989],{"class":982},[976,1401,1402],{"class":992},"source_info",[976,1404,996],{"class":982},[976,1406,999],{"class":982},[976,1408,1002],{"class":982},[976,1410,1411,1413,1416,1418,1420,1422,1425,1427],{"class":978,"line":1066},[976,1412,1008],{"class":982},[976,1414,1415],{"class":1011},"id",[976,1417,996],{"class":982},[976,1419,999],{"class":982},[976,1421,1019],{"class":982},[976,1423,1424],{"class":1022},"02e4e876-4c33-b27b-9c8e-82cd402f0f85",[976,1426,996],{"class":982},[976,1428,1027],{"class":982},[976,1430,1431,1433,1436,1438,1440,1442,1445,1447],{"class":978,"line":1164},[976,1432,1008],{"class":982},[976,1434,1435],{"class":1011},"name",[976,1437,996],{"class":982},[976,1439,999],{"class":982},[976,1441,1019],{"class":982},[976,1443,1444],{"class":1022},"Akamai App & API Protector",[976,1446,996],{"class":982},[976,1448,1027],{"class":982},[976,1450,1451,1453,1456,1458,1460,1462,1464,1466],{"class":978,"line":1177},[976,1452,1008],{"class":982},[976,1454,1455],{"class":1011},"type",[976,1457,996],{"class":982},[976,1459,999],{"class":982},[976,1461,1019],{"class":982},[976,1463,1391],{"class":1022},[976,1465,996],{"class":982},[976,1467,1027],{"class":982},[976,1469,1470,1472,1475,1477,1479,1481,1483,1485,1487,1489,1491,1494,1496,1498,1500,1502,1504,1506,1508,1511,1513],{"class":978,"line":1198},[976,1471,1008],{"class":982},[976,1473,1474],{"class":1011},"vendor",[976,1476,996],{"class":982},[976,1478,999],{"class":982},[976,1480,1245],{"class":982},[976,1482,1019],{"class":982},[976,1484,1415],{"class":1250},[976,1486,996],{"class":982},[976,1488,999],{"class":982},[976,1490,1019],{"class":982},[976,1492,1493],{"class":1022},"...",[976,1495,996],{"class":982},[976,1497,1220],{"class":982},[976,1499,1019],{"class":982},[976,1501,1435],{"class":1250},[976,1503,996],{"class":982},[976,1505,999],{"class":982},[976,1507,1019],{"class":982},[976,1509,1510],{"class":1022},"Akamai",[976,1512,996],{"class":982},[976,1514,1261],{"class":982},[976,1516,1517],{"class":978,"line":1233},[976,1518,1127],{"class":982},[976,1520,1521,1523,1526,1528,1530,1532,1535,1537],{"class":978,"line":1264},[976,1522,989],{"class":982},[976,1524,1525],{"class":992},"task_directive",[976,1527,996],{"class":982},[976,1529,999],{"class":982},[976,1531,1019],{"class":982},[976,1533,1534],{"class":1022},"You are conducting a structured interview to gather information about a product...",[976,1536,996],{"class":982},[976,1538,1027],{"class":982},[976,1540,1541,1543,1546,1548,1550],{"class":978,"line":1269},[976,1542,989],{"class":982},[976,1544,1545],{"class":992},"interview_transcript",[976,1547,996],{"class":982},[976,1549,999],{"class":982},[976,1551,1552],{"class":982}," [\n",[976,1554,1556],{"class":978,"line":1555},14,[976,1557,1558],{"class":982},"    {\n",[976,1560,1562,1565,1567,1569,1571,1573,1576,1578],{"class":978,"line":1561},15,[976,1563,1564],{"class":982},"      \"",[976,1566,1415],{"class":1011},[976,1568,996],{"class":982},[976,1570,999],{"class":982},[976,1572,1019],{"class":982},[976,1574,1575],{"class":1022},"rufsaa",[976,1577,996],{"class":982},[976,1579,1027],{"class":982},[976,1581,1583,1585,1588,1590,1592,1594,1597,1599],{"class":978,"line":1582},16,[976,1584,1564],{"class":982},[976,1586,1587],{"class":1011},"role",[976,1589,996],{"class":982},[976,1591,999],{"class":982},[976,1593,1019],{"class":982},[976,1595,1596],{"class":1022},"assistant",[976,1598,996],{"class":982},[976,1600,1027],{"class":982},[976,1602,1604,1606,1608,1610,1612,1614,1617,1619],{"class":978,"line":1603},17,[976,1605,1564],{"class":982},[976,1607,634],{"class":1011},[976,1609,996],{"class":982},[976,1611,999],{"class":982},[976,1613,1019],{"class":982},[976,1615,1616],{"class":1022},"Hi Marcus! I'm ESPi...",[976,1618,996],{"class":982},[976,1620,1027],{"class":982},[976,1622,1624,1626,1629,1631,1633],{"class":978,"line":1623},18,[976,1625,1564],{"class":982},[976,1627,1628],{"class":1011},"isThoughts",[976,1630,996],{"class":982},[976,1632,999],{"class":982},[976,1634,1635],{"class":982}," false,\n",[976,1637,1639,1641,1644,1646,1648],{"class":978,"line":1638},19,[976,1640,1564],{"class":982},[976,1642,1643],{"class":1011},"imageStr",[976,1645,996],{"class":982},[976,1647,999],{"class":982},[976,1649,1650],{"class":982}," null\n",[976,1652,1654],{"class":978,"line":1653},20,[976,1655,1656],{"class":982},"    },\n",[976,1658,1660],{"class":978,"line":1659},21,[976,1661,1558],{"class":982},[976,1663,1665,1667,1669,1671,1673,1675,1678,1680],{"class":978,"line":1664},22,[976,1666,1564],{"class":982},[976,1668,1415],{"class":1011},[976,1670,996],{"class":982},[976,1672,999],{"class":982},[976,1674,1019],{"class":982},[976,1676,1677],{"class":1022},"oywj8t",[976,1679,996],{"class":982},[976,1681,1027],{"class":982},[976,1683,1685,1687,1689,1691,1693,1695,1698,1700],{"class":978,"line":1684},23,[976,1686,1564],{"class":982},[976,1688,1587],{"class":1011},[976,1690,996],{"class":982},[976,1692,999],{"class":982},[976,1694,1019],{"class":982},[976,1696,1697],{"class":1022},"user",[976,1699,996],{"class":982},[976,1701,1027],{"class":982},[976,1703,1705,1707,1709,1711,1713,1715,1718,1720],{"class":978,"line":1704},24,[976,1706,1564],{"class":982},[976,1708,634],{"class":1011},[976,1710,996],{"class":982},[976,1712,999],{"class":982},[976,1714,1019],{"class":982},[976,1716,1717],{"class":1022},"Yes",[976,1719,996],{"class":982},[976,1721,1027],{"class":982},[976,1723,1725,1727,1729,1731,1733],{"class":978,"line":1724},25,[976,1726,1564],{"class":982},[976,1728,1628],{"class":1011},[976,1730,996],{"class":982},[976,1732,999],{"class":982},[976,1734,1635],{"class":982},[976,1736,1738,1740,1742,1744,1746],{"class":978,"line":1737},26,[976,1739,1564],{"class":982},[976,1741,1643],{"class":1011},[976,1743,996],{"class":982},[976,1745,999],{"class":982},[976,1747,1650],{"class":982},[976,1749,1751],{"class":978,"line":1750},27,[976,1752,1753],{"class":982},"    }\n",[976,1755,1757],{"class":978,"line":1756},28,[976,1758,1759],{"class":982},"  ]\n",[976,1761,1763],{"class":978,"line":1762},29,[976,1764,1069],{"class":982},[375,1766,1767],{},[396,1768,1769,1771],{},[404,1770,410],{}," (guidance)",[629,1773,1775],{"className":970,"code":1774,"language":972,"meta":635,"style":635},"{\n  \"expected_summary_focus\": \"Summarize strategic state, material risks\u002Finsights, and decision-relevant takeaways. Surface only transcript-grounded points.\",\n  \"expected_min_findings\": 2,\n  \"expected_max_findings\": 11,\n  \"expected_required_finding_types\": [\"insight\", \"risk\", \"sentiment\"],\n  \"expected_report_sections\": [\n    \"Overview\",\n    \"Strategic State\",\n    \"Capabilities and Usage\",\n    \"Coverage and Controls\",\n    \"Sentiment and Operational Experience\",\n    \"Risks, Gaps and Opportunities\",\n    \"Stakeholders\",\n    \"Conclusion\"\n  ],\n  \"must_align_to_directive\": true,\n  \"must_be_grounded_in_transcript\": true,\n  \"schema_notes\": {\n    \"RISK_requires\": [\"severity\"],\n    \"INSIGHT_requires\": [\"category\"],\n    \"SENTIMENT_requires\": [\"category\", \"rating\"]\n  }\n}\n",[404,1776,1777,1781,1801,1817,1833,1873,1886,1897,1908,1919,1930,1941,1952,1963,1972,1977,1990,2003,2016,2038,2060,2091,2095],{"__ignoreMap":635},[976,1778,1779],{"class":978,"line":979},[976,1780,983],{"class":982},[976,1782,1783,1785,1788,1790,1792,1794,1797,1799],{"class":978,"line":986},[976,1784,989],{"class":982},[976,1786,1787],{"class":992},"expected_summary_focus",[976,1789,996],{"class":982},[976,1791,999],{"class":982},[976,1793,1019],{"class":982},[976,1795,1796],{"class":1022},"Summarize strategic state, material risks\u002Finsights, and decision-relevant takeaways. Surface only transcript-grounded points.",[976,1798,996],{"class":982},[976,1800,1027],{"class":982},[976,1802,1803,1805,1808,1810,1812,1815],{"class":978,"line":1005},[976,1804,989],{"class":982},[976,1806,1807],{"class":992},"expected_min_findings",[976,1809,996],{"class":982},[976,1811,999],{"class":982},[976,1813,1814],{"class":1250}," 2",[976,1816,1027],{"class":982},[976,1818,1819,1821,1824,1826,1828,1831],{"class":978,"line":1030},[976,1820,989],{"class":982},[976,1822,1823],{"class":992},"expected_max_findings",[976,1825,996],{"class":982},[976,1827,999],{"class":982},[976,1829,1830],{"class":1250}," 11",[976,1832,1027],{"class":982},[976,1834,1835,1837,1840,1842,1844,1846,1848,1851,1853,1855,1857,1860,1862,1864,1866,1869,1871],{"class":978,"line":1045},[976,1836,989],{"class":982},[976,1838,1839],{"class":992},"expected_required_finding_types",[976,1841,996],{"class":982},[976,1843,999],{"class":982},[976,1845,1210],{"class":982},[976,1847,996],{"class":982},[976,1849,1850],{"class":1022},"insight",[976,1852,996],{"class":982},[976,1854,1220],{"class":982},[976,1856,1019],{"class":982},[976,1858,1859],{"class":1022},"risk",[976,1861,996],{"class":982},[976,1863,1220],{"class":982},[976,1865,1019],{"class":982},[976,1867,1868],{"class":1022},"sentiment",[976,1870,996],{"class":982},[976,1872,1230],{"class":982},[976,1874,1875,1877,1880,1882,1884],{"class":978,"line":1060},[976,1876,989],{"class":982},[976,1878,1879],{"class":992},"expected_report_sections",[976,1881,996],{"class":982},[976,1883,999],{"class":982},[976,1885,1552],{"class":982},[976,1887,1888,1890,1893,1895],{"class":978,"line":1066},[976,1889,1008],{"class":982},[976,1891,1892],{"class":1022},"Overview",[976,1894,996],{"class":982},[976,1896,1027],{"class":982},[976,1898,1899,1901,1904,1906],{"class":978,"line":1164},[976,1900,1008],{"class":982},[976,1902,1903],{"class":1022},"Strategic State",[976,1905,996],{"class":982},[976,1907,1027],{"class":982},[976,1909,1910,1912,1915,1917],{"class":978,"line":1177},[976,1911,1008],{"class":982},[976,1913,1914],{"class":1022},"Capabilities and Usage",[976,1916,996],{"class":982},[976,1918,1027],{"class":982},[976,1920,1921,1923,1926,1928],{"class":978,"line":1198},[976,1922,1008],{"class":982},[976,1924,1925],{"class":1022},"Coverage and Controls",[976,1927,996],{"class":982},[976,1929,1027],{"class":982},[976,1931,1932,1934,1937,1939],{"class":978,"line":1233},[976,1933,1008],{"class":982},[976,1935,1936],{"class":1022},"Sentiment and Operational Experience",[976,1938,996],{"class":982},[976,1940,1027],{"class":982},[976,1942,1943,1945,1948,1950],{"class":978,"line":1264},[976,1944,1008],{"class":982},[976,1946,1947],{"class":1022},"Risks, Gaps and Opportunities",[976,1949,996],{"class":982},[976,1951,1027],{"class":982},[976,1953,1954,1956,1959,1961],{"class":978,"line":1269},[976,1955,1008],{"class":982},[976,1957,1958],{"class":1022},"Stakeholders",[976,1960,996],{"class":982},[976,1962,1027],{"class":982},[976,1964,1965,1967,1970],{"class":978,"line":1555},[976,1966,1008],{"class":982},[976,1968,1969],{"class":1022},"Conclusion",[976,1971,1122],{"class":982},[976,1973,1974],{"class":978,"line":1561},[976,1975,1976],{"class":982},"  ],\n",[976,1978,1979,1981,1984,1986,1988],{"class":978,"line":1582},[976,1980,989],{"class":982},[976,1982,1983],{"class":992},"must_align_to_directive",[976,1985,996],{"class":982},[976,1987,999],{"class":982},[976,1989,1042],{"class":982},[976,1991,1992,1994,1997,1999,2001],{"class":978,"line":1603},[976,1993,989],{"class":982},[976,1995,1996],{"class":992},"must_be_grounded_in_transcript",[976,1998,996],{"class":982},[976,2000,999],{"class":982},[976,2002,1042],{"class":982},[976,2004,2005,2007,2010,2012,2014],{"class":978,"line":1623},[976,2006,989],{"class":982},[976,2008,2009],{"class":992},"schema_notes",[976,2011,996],{"class":982},[976,2013,999],{"class":982},[976,2015,1002],{"class":982},[976,2017,2018,2020,2023,2025,2027,2029,2031,2034,2036],{"class":978,"line":1638},[976,2019,1008],{"class":982},[976,2021,2022],{"class":1011},"RISK_requires",[976,2024,996],{"class":982},[976,2026,999],{"class":982},[976,2028,1210],{"class":982},[976,2030,996],{"class":982},[976,2032,2033],{"class":1022},"severity",[976,2035,996],{"class":982},[976,2037,1230],{"class":982},[976,2039,2040,2042,2045,2047,2049,2051,2053,2056,2058],{"class":978,"line":1653},[976,2041,1008],{"class":982},[976,2043,2044],{"class":1011},"INSIGHT_requires",[976,2046,996],{"class":982},[976,2048,999],{"class":982},[976,2050,1210],{"class":982},[976,2052,996],{"class":982},[976,2054,2055],{"class":1022},"category",[976,2057,996],{"class":982},[976,2059,1230],{"class":982},[976,2061,2062,2064,2067,2069,2071,2073,2075,2077,2079,2081,2083,2086,2088],{"class":978,"line":1659},[976,2063,1008],{"class":982},[976,2065,2066],{"class":1011},"SENTIMENT_requires",[976,2068,996],{"class":982},[976,2070,999],{"class":982},[976,2072,1210],{"class":982},[976,2074,996],{"class":982},[976,2076,2055],{"class":1022},[976,2078,996],{"class":982},[976,2080,1220],{"class":982},[976,2082,1019],{"class":982},[976,2084,2085],{"class":1022},"rating",[976,2087,996],{"class":982},[976,2089,2090],{"class":982},"]\n",[976,2092,2093],{"class":978,"line":1664},[976,2094,1063],{"class":982},[976,2096,2097],{"class":978,"line":1684},[976,2098,1069],{"class":982},[375,2100,2101],{},[396,2102,2103,2105],{},[404,2104,413],{}," (example)",[629,2107,2109],{"className":970,"code":2108,"language":972,"meta":635,"style":635},"{\n  \"dataset\": \"findings\u002Fgate_test\",\n  \"tags\": [\"findings_agent\", \"black_mesa\", \"anonymized\"],\n  \"original_task_id_hash\": \"4ec6af6be8582669\",\n  \"product_name\": \"Akamai App & API Protector\",\n  \"expected_tools\": [\"platformSearch\"],\n  \"reference_output\": {\n    \"summary\": \"...\",\n    \"report\": \"...\",\n    \"findings\": []\n  },\n  \"anonymization\": {\n    \"tenant_replaced_with\": \"Black Mesa\",\n    \"people_synthetic\": true,\n    \"images_stripped\": true\n  }\n}\n",[404,2110,2111,2115,2133,2172,2191,2210,2232,2245,2264,2283,2297,2301,2313,2331,2343,2355,2359],{"__ignoreMap":635},[976,2112,2113],{"class":978,"line":979},[976,2114,983],{"class":982},[976,2116,2117,2119,2121,2123,2125,2127,2129,2131],{"class":978,"line":986},[976,2118,989],{"class":982},[976,2120,1182],{"class":992},[976,2122,996],{"class":982},[976,2124,999],{"class":982},[976,2126,1019],{"class":982},[976,2128,512],{"class":1022},[976,2130,996],{"class":982},[976,2132,1027],{"class":982},[976,2134,2135,2137,2139,2141,2143,2145,2147,2150,2152,2154,2156,2159,2161,2163,2165,2168,2170],{"class":978,"line":1005},[976,2136,989],{"class":982},[976,2138,1203],{"class":992},[976,2140,996],{"class":982},[976,2142,999],{"class":982},[976,2144,1210],{"class":982},[976,2146,996],{"class":982},[976,2148,2149],{"class":1022},"findings_agent",[976,2151,996],{"class":982},[976,2153,1220],{"class":982},[976,2155,1019],{"class":982},[976,2157,2158],{"class":1022},"black_mesa",[976,2160,996],{"class":982},[976,2162,1220],{"class":982},[976,2164,1019],{"class":982},[976,2166,2167],{"class":1022},"anonymized",[976,2169,996],{"class":982},[976,2171,1230],{"class":982},[976,2173,2174,2176,2178,2180,2182,2184,2187,2189],{"class":978,"line":1030},[976,2175,989],{"class":982},[976,2177,952],{"class":992},[976,2179,996],{"class":982},[976,2181,999],{"class":982},[976,2183,1019],{"class":982},[976,2185,2186],{"class":1022},"4ec6af6be8582669",[976,2188,996],{"class":982},[976,2190,1027],{"class":982},[976,2192,2193,2195,2198,2200,2202,2204,2206,2208],{"class":978,"line":1045},[976,2194,989],{"class":982},[976,2196,2197],{"class":992},"product_name",[976,2199,996],{"class":982},[976,2201,999],{"class":982},[976,2203,1019],{"class":982},[976,2205,1444],{"class":1022},[976,2207,996],{"class":982},[976,2209,1027],{"class":982},[976,2211,2212,2214,2217,2219,2221,2223,2225,2228,2230],{"class":978,"line":1060},[976,2213,989],{"class":982},[976,2215,2216],{"class":992},"expected_tools",[976,2218,996],{"class":982},[976,2220,999],{"class":982},[976,2222,1210],{"class":982},[976,2224,996],{"class":982},[976,2226,2227],{"class":1022},"platformSearch",[976,2229,996],{"class":982},[976,2231,1230],{"class":982},[976,2233,2234,2236,2239,2241,2243],{"class":978,"line":1066},[976,2235,989],{"class":982},[976,2237,2238],{"class":992},"reference_output",[976,2240,996],{"class":982},[976,2242,999],{"class":982},[976,2244,1002],{"class":982},[976,2246,2247,2249,2252,2254,2256,2258,2260,2262],{"class":978,"line":1164},[976,2248,1008],{"class":982},[976,2250,2251],{"class":1011},"summary",[976,2253,996],{"class":982},[976,2255,999],{"class":982},[976,2257,1019],{"class":982},[976,2259,1493],{"class":1022},[976,2261,996],{"class":982},[976,2263,1027],{"class":982},[976,2265,2266,2268,2271,2273,2275,2277,2279,2281],{"class":978,"line":1177},[976,2267,1008],{"class":982},[976,2269,2270],{"class":1011},"report",[976,2272,996],{"class":982},[976,2274,999],{"class":982},[976,2276,1019],{"class":982},[976,2278,1493],{"class":1022},[976,2280,996],{"class":982},[976,2282,1027],{"class":982},[976,2284,2285,2287,2290,2292,2294],{"class":978,"line":1198},[976,2286,1008],{"class":982},[976,2288,2289],{"class":1011},"findings",[976,2291,996],{"class":982},[976,2293,999],{"class":982},[976,2295,2296],{"class":982}," []\n",[976,2298,2299],{"class":978,"line":1233},[976,2300,1127],{"class":982},[976,2302,2303,2305,2307,2309,2311],{"class":978,"line":1264},[976,2304,989],{"class":982},[976,2306,993],{"class":992},[976,2308,996],{"class":982},[976,2310,999],{"class":982},[976,2312,1002],{"class":982},[976,2314,2315,2317,2319,2321,2323,2325,2327,2329],{"class":978,"line":1269},[976,2316,1008],{"class":982},[976,2318,1012],{"class":1011},[976,2320,996],{"class":982},[976,2322,999],{"class":982},[976,2324,1019],{"class":982},[976,2326,915],{"class":1022},[976,2328,996],{"class":982},[976,2330,1027],{"class":982},[976,2332,2333,2335,2337,2339,2341],{"class":978,"line":1555},[976,2334,1008],{"class":982},[976,2336,1035],{"class":1011},[976,2338,996],{"class":982},[976,2340,999],{"class":982},[976,2342,1042],{"class":982},[976,2344,2345,2347,2349,2351,2353],{"class":978,"line":1561},[976,2346,1008],{"class":982},[976,2348,1050],{"class":1011},[976,2350,996],{"class":982},[976,2352,999],{"class":982},[976,2354,1057],{"class":982},[976,2356,2357],{"class":978,"line":1582},[976,2358,1063],{"class":982},[976,2360,2361],{"class":978,"line":1603},[976,2362,1069],{"class":982},[728,2364,2367,2413,2570],{"icon":2365,"label":2366},"i-lucide-code","1.4.7 Python SDK upload (nested JSON)",[629,2368,2372],{"className":2369,"code":2370,"language":2371,"meta":635,"style":635},"language-bash shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","export LANGFUSE_PUBLIC_KEY=pk-lf-...\nexport LANGFUSE_SECRET_KEY=sk-lf-...\nexport LANGFUSE_HOST=https:\u002F\u002Fcloud.langfuse.com\n","bash",[404,2373,2374,2389,2401],{"__ignoreMap":635},[976,2375,2376,2379,2383,2386],{"class":978,"line":979},[976,2377,2378],{"class":992},"export",[976,2380,2382],{"class":2381},"sTEyZ"," LANGFUSE_PUBLIC_KEY",[976,2384,2385],{"class":982},"=",[976,2387,2388],{"class":2381},"pk-lf-...\n",[976,2390,2391,2393,2396,2398],{"class":978,"line":986},[976,2392,2378],{"class":992},[976,2394,2395],{"class":2381}," LANGFUSE_SECRET_KEY",[976,2397,2385],{"class":982},[976,2399,2400],{"class":2381},"sk-lf-...\n",[976,2402,2403,2405,2408,2410],{"class":978,"line":1005},[976,2404,2378],{"class":992},[976,2406,2407],{"class":2381}," LANGFUSE_HOST",[976,2409,2385],{"class":982},[976,2411,2412],{"class":2381},"https:\u002F\u002Fcloud.langfuse.com\n",[629,2414,2418],{"className":2415,"code":2416,"language":2417,"meta":635,"style":635},"language-python shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","from langfuse import get_client\n\nlangfuse = get_client()\n\ndataset_name = \"conversation-namer\u002Fsmoke_test\"\n\nlangfuse.create_dataset(\n    name=dataset_name,\n    description=\"Smoke test set for Conversation Namer\",\n)\n\nitems = [\n    {\n        \"input\": {\n            \"message\": \"Can you compare our CrowdStrike and Wiz renewals for next quarter?\"\n        },\n        \"expected_output\": {\"title\": \"CrowdStrike vs Wiz renewals\"},\n        \"metadata\": {\n            \"dataset\": dataset_name,\n            \"tags\": [\"conversation_namer\", \"synthetic\"],\n        },\n    },\n]\n\nfor item in items:\n    langfuse.create_dataset_item(\n        dataset_name=dataset_name,\n        input=item[\"input\"],\n        expected_output=item[\"expected_output\"],\n        metadata=item.get(\"metadata\"),\n    )\n","python",[404,2419,2420,2425,2431,2436,2440,2445,2449,2454,2459,2464,2469,2473,2478,2482,2487,2492,2497,2502,2507,2512,2517,2521,2525,2529,2533,2538,2543,2548,2553,2558,2564],{"__ignoreMap":635},[976,2421,2422],{"class":978,"line":979},[976,2423,2424],{},"from langfuse import get_client\n",[976,2426,2427],{"class":978,"line":986},[976,2428,2430],{"emptyLinePlaceholder":2429},true,"\n",[976,2432,2433],{"class":978,"line":1005},[976,2434,2435],{},"langfuse = get_client()\n",[976,2437,2438],{"class":978,"line":1030},[976,2439,2430],{"emptyLinePlaceholder":2429},[976,2441,2442],{"class":978,"line":1045},[976,2443,2444],{},"dataset_name = \"conversation-namer\u002Fsmoke_test\"\n",[976,2446,2447],{"class":978,"line":1060},[976,2448,2430],{"emptyLinePlaceholder":2429},[976,2450,2451],{"class":978,"line":1066},[976,2452,2453],{},"langfuse.create_dataset(\n",[976,2455,2456],{"class":978,"line":1164},[976,2457,2458],{},"    name=dataset_name,\n",[976,2460,2461],{"class":978,"line":1177},[976,2462,2463],{},"    description=\"Smoke test set for Conversation Namer\",\n",[976,2465,2466],{"class":978,"line":1198},[976,2467,2468],{},")\n",[976,2470,2471],{"class":978,"line":1233},[976,2472,2430],{"emptyLinePlaceholder":2429},[976,2474,2475],{"class":978,"line":1264},[976,2476,2477],{},"items = [\n",[976,2479,2480],{"class":978,"line":1269},[976,2481,1558],{},[976,2483,2484],{"class":978,"line":1555},[976,2485,2486],{},"        \"input\": {\n",[976,2488,2489],{"class":978,"line":1561},[976,2490,2491],{},"            \"message\": \"Can you compare our CrowdStrike and Wiz renewals for next quarter?\"\n",[976,2493,2494],{"class":978,"line":1582},[976,2495,2496],{},"        },\n",[976,2498,2499],{"class":978,"line":1603},[976,2500,2501],{},"        \"expected_output\": {\"title\": \"CrowdStrike vs Wiz renewals\"},\n",[976,2503,2504],{"class":978,"line":1623},[976,2505,2506],{},"        \"metadata\": {\n",[976,2508,2509],{"class":978,"line":1638},[976,2510,2511],{},"            \"dataset\": dataset_name,\n",[976,2513,2514],{"class":978,"line":1653},[976,2515,2516],{},"            \"tags\": [\"conversation_namer\", \"synthetic\"],\n",[976,2518,2519],{"class":978,"line":1659},[976,2520,2496],{},[976,2522,2523],{"class":978,"line":1664},[976,2524,1656],{},[976,2526,2527],{"class":978,"line":1684},[976,2528,2090],{},[976,2530,2531],{"class":978,"line":1704},[976,2532,2430],{"emptyLinePlaceholder":2429},[976,2534,2535],{"class":978,"line":1724},[976,2536,2537],{},"for item in items:\n",[976,2539,2540],{"class":978,"line":1737},[976,2541,2542],{},"    langfuse.create_dataset_item(\n",[976,2544,2545],{"class":978,"line":1750},[976,2546,2547],{},"        dataset_name=dataset_name,\n",[976,2549,2550],{"class":978,"line":1756},[976,2551,2552],{},"        input=item[\"input\"],\n",[976,2554,2555],{"class":978,"line":1762},[976,2556,2557],{},"        expected_output=item[\"expected_output\"],\n",[976,2559,2561],{"class":978,"line":2560},30,[976,2562,2563],{},"        metadata=item.get(\"metadata\"),\n",[976,2565,2567],{"class":978,"line":2566},31,[976,2568,2569],{},"    )\n",[375,2571,2572],{},"Same pattern for Findings — pass the full nested objects; no need to flatten.",[728,2574,2577],{"icon":2575,"label":2576},"i-lucide-square-check","1.4.8 Checklist before upload",[487,2578,2581,2589,2597,2608,2619,2625,2636],{"className":2579},[2580],"contains-task-list",[393,2582,2585,2588],{"className":2583},[2584],"task-list-item",[406,2586],{"disabled":2429,"type":2587},"checkbox"," Agent under test is clear",[393,2590,2592,714,2594,2596],{"className":2591},[2584],[406,2593],{"disabled":2429,"type":2587},[404,2595,406],{}," keys match prompt variables 1:1",[393,2598,2600,714,2602,2604,2605,2607],{"className":2599},[2584],[406,2601],{"disabled":2429,"type":2587},[404,2603,410],{}," is a golden answer ",[396,2606,562],{}," judge guidance (not mixed without intent)",[393,2609,2611,2613,2614,2616,2617],{"className":2610},[2584],[406,2612],{"disabled":2429,"type":2587}," For open-ended agents: guidance in ",[404,2615,410],{},"; full goldens (if kept) in ",[404,2618,785],{},[393,2620,2622,2624],{"className":2621},[2584],[406,2623],{"disabled":2429,"type":2587}," Production data is anonymized",[393,2626,2628,2630,2631,2633,2634],{"className":2627},[2584],[406,2629],{"disabled":2429,"type":2587}," Name is ",[404,2632,659],{}," or ",[404,2635,672],{},[393,2637,2639,2641],{"className":2638},[2584],[406,2640],{"disabled":2429,"type":2587}," Spot-check 2–3 items in the Langfuse UI",[518,2643],{},[521,2645,2647],{"id":2646},"_2-creating-evaluators","2. Creating Evaluators",[375,2649,2650],{},"Evaluators are the scoring definitions you attach when you run prompt experiments.",[375,2652,2653,2654,2657],{},"Today we ",[396,2655,2656],{},"mostly set up LLM-as-judge evaluators"," in Langfuse.",[375,2659,2660,2663,2664,2667,2668,1290,2671,2673,2674,2679],{},[396,2661,2662],{},"Code-based checks"," still matter for deterministic rules, but we typically implement them in the ",[396,2665,2666],{},"application"," and\u002For ",[396,2669,2670],{},"unit\u002Fintegration tests",[404,2672,757],{},", rather than as the primary Langfuse experiment evaluators. Langfuse also supports ",[379,2675,2678],{"href":2676,"rel":2677},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fevaluation-methods\u002Fcode-evaluators",[383],"Code Evaluators"," if needed later.",[375,2681,2682],{},"Rule of thumb:",[487,2684,2685,2691],{},[393,2686,2687,2688,2690],{},"Needs reading comprehension \u002F judgment → ",[396,2689,838],{}," (Langfuse)",[393,2692,2693,2694,2696],{},"A junior engineer could assert it in a test → ",[396,2695,404],{}," (app or unit tests)",[574,2698,2700],{"id":2699},"_21-create-your-first-llm-as-judge-5-steps","2.1 Create your first LLM-as-judge (5 steps)",[390,2702,2703,2709,2715,2721,2738],{},[393,2704,2705,2708],{},[396,2706,2707],{},"Pick one dimension"," — e.g. “grounded in transcript” or “aligns with directive”.",[393,2710,2711,2714],{},[396,2712,2713],{},"Choose score shape"," — boolean \u002F categorical \u002F numeric.",[393,2716,2717,2720],{},[396,2718,2719],{},"Write the judge prompt"," — explicit pass\u002Ffail or category rules; one job only.",[393,2722,2723,2726,2727,407,2730,2733,2734,2737],{},[396,2724,2725],{},"Create it in Langfuse"," — map ",[404,2728,2729],{},"{{input}}",[404,2731,2732],{},"{{output}}"," (and ",[404,2735,2736],{},"{{expected_output}}"," only if needed).",[393,2739,2740,2743],{},[396,2741,2742],{},"Verify mapping"," — use Prompt Preview; spot-check that variables populate as expected.",[375,2745,2746,2747,2750],{},"Start with ",[396,2748,2749],{},"2–3 focused judges"," per agent. Add more only when debugging a specific failure class.",[375,2752,2753],{},"Findings starter pack:",[390,2755,2756,2762],{},[393,2757,2758,2761],{},[404,2759,2760],{},"findings_agent.grounding"," — Boolean",[393,2763,2764,2767,2768,407,2771,407,2774,414],{},[404,2765,2766],{},"findings_agent.directive_alignment"," — Categorical (",[404,2769,2770],{},"fail",[404,2772,2773],{},"partial",[404,2775,2776],{},"pass",[574,2778,2780],{"id":2779},"_22-naming","2.2 Naming",[629,2782,2785],{"className":2783,"code":2784,"language":634,"meta":635},[632],"{agentName}.{dimension}\n",[404,2786,2784],{"__ignoreMap":635},[375,2788,681,2789,685,2792,685,2794,588],{},[404,2790,2791],{},"conversation_namer.title_quality",[404,2793,2760],{},[404,2795,2766],{},[574,2797,2799],{"id":2798},"_23-create-in-langfuse-ui","2.3 Create in Langfuse (UI)",[390,2801,2802,2813,2823,2833,2840,2843],{},[393,2803,2804,2805,2808,2809,2812],{},"Ensure an ",[396,2806,2807],{},"LLM Connection"," exists (",[396,2810,2811],{},"Settings → LLM Connections","). The judge model must support structured output.",[393,2814,2815,2816,2819,2820,588],{},"Open ",[396,2817,2818],{},"Evaluators"," → ",[396,2821,2822],{},"+ Set up Evaluator",[393,2824,2825,2826,2829,2830,588],{},"Pick a managed template, or ",[396,2827,2828],{},"Custom"," and paste your judge prompt with ",[404,2831,2832],{},"{{variables}}",[393,2834,2835,2836,2839],{},"Choose ",[396,2837,2838],{},"score type"," (boolean \u002F categorical \u002F numeric). For categorical, define labels and numeric mapping.",[393,2841,2842],{},"Map variables to Input \u002F Output \u002F Expected output (add JSONPath if needed).",[393,2844,2845,2846,588],{},"Save. Attach these evaluators when ",[379,2847,2848],{"href":430},"running experiments",[375,2850,2851,2852,588],{},"Official guide: ",[379,2853,2856],{"href":2854,"rel":2855},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fevaluation-methods\u002Fllm-as-a-judge",[383],"LLM-as-a-Judge",[574,2858,2860],{"id":2859},"_24-evaluator-reference-expand-as-needed","2.4 Evaluator reference (expand as needed)",[725,2862,2863,2920,3008,3054,3131,3212,3244,3294],{},[728,2864,2867],{"icon":2865,"label":2866},"i-lucide-layout-list","2.4.1 Where evaluators fit in Langfuse",[434,2868,2869,2879],{},[437,2870,2871],{},[440,2872,2873,2876],{},[443,2874,2875],{},"If you want to...",[443,2877,2878],{},"Langfuse \u002F approach we use",[450,2880,2881,2890,2900,2909],{},[440,2882,2883,2886],{},[455,2884,2885],{},"Build a reusable set of test cases",[455,2887,2888],{},[379,2889,495],{"href":400},[440,2891,2892,2895],{},[455,2893,2894],{},"Compare prompt or model changes",[455,2896,2897],{},[379,2898,2899],{"href":430},"Experiments",[440,2901,2902,2905],{},[455,2903,2904],{},"Automatically score quality (grounding, alignment, tone, …)",[455,2906,2907],{},[396,2908,2856],{},[440,2910,2911,2914],{},[455,2912,2913],{},"Run deterministic checks (schema, length, exact match)",[455,2915,2916,2919],{},[396,2917,2918],{},"Code checks"," in the app or unit\u002Fintegration tests",[728,2921,2924,2986,3005],{"icon":2922,"label":2923},"i-lucide-gauge","2.4.2 Score types",[434,2925,2926,2939],{},[437,2927,2928],{},[440,2929,2930,2933,2936],{},[443,2931,2932],{},"Score type",[443,2934,2935],{},"Use when",[443,2937,2938],{},"Typical use",[450,2940,2941,2954,2973],{},[440,2942,2943,2948,2951],{},[455,2944,2945],{},[396,2946,2947],{},"Boolean",[455,2949,2950],{},"Clear pass\u002Ffail",[455,2952,2953],{},"Hard checks (e.g. grounding)",[440,2955,2956,2961,2970],{},[455,2957,2958],{},[396,2959,2960],{},"Categorical",[455,2962,2963,2964,407,2966,407,2968,414],{},"Small fixed tiers (",[404,2965,2770],{},[404,2967,2773],{},[404,2969,2776],{},[455,2971,2972],{},"Soft quality bands",[440,2974,2975,2980,2983],{},[455,2976,2977],{},[396,2978,2979],{},"Numeric",[455,2981,2982],{},"Fine-grained ranking",[455,2984,2985],{},"When tiers are too coarse",[375,2987,2988,2989,2819,2991,685,2994,2819,2996,685,2999,2819,3001,3004],{},"Map categorical labels to numbers when useful (e.g. ",[404,2990,2770],{},[404,2992,2993],{},"0",[404,2995,2773],{},[404,2997,2998],{},"0.5",[404,3000,2776],{},[404,3002,3003],{},"1",").",[375,3006,3007],{},"Prefer boolean for hard correctness; categorical (3-tier) for softer quality.",[728,3009,3012],{"icon":3010,"label":3011},"i-lucide-pencil","2.4.3 Design principles",[390,3013,3014,3020,3026,3032,3038,3044],{},[393,3015,3016,3019],{},[396,3017,3018],{},"One job per evaluator"," — do not mix grounding + writing quality in one score.",[393,3021,3022,3025],{},[396,3023,3024],{},"Say what to check"," — avoid long “do not score X” lists.",[393,3027,3028,3031],{},[396,3029,3030],{},"Define pass\u002Ffail or categories explicitly"," with short examples.",[393,3033,3034,3037],{},[396,3035,3036],{},"Derive criteria from the use case"," — do not hard-code one product’s interview branch unless it is universal.",[393,3039,3040,3043],{},[396,3041,3042],{},"Map only the data needed"," — use JSONPath when you only need a nested field.",[393,3045,3046,3049,3050,3053],{},[396,3047,3048],{},"Name and version stably"," — keep ",[404,3051,3052],{},"{agent}.{dimension}"," names unchanged so later experiment comparisons stay readable.",[728,3055,3058,3099,3113],{"icon":3056,"label":3057},"i-lucide-link","2.4.4 Variable mapping",[434,3059,3060,3070],{},[437,3061,3062],{},[440,3063,3064,3067],{},[443,3065,3066],{},"Common variable",[443,3068,3069],{},"Typical source",[450,3071,3072,3081,3090],{},[440,3073,3074,3078],{},[455,3075,3076],{},[404,3077,2729],{},[455,3079,3080],{},"Dataset item input",[440,3082,3083,3087],{},[455,3084,3085],{},[404,3086,2732],{},[455,3088,3089],{},"Run output (filled when an experiment executes)",[440,3091,3092,3096],{},[455,3093,3094],{},[404,3095,2736],{},[455,3097,3098],{},"Dataset expected output (optional)",[375,3100,3101,3102,3105,3106,3109,3110,588],{},"Use ",[396,3103,3104],{},"JSONPath"," when you only need a nested field (e.g. Output → ",[404,3107,3108],{},"$.findings","). Confirm in Langfuse ",[396,3111,3112],{},"Prompt Preview",[375,3114,3115,3121,3122,3004,3124,3127,3130],{},[396,3116,3117,3118],{},"Skip ",[404,3119,3120],{},"expected_output"," when the rubric lives fully in the judge prompt, or the check is reference-free (e.g. grounding against transcript in ",[404,3123,406],{},[3125,3126],"br",{},[396,3128,3129],{},"Use it"," when item-specific guidance lives in the dataset.",[728,3132,3135,3138,3186,3192,3201],{"icon":3133,"label":3134},"i-lucide-sparkles","2.4.5 Simple example: Conversation Namer title quality",[375,3136,3137],{},"A light judge — good first custom evaluator to practise the shape.",[434,3139,3140,3149],{},[437,3141,3142],{},[440,3143,3144,3146],{},[443,3145,536],{},[443,3147,3148],{},"Value",[450,3150,3151,3162,3171],{},[440,3152,3153,3158],{},[455,3154,3155],{},[396,3156,3157],{},"Name",[455,3159,3160],{},[404,3161,2791],{},[440,3163,3164,3169],{},[455,3165,3166],{},[396,3167,3168],{},"Score",[455,3170,2947],{},[440,3172,3173,3178],{},[455,3174,3175],{},[396,3176,3177],{},"Maps",[455,3179,3180,3182,3183,3185],{},[404,3181,2729],{}," → Input, ",[404,3184,2732],{}," → Output",[629,3187,3190],{"className":3188,"code":3189,"language":634,"meta":635},[632],"You evaluate whether the OUTPUT title is a good short label for the INPUT user message.\n\nPass (true) if ALL are true:\n- Title is non-empty\n- Title is at most 5 words\n- Title reflects the main topic of the user message (no unrelated subject)\n\nFail (false) otherwise.\nIf OUTPUT is empty or malformed, return false.\n\nINPUT:\n{{input}}\n\nOUTPUT:\n{{output}}\n",[404,3191,3189],{"__ignoreMap":635},[375,3193,3194,3195,2633,3198,588],{},"Score output: return ONLY ",[404,3196,3197],{},"true",[404,3199,3200],{},"false",[375,3202,3203,3204,3207,3208,3211],{},"For a pure length rule (",[404,3205,3206],{},"≤ 5 words","), prefer a ",[396,3209,3210],{},"unit test"," or in-app validation instead of a judge.",[728,3213,3216,3228,3234],{"icon":3214,"label":3215},"i-lucide-shield-check","2.4.6 Findings: grounding judge (Boolean)",[375,3217,3218,714,3221,3182,3223,3225,3226,414],{},[396,3219,3220],{},"Maps:",[404,3222,2729],{},[404,3224,2732],{}," → Output (no ",[404,3227,3120],{},[629,3229,3232],{"className":3230,"code":3231,"language":634,"meta":635},[632],"You evaluate whether the agent OUTPUT is grounded in the interview transcript from INPUT.\n\nTask:\nDecide if material claims in the summary, report, and findings are supported by the transcript.\n\nRules:\n- Every material claim must be supported by the transcript\n- Paraphrase is allowed; invention is not\n- Minor omissions are OK; fabrication is not\n\nFail (false) if any material claim is invented, over-precise beyond the transcript, contradicts the transcript without uncertainty, or invents rationale\u002Fabbreviation expansions not stated.\n\nPass (true) only if all material claims are transcript-supported.\nIf OUTPUT is empty or malformed, return false.\n\nList any ungrounded claims briefly before deciding.\n\nINPUT:\n{{input}}\n\nOUTPUT:\n{{output}}\n",[404,3233,3231],{"__ignoreMap":635},[375,3235,3236,3237,3239,3240,2633,3242,588],{},"Score reasoning: 1–3 sentences; if false, name the worst ungrounded claim(s).",[3125,3238],{},"\nScore output: return ONLY ",[404,3241,3197],{},[404,3243,3200],{},[728,3245,3248,3276,3282],{"icon":3246,"label":3247},"i-lucide-list-checks","2.4.7 Findings: directive alignment judge (Categorical)",[375,3249,3250,714,3253,2819,3255,685,3257,2819,3259,685,3261,2819,3263,3265,3267,714,3269,3182,3271,3273,3274,414],{},[396,3251,3252],{},"Categories:",[404,3254,2770],{},[404,3256,2993],{},[404,3258,2773],{},[404,3260,2998],{},[404,3262,2776],{},[404,3264,3003],{},[3125,3266],{},[396,3268,3220],{},[404,3270,2729],{},[404,3272,2732],{}," → Output (directive usually inside ",[404,3275,406],{},[629,3277,3280],{"className":3278,"code":3279,"language":634,"meta":635},[632],"You evaluate whether the Findings\u002FReport agent OUTPUT aligns with the interview DIRECTIVE in INPUT.\n\nTask:\nJudge how well the REPORT and FINDINGS deliver the directive’s information goals, based on what the transcript actually captured.\n\nDo not assume a fixed interview structure. Derive success criteria from the directive itself.\n\nCheck only:\n1) Main directive objective(s) appear in report and\u002For findings\n2) Key requested topics are covered when the transcript has answers\n3) Findings are decision-useful for the directive\n4) Important directive-relevant transcript content is not systematically missing\n\nIf the transcript lacked answers for a topic, do not penalize missing content for that topic.\n\nCategories:\n- fail: largely misses the directive, or major goals missing\n- partial: some alignment, important gaps remain\n- pass: strong alignment; minor gaps only\n\nINPUT:\n{{input}}\n\nOUTPUT:\n{{output}}\n",[404,3281,3279],{"__ignoreMap":635},[375,3283,3284,3285,3239,3287,685,3289,3291,3292,588],{},"Score reasoning: 2–4 sentences with strongest coverage and most important gaps.",[3125,3286],{},[404,3288,2770],{},[404,3290,2773],{},", or ",[404,3293,2776],{},[728,3295,3297],{"icon":2575,"label":3296},"2.4.8 Checklist before saving an evaluator",[487,3298,3300,3306,3312,3318,3324,3330,3336,3344],{"className":3299},[2580],[393,3301,3303,3305],{"className":3302},[2584],[406,3304],{"disabled":2429,"type":2587}," One clear dimension",[393,3307,3309,3311],{"className":3308},[2584],[406,3310],{"disabled":2429,"type":2587}," Score type matches the check (boolean \u002F categorical \u002F numeric)",[393,3313,3315,3317],{"className":3314},[2584],[406,3316],{"disabled":2429,"type":2587}," Categorical labels + numeric mapping defined (if used)",[393,3319,3321,3323],{"className":3320},[2584],[406,3322],{"disabled":2429,"type":2587}," Judge prompt states positive checks",[393,3325,3327,3329],{"className":3326},[2584],[406,3328],{"disabled":2429,"type":2587}," Variable mapping verified in Prompt Preview",[393,3331,3333,3335],{"className":3332},[2584],[406,3334],{"disabled":2429,"type":2587}," JSONPath used when only a nested field is needed",[393,3337,3339,714,3341,3343],{"className":3338},[2584],[406,3340],{"disabled":2429,"type":2587},[404,3342,3120],{}," only when item-specific guidance is required",[393,3345,3347,3349,3350],{"className":3346},[2584],[406,3348],{"disabled":2429,"type":2587}," Name follows ",[404,3351,3052],{},[518,3353],{},[521,3355,3357],{"id":3356},"_3-running-experiments","3. Running Experiments",[375,3359,3360,3361,3363,3364,3367,3368,3371],{},"Use a ",[396,3362,1182],{}," and ",[396,3365,3366],{},"evaluators"," together in a Langfuse ",[396,3369,3370],{},"Prompt Experiment"," to compare prompt versions and decide whether to ship a change.",[375,3373,3374,3375,588],{},"Example completed runs: ",[379,3376,3378,3380],{"href":507,"rel":3377},[383],[404,3379,512],{}," experiments",[375,3382,3383,3384,3389,3390],{},"Official docs: ",[379,3385,3388],{"href":3386,"rel":3387},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fexperiments\u002Fexperiments-via-ui",[383],"Experiments via UI"," · ",[379,3391,3394],{"href":3392,"rel":3393},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fexperiments\u002Fdata-model",[383],"Experiments data model",[574,3396,3398],{"id":3397},"_31-what-a-langfuse-experiment-is","3.1 What a Langfuse experiment is",[434,3400,3401,3411],{},[437,3402,3403],{},[440,3404,3405,3408],{},[443,3406,3407],{},"Concept",[443,3409,3410],{},"Meaning",[450,3412,3413,3429,3439,3449,3458],{},[440,3414,3415,3419],{},[455,3416,3417],{},[396,3418,459],{},[455,3420,3421,3422,3424,3425,685,3427,414],{},"Frozen test cases (",[404,3423,406],{},", optional ",[404,3426,410],{},[404,3428,413],{},[440,3430,3431,3436],{},[455,3432,3433],{},[396,3434,3435],{},"Prompt",[455,3437,3438],{},"Versioned prompt from Prompt Management",[440,3440,3441,3446],{},[455,3442,3443],{},[396,3444,3445],{},"Experiment (Prompt Experiment)",[455,3447,3448],{},"Runs the selected prompt on each dataset item",[440,3450,3451,3455],{},[455,3452,3453],{},[396,3454,469],{},[455,3456,3457],{},"Scores each experiment item output (LLM-as-judge and\u002For code)",[440,3459,3460,3465],{},[455,3461,3462],{},[396,3463,3464],{},"Experiment comparison",[455,3466,3467],{},"Side-by-side aggregate + item-level score comparison across runs",[629,3469,3472],{"className":3470,"code":3471,"language":634,"meta":635},[632],"Dataset item input\n        │\n        ▼\nPrompt version (variables filled from input)\n        │\n        ▼\nModel output\n        │\n        ▼\nEvaluators attach scores\n        │\n        ▼\nCompare runs → promote or reject prompt\n",[404,3473,3471],{"__ignoreMap":635},[375,3475,3476,3479,3480,3483],{},[396,3477,3478],{},"Important:"," one experiment can attach ",[396,3481,3482],{},"multiple evaluators",". Do not create one experiment per score.",[574,3485,3487],{"id":3486},"_32-prerequisites","3.2 Prerequisites",[375,3489,3490],{},"Before running an experiment, confirm:",[390,3492,3493,3504,3517,3523],{},[393,3494,3495,3497,3498,3500,3501,3503],{},[396,3496,3435],{}," in Prompt Management with ",[404,3499,2832],{}," matching dataset ",[404,3502,406],{}," keys",[393,3505,3506,3508,3509,2633,3511,3513,3514],{},[396,3507,459],{}," uploaded (",[404,3510,880],{},[404,3512,876],{},") — see ",[379,3515,3516],{"href":400},"§1",[393,3518,3519,3522],{},[396,3520,3521],{},"LLM connection"," configured; default evaluation model supports structured output for judges",[393,3524,3525,3527,3528],{},[396,3526,2818],{}," created and able to target Experiments — see ",[379,3529,3530],{"href":421},"§2",[574,3532,3534],{"id":3533},"_33-run-a-prompt-experiment-ui","3.3 Run a Prompt Experiment (UI)",[390,3536,3537,3546,3560,3607],{},[393,3538,3539,3540,3542,3543,3545],{},"Go to ",[396,3541,495],{}," → open the dataset (e.g. ",[404,3544,512],{},") → spot-check 1–2 items",[393,3547,3548,3549,407,3552,2819,3555,2819,3557],{},"Click ",[396,3550,3551],{},"Start Experiment",[396,3553,3554],{},"Run Experiment",[396,3556,3370],{},[396,3558,3559],{},"Create",[393,3561,3562,3563],{},"Configure:\n",[487,3564,3565,3571,3579,3584,3589,3602],{},[393,3566,3567,3570],{},[396,3568,3569],{},"Experiment name"," (see naming below)",[393,3572,3573,3575,3576],{},[396,3574,3435],{}," + ",[396,3577,3578],{},"prompt version",[393,3580,3581,3583],{},[396,3582,3521],{}," \u002F model settings",[393,3585,3586,3588],{},[396,3587,459],{}," (usually already selected)",[393,3590,3591,3592,3595,3596,685,3598,685,3600,414],{},"Optional: ",[396,3593,3594],{},"structured output"," schema (recommended for Findings: ",[404,3597,2251],{},[404,3599,2270],{},[404,3601,2289],{},[393,3603,3604,3606],{},[396,3605,2818],{}," to attach (all gate evaluators)",[393,3608,3548,3609],{},[396,3610,3559],{},[375,3612,3613],{},"Langfuse runs the prompt per item, stores outputs, runs evaluators asynchronously, and shows aggregate scores. Runtime depends on dataset size, prompt length, and judge count.",[574,3615,3617],{"id":3616},"_34-experiment-naming","3.4 Experiment naming",[629,3619,3622],{"className":3620,"code":3621,"language":634,"meta":635},[632],"{agent}-{role}-{promptVersion}-{yyyymmdd}\n",[404,3623,3621],{"__ignoreMap":635},[375,3625,3626],{},"Examples:",[487,3628,3629,3634],{},[393,3630,3631],{},[404,3632,3633],{},"findings-baseline-v12-20260729",[393,3635,3636],{},[404,3637,3638],{},"findings-candidate-v13-20260729",[375,3640,3641,3642,3645,3646,3649,3650,3653],{},"For prompt gates: ",[396,3643,3644],{},"baseline"," = current production prompt; ",[396,3647,3648],{},"candidate"," = proposed version; ",[396,3651,3652],{},"same dataset + same evaluators"," for both.",[574,3655,3657],{"id":3656},"_35-compare-experiments","3.5 Compare experiments",[375,3659,3660],{},"After runs complete:",[390,3662,3663,3668,3671],{},[393,3664,2815,3665,3667],{},[396,3666,2899],{}," (or the dataset’s Experiments tab)",[393,3669,3670],{},"Select baseline and candidate runs",[393,3672,3673],{},"Compare aggregate scores, item-level regressions (especially boolean fails), and judge comments on failures",[375,3675,3676],{},"Always spot-check a few failed items manually before promoting.",[574,3678,3680],{"id":3679},"_36-experiment-reference-expand-as-needed","3.6 Experiment reference (expand as needed)",[725,3682,3683,3792,3929,3979,4020,4054,4144,4216],{},[728,3684,3686,3695,3789],{"icon":3056,"label":3685},"3.6.1 Prompt ↔ dataset variable mapping (Findings)",[375,3687,3688,3689,3691,3692,588],{},"A prompt is usable for Prompt Experiments when its ",[404,3690,2832],{}," match dataset item ",[396,3693,3694],{},"input keys",[434,3696,3697,3710],{},[437,3698,3699],{},[440,3700,3701,3704],{},[443,3702,3703],{},"Prompt variable",[443,3705,3706,3707,3709],{},"Dataset ",[404,3708,406],{}," key",[450,3711,3712,3723,3734,3745,3756,3767,3778],{},[440,3713,3714,3719],{},[455,3715,3716],{},[404,3717,3718],{},"{{user_name}}",[455,3720,3721],{},[404,3722,1322],{},[440,3724,3725,3730],{},[455,3726,3727],{},[404,3728,3729],{},"{{user_role}}",[455,3731,3732],{},[404,3733,1342],{},[440,3735,3736,3741],{},[455,3737,3738],{},[404,3739,3740],{},"{{organisation_context}}",[455,3742,3743],{},[404,3744,1362],{},[440,3746,3747,3752],{},[455,3748,3749],{},[404,3750,3751],{},"{{source_type}}",[455,3753,3754],{},[404,3755,1382],{},[440,3757,3758,3763],{},[455,3759,3760],{},[404,3761,3762],{},"{{source_info}}",[455,3764,3765],{},[404,3766,1402],{},[440,3768,3769,3774],{},[455,3770,3771],{},[404,3772,3773],{},"{{task_directive}}",[455,3775,3776],{},[404,3777,1525],{},[440,3779,3780,3785],{},[455,3781,3782],{},[404,3783,3784],{},"{{interview_transcript}}",[455,3786,3787],{},[404,3788,1545],{},[375,3790,3791],{},"If variables and input keys do not match, the experiment will fail or run with empty fields.",[728,3793,3796,3801,3832,3837,3866,3871,3874,3877,3883,3888],{"icon":3794,"label":3795},"i-lucide-git-branch","3.6.2 Findings Agent workflow (baseline → candidate → decide)",[375,3797,3798],{},[396,3799,3800],{},"A) Baseline (current production prompt)",[390,3802,3803,3807,3810,3816,3823,3829],{},[393,3804,2815,3805],{},[404,3806,512],{},[393,3808,3809],{},"Start Prompt Experiment",[393,3811,3812,3813],{},"Select ",[396,3814,3815],{},"current production prompt version",[393,3817,3818,3819,685,3821,414],{},"Attach gate evaluators (",[404,3820,2760],{},[404,3822,2766],{},[393,3824,3825,3826],{},"Name: ",[404,3827,3828],{},"findings-baseline-\u003Cprod-version>",[393,3830,3831],{},"Wait for scores",[375,3833,3834],{},[396,3835,3836],{},"B) Candidate (new prompt)",[390,3838,3839,3846,3855,3861],{},[393,3840,3841,3842,3845],{},"Save prompt edits as a ",[396,3843,3844],{},"new prompt version"," in Prompt Management",[393,3847,3848,3849,714,3852,3854],{},"Run another Prompt Experiment on the ",[396,3850,3851],{},"same",[404,3853,512],{}," dataset",[393,3856,3857,3858,3860],{},"Attach the ",[396,3859,3851],{}," evaluators",[393,3862,3825,3863],{},[404,3864,3865],{},"findings-candidate-\u003Cnew-version>",[375,3867,3868],{},[396,3869,3870],{},"C) Decide",[375,3872,3873],{},"Promote only if grounding and directive alignment do not regress materially, and no new systemic failure pattern appears in item review.",[375,3875,3876],{},"If it fails: revise the candidate prompt and rerun candidate only (keep baseline fixed).",[629,3878,3881],{"className":3879,"code":3880,"language":634,"meta":635},[632],"Prod prompt vN\n   │\n   ▼\nBaseline experiment on findings\u002Fgate_test + gate evaluators\n   │\n   ▼\nEdit prompt → save vN+1\n   │\n   ▼\nCandidate experiment on SAME dataset + SAME evaluators\n   │\n   ▼\nCompare in Langfuse Experiments\n   │\n   ├─ Pass → promote vN+1\n   └─ Fail → revise prompt, rerun candidate\n",[404,3882,3880],{"__ignoreMap":635},[375,3884,3885],{},[396,3886,3887],{},"What to look for",[434,3889,3890,3899],{},[437,3891,3892],{},[440,3893,3894,3896],{},[443,3895,469],{},[443,3897,3898],{},"Prefer",[450,3900,3901,3911],{},[440,3902,3903,3908],{},[455,3904,3905],{},[404,3906,3907],{},"grounding",[455,3909,3910],{},"pass-rate ≥ baseline",[440,3912,3913,3918],{},[455,3914,3915],{},[404,3916,3917],{},"directive_alignment",[455,3919,3920,3921,3924,3925,3928],{},"higher ",[404,3922,3923],{},"% pass",", lower ",[404,3926,3927],{},"% fail",", mean mapped score ≥ baseline − tolerance",[728,3930,3933,3966,3973],{"icon":3931,"label":3932},"i-lucide-app-window","3.6.3 UI vs SDK experiments",[434,3934,3935,3944],{},[437,3936,3937],{},[440,3938,3939,3942],{},[443,3940,3941],{},"Approach",[443,3943,2935],{},[450,3945,3946,3956],{},[440,3947,3948,3953],{},[455,3949,3950],{},[396,3951,3952],{},"Experiments via UI (Prompt Experiments)",[455,3954,3955],{},"Prompt-only changes; variables map cleanly from dataset input",[440,3957,3958,3963],{},[455,3959,3960],{},[396,3961,3962],{},"Experiments via SDK",[455,3964,3965],{},"Full app\u002Fagent logic, tools, retrieval, custom runtime config",[375,3967,3968,3969,3972],{},"Findings Agent ",[396,3970,3971],{},"prompt iteration"," fits UI Prompt Experiments well when structured output is enforced.",[375,3974,3975,3976,3978],{},"If the flow depends heavily on tool calls \u002F multi-step orchestration that Prompt Experiments cannot reproduce, use ",[396,3977,3962],{}," (or hybrid: UI for prompt drafts, SDK for full-agent realism).",[728,3980,3983,3986,4011,4017],{"icon":3981,"label":3982},"i-lucide-snowflake","3.6.4 Dataset freeze rules during experiments",[375,3984,3985],{},"To keep comparisons fair:",[390,3987,3988,3997,4005],{},[393,3989,3990,3991,3993,3994,3996],{},"Do ",[396,3992,795],{}," edit\u002Fadd\u002Fdelete ",[404,3995,876],{}," items while comparing prompts",[393,3998,3999,4000,4002,4003],{},"Put fresh cases into ",[404,4001,880],{}," (or a scratch set), not into ",[404,4004,876],{},[393,4006,4007,4008,4010],{},"Only expand ",[404,4009,876],{}," with reviewed items after the current comparison cycle",[375,4012,4013,4014,4016],{},"Langfuse experiments run against the dataset state at experiment time. Treat ",[404,4015,876],{}," as frozen for the duration of a promotion decision.",[375,4018,4019],{},"Optional: use dataset versioning (Items tab → version view) when available, so you can re-run against a historical snapshot.",[728,4021,4024,4030,4046,4051],{"icon":4022,"label":4023},"i-lucide-braces","3.6.5 Structured output tip (Findings)",[375,4025,4026,4027,4029],{},"For Findings experiments, enable ",[396,4028,3594],{}," with a schema requiring:",[487,4031,4032,4037,4041],{},[393,4033,4034,4036],{},[404,4035,2251],{}," (string)",[393,4038,4039,4036],{},[404,4040,2270],{},[393,4042,4043,4045],{},[404,4044,2289],{}," (array)",[375,4047,4048,4049,3004],{},"This improves parseability for judges, consistency across items, and JSONPath mapping (e.g. ",[404,4050,3108],{},[375,4052,4053],{},"Schemas can be created\u002Fsaved in Langfuse Playground and reused in experiments.",[728,4055,4058],{"icon":4056,"label":4057},"i-lucide-bug","3.6.6 Debugging failed or empty scores",[434,4059,4060,4073],{},[437,4061,4062],{},[440,4063,4064,4067,4070],{},[443,4065,4066],{},"Symptom",[443,4068,4069],{},"Likely cause",[443,4071,4072],{},"Fix",[450,4074,4075,4086,4097,4108,4119,4130],{},[440,4076,4077,4080,4083],{},[455,4078,4079],{},"Experiment fails immediately",[455,4081,4082],{},"Prompt variables ≠ dataset input keys",[455,4084,4085],{},"Align names exactly",[440,4087,4088,4091,4094],{},[455,4089,4090],{},"Empty outputs",[455,4092,4093],{},"LLM connection \u002F model issue",[455,4095,4096],{},"Check project LLM connection + logs",[440,4098,4099,4102,4105],{},[455,4100,4101],{},"No evaluator scores",[455,4103,4104],{},"Evaluator not attached \u002F wrong target",[455,4106,4107],{},"Attach evaluators; target Experiments",[440,4109,4110,4113,4116],{},[455,4111,4112],{},"Judge mapping empty",[455,4114,4115],{},"Wrong source or JSONPath",[455,4117,4118],{},"Fix mapping; use Prompt Preview",[440,4120,4121,4124,4127],{},[455,4122,4123],{},"Noisy scores",[455,4125,4126],{},"Judge prompt too broad",[455,4128,4129],{},"Split into one-dimension evaluators",[440,4131,4132,4135,4138],{},[455,4133,4134],{},"Need judge internals",[455,4136,4137],{},"—",[455,4139,4140,4141],{},"Filter traces by environment ",[404,4142,4143],{},"langfuse-llm-as-a-judge",[728,4145,4147],{"icon":2575,"label":4146},"3.6.7 Checklist before promoting a prompt",[487,4148,4150,4156,4162,4173,4180,4186,4192,4198,4204,4210],{"className":4149},[2580],[393,4151,4153,4155],{"className":4152},[2584],[406,4154],{"disabled":2429,"type":2587}," Prompt version is saved in Prompt Management (not an unsaved playground edit)",[393,4157,4159,4161],{"className":4158},[2584],[406,4160],{"disabled":2429,"type":2587}," Baseline experiment exists for current production prompt",[393,4163,4165,4167,4168,4170,4171,414],{"className":4164},[2584],[406,4166],{"disabled":2429,"type":2587}," Candidate experiment used the ",[396,4169,3851],{}," dataset (e.g. ",[404,4172,512],{},[393,4174,4176,4167,4178,3860],{"className":4175},[2584],[406,4177],{"disabled":2429,"type":2587},[396,4179,3851],{},[393,4181,4183,4185],{"className":4182},[2584],[406,4184],{"disabled":2429,"type":2587}," Structured output schema enabled (if required by agent)",[393,4187,4189,4191],{"className":4188},[2584],[406,4190],{"disabled":2429,"type":2587}," Aggregate scores reviewed",[393,4193,4195,4197],{"className":4194},[2584],[406,4196],{"disabled":2429,"type":2587}," Item-level failures reviewed (especially grounding fails)",[393,4199,4201,4203],{"className":4200},[2584],[406,4202],{"disabled":2429,"type":2587}," No dataset edits happened between baseline and candidate",[393,4205,4207,4209],{"className":4206},[2584],[406,4208],{"disabled":2429,"type":2587}," Promotion decision documented (pass \u002F fail + reason)",[393,4211,4213,4215],{"className":4212},[2584],[406,4214],{"disabled":2429,"type":2587}," Production prompt pointer updated only after pass",[728,4217,4220,4297,4302],{"icon":4218,"label":4219},"i-lucide-bookmark","3.6.8 Quick reference — Findings Agent",[434,4221,4222,4231],{},[437,4223,4224],{},[440,4225,4226,4228],{},[443,4227,445],{},[443,4229,4230],{},"Langfuse object",[450,4232,4233,4242,4252,4259,4270,4278,4289],{},[440,4234,4235,4238],{},[455,4236,4237],{},"Gate dataset",[455,4239,4240],{},[404,4241,512],{},[440,4243,4244,4247],{},[455,4245,4246],{},"Smoke dataset",[455,4248,4249],{},[404,4250,4251],{},"findings\u002Fsmoke_test",[440,4253,4254,4256],{},[455,4255,3435],{},[455,4257,4258],{},"Findings Agent prompt in Prompt Management",[440,4260,4261,4264],{},[455,4262,4263],{},"Gate evaluators",[455,4265,4266,685,4268],{},[404,4267,2760],{},[404,4269,2766],{},[440,4271,4272,4275],{},[455,4273,4274],{},"Experiment type",[455,4276,4277],{},"Prompt Experiment (UI)",[440,4279,4280,4283],{},[455,4281,4282],{},"Example runs",[455,4284,4285],{},[379,4286,4288],{"href":507,"rel":4287},[383],"findings\u002Fgate_test experiments",[440,4290,4291,4294],{},[455,4292,4293],{},"Decision",[455,4295,4296],{},"Compare baseline vs candidate → promote only on gate pass",[375,4298,4299],{},[396,4300,4301],{},"Tips",[390,4303,4304,4312,4318,4321],{},[393,4305,4306,4307,407,4309,4311],{},"Keep experiment names searchable (",[404,4308,3644],{},[404,4310,3648],{}," + prompt version).",[393,4313,4314,4315,4317],{},"Prefer foldered datasets (",[404,4316,512],{},") for clarity in the Datasets UI.",[393,4319,4320],{},"Attach all gate evaluators on every promotion run — do not compare incomplete score sets.",[393,4322,3101,4323,4325,4326,588],{},[404,4324,880],{}," for quick checks; promote prompts using ",[404,4327,876],{},[518,4329],{},[521,4331,4333],{"id":4332},"_4-related-docs","4. Related docs",[487,4335,4336,4340,4347,4353,4359,4364,4369],{},[393,4337,4338],{},[379,4339,350],{"href":351},[393,4341,4342],{},[379,4343,4346],{"href":4344,"rel":4345},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Foverview",[383],"Langfuse evaluation overview",[393,4348,4349],{},[379,4350,4352],{"href":2854,"rel":4351},[383],"Langfuse LLM-as-a-Judge",[393,4354,4355],{},[379,4356,4358],{"href":2676,"rel":4357},[383],"Langfuse Code Evaluators",[393,4360,4361],{},[379,4362,719],{"href":717,"rel":4363},[383],[393,4365,4366],{},[379,4367,3388],{"href":3386,"rel":4368},[383],[393,4370,4371],{},[379,4372,4375],{"href":4373,"rel":4374},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fscores\u002Fdata-model",[383],"Scores data model",[4377,4378,4379],"style",{},"html pre.shiki code .sMK4o, html code.shiki .sMK4o{--shiki-light:#39ADB5;--shiki-default:#89DDFF;--shiki-dark:#89DDFF}html pre.shiki code .spNyl, html code.shiki .spNyl{--shiki-light:#9C3EDA;--shiki-default:#C792EA;--shiki-dark:#C792EA}html pre.shiki code .sBMFI, html code.shiki .sBMFI{--shiki-light:#E2931D;--shiki-default:#FFCB6B;--shiki-dark:#FFCB6B}html pre.shiki code .sfazB, html code.shiki .sfazB{--shiki-light:#91B859;--shiki-default:#C3E88D;--shiki-dark:#C3E88D}html .light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html.light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .sbssI, html code.shiki .sbssI{--shiki-light:#F76D47;--shiki-default:#F78C6C;--shiki-dark:#F78C6C}html pre.shiki code .sTEyZ, html code.shiki .sTEyZ{--shiki-light:#90A4AE;--shiki-default:#EEFFFF;--shiki-dark:#BABED8}",{"title":635,"searchDepth":986,"depth":986,"links":4381},[4382,4388,4394,4402],{"id":523,"depth":986,"text":524,"children":4383},[4384,4385,4386,4387],{"id":576,"depth":1005,"text":577},{"id":626,"depth":1005,"text":627},{"id":693,"depth":1005,"text":694},{"id":722,"depth":1005,"text":723},{"id":2646,"depth":986,"text":2647,"children":4389},[4390,4391,4392,4393],{"id":2699,"depth":1005,"text":2700},{"id":2779,"depth":1005,"text":2780},{"id":2798,"depth":1005,"text":2799},{"id":2859,"depth":1005,"text":2860},{"id":3356,"depth":986,"text":3357,"children":4395},[4396,4397,4398,4399,4400,4401],{"id":3397,"depth":1005,"text":3398},{"id":3486,"depth":1005,"text":3487},{"id":3533,"depth":1005,"text":3534},{"id":3616,"depth":1005,"text":3617},{"id":3656,"depth":1005,"text":3657},{"id":3679,"depth":1005,"text":3680},{"id":4332,"depth":986,"text":4333},"How to evaluate LLM agents with Langfuse — datasets, evaluators, and experiments.","md",null,{},{"title":358,"description":4403},"0EnGvDfgEMon3yvnwyp4Na6BavnAN1Xc_jwDy73oPok",[4410,4412],{"title":354,"path":355,"stem":356,"description":4411,"children":-1},"This document provides a high-level overview, a component breakdown, and detailed diagrams of how ESPi (ESPROFILER Intelligence), the AI co-pilot and agent orchestrator built into ESPROFILER, handles and processes user queries.",{"title":362,"path":363,"stem":364,"description":635,"children":-1},1790076748517]