[{"data":1,"prerenderedAt":4392},["ShallowReactive",2],{"navigation_docs":3,"-data-ops-llm-ops-evaluations":348,"-data-ops-llm-ops-evaluations-surround":4388},[4,8,68,98,216,245,259,280,344],{"title":5,"path":6,"stem":7},"Introduction","\u002Fintroduction","0.introduction",{"title":9,"icon":10,"path":11,"stem":12,"children":13,"page":63},"Company","i-lucide-building-2","\u002Fcompany","1.company",[14,18,22,26,30,34,38,42,46,50,64],{"title":15,"path":16,"stem":17},"About","\u002Fcompany\u002Fabout","1.company\u002F0.about",{"title":19,"path":20,"stem":21},"Values","\u002Fcompany\u002Fvalues","1.company\u002F1.values",{"title":23,"path":24,"stem":25},"Communication","\u002Fcompany\u002Fcommunication","1.company\u002Fcommunication",{"title":27,"path":28,"stem":29},"Competition","\u002Fcompany\u002Fcompetition","1.company\u002Fcompetition",{"title":31,"path":32,"stem":33},"Hybrid Working","\u002Fcompany\u002Fhybrid-working","1.company\u002Fhybrid-working",{"title":35,"path":36,"stem":37},"Manchester Office","\u002Fcompany\u002Foffice","1.company\u002Foffice",{"title":39,"path":40,"stem":41},"Operations","\u002Fcompany\u002Foperations","1.company\u002Foperations",{"title":43,"path":44,"stem":45},"Policies","\u002Fcompany\u002Fpolicies","1.company\u002Fpolicies",{"title":47,"path":48,"stem":49},"Product Strategy","\u002Fcompany\u002Fproduct-strategy","1.company\u002Fproduct-strategy",{"title":51,"path":52,"stem":53,"children":54,"page":63},"Products","\u002Fcompany\u002Fproducts","1.company\u002Fproducts",[55,59],{"title":56,"path":57,"stem":58},"Capability Exchange","\u002Fcompany\u002Fproducts\u002Fcapability-exchange","1.company\u002Fproducts\u002Fcapability-exchange",{"title":60,"path":61,"stem":62},"ESProfiler Platform","\u002Fcompany\u002Fproducts\u002Fesprofiler","1.company\u002Fproducts\u002Fesprofiler",false,{"title":65,"path":66,"stem":67},"Security","\u002Fcompany\u002Fsecurity","1.company\u002Fsecurity",{"title":69,"icon":70,"path":71,"stem":72,"children":73,"page":63},"People Ops","i-lucide-users","\u002Fpeople-ops","2.people-ops",[74,78,82,86,90,94],{"title":75,"path":76,"stem":77},"Compensation","\u002Fpeople-ops\u002Fcompensation","2.people-ops\u002Fcompensation",{"title":79,"path":80,"stem":81},"Education","\u002Fpeople-ops\u002Feducation","2.people-ops\u002Feducation",{"title":83,"path":84,"stem":85},"Expenses","\u002Fpeople-ops\u002Fexpenses","2.people-ops\u002Fexpenses",{"title":87,"path":88,"stem":89},"Holiday & Leave","\u002Fpeople-ops\u002Fleave","2.people-ops\u002Fleave",{"title":91,"path":92,"stem":93},"Onboarding","\u002Fpeople-ops\u002Fonboarding","2.people-ops\u002Fonboarding",{"title":95,"path":96,"stem":97},"Recruitment","\u002Fpeople-ops\u002Frecruitment","2.people-ops\u002Frecruitment",{"title":99,"icon":100,"path":101,"stem":102,"children":103,"page":63},"Engineering","i-lucide-rocket","\u002Fengineering","3.engineering",[104,147,151,171,192,196,204,208,212],{"title":105,"path":106,"stem":107,"children":108,"page":63},"Contributing","\u002Fengineering\u002Fcontributing","3.engineering\u002Fcontributing",[109,113,117,121,125,138],{"title":110,"path":111,"stem":112},"Development Setup","\u002Fengineering\u002Fcontributing\u002Fdevelopment-setup","3.engineering\u002Fcontributing\u002F1.development-setup",{"title":114,"path":115,"stem":116},"Engineering Operations","\u002Fengineering\u002Fcontributing\u002Fengineering-operations","3.engineering\u002Fcontributing\u002F2.engineering-operations",{"title":118,"path":119,"stem":120},"Documentation","\u002Fengineering\u002Fcontributing\u002Fdocumentation","3.engineering\u002Fcontributing\u002F3.documentation",{"title":122,"path":123,"stem":124},"Agentic Coding","\u002Fengineering\u002Fcontributing\u002Fagentic-coding","3.engineering\u002Fcontributing\u002Fagentic-coding",{"title":126,"path":127,"stem":128,"children":129,"page":63},"Back End","\u002Fengineering\u002Fcontributing\u002Fback-end","3.engineering\u002Fcontributing\u002Fback-end",[130,134],{"title":131,"path":132,"stem":133},"API Guidelines","\u002Fengineering\u002Fcontributing\u002Fback-end\u002Fapi-guidelines","3.engineering\u002Fcontributing\u002Fback-end\u002Fapi-guidelines",{"title":135,"path":136,"stem":137},"LLM Prompts & Langfuse Integration","\u002Fengineering\u002Fcontributing\u002Fback-end\u002Fllm-prompts","3.engineering\u002Fcontributing\u002Fback-end\u002Fllm-prompts",{"title":139,"path":140,"stem":141,"children":142,"page":63},"Front End","\u002Fengineering\u002Fcontributing\u002Ffront-end","3.engineering\u002Fcontributing\u002Ffront-end",[143],{"title":144,"path":145,"stem":146},"Testing","\u002Fengineering\u002Fcontributing\u002Ffront-end\u002Ftesting","3.engineering\u002Fcontributing\u002Ffront-end\u002Ftesting",{"title":148,"path":149,"stem":150},"Production Database","\u002Fengineering\u002Fdatabase-connection","3.engineering\u002Fdatabase-connection",{"title":152,"path":153,"stem":154,"children":155},"Deployment","\u002Fengineering\u002Fdeployment","3.engineering\u002Fdeployment",[156,159,163,167],{"title":56,"path":157,"stem":158},"\u002Fengineering\u002Fdeployment\u002Fcapability-exchange","3.engineering\u002Fdeployment\u002Fcapability-exchange",{"title":160,"path":161,"stem":162},"Langfuse Deployment","\u002Fengineering\u002Fdeployment\u002Fecs-langfuse-deployment","3.engineering\u002Fdeployment\u002Fecs-langfuse-deployment",{"title":164,"path":165,"stem":166},"ESP Platform Configuration","\u002Fengineering\u002Fdeployment\u002Fesp-platform-configuration","3.engineering\u002Fdeployment\u002Fesp-platform-configuration",{"title":168,"path":169,"stem":170},"Platform","\u002Fengineering\u002Fdeployment\u002Fplatform","3.engineering\u002Fdeployment\u002Fplatform",{"title":172,"path":173,"stem":174,"children":175,"page":63},"Github","\u002Fengineering\u002Fgithub","3.engineering\u002Fgithub",[176,180,184,188],{"title":177,"path":178,"stem":179},"Packages","\u002Fengineering\u002Fgithub\u002Fpackages","3.engineering\u002Fgithub\u002Fpackages",{"title":181,"path":182,"stem":183},"Personal Access Token","\u002Fengineering\u002Fgithub\u002Fpersonal-access-token","3.engineering\u002Fgithub\u002Fpersonal-access-token",{"title":185,"path":186,"stem":187},"Troubleshooting","\u002Fengineering\u002Fgithub\u002Ftroubleshooting","3.engineering\u002Fgithub\u002Ftroubleshooting",{"title":189,"path":190,"stem":191},"Workflows","\u002Fengineering\u002Fgithub\u002Fworkflows","3.engineering\u002Fgithub\u002Fworkflows",{"title":193,"path":194,"stem":195},"Platform Ops","\u002Fengineering\u002Fplatform-ops","3.engineering\u002Fplatform-ops",{"title":168,"path":197,"stem":198,"children":199,"page":63},"\u002Fengineering\u002Fplatform","3.engineering\u002Fplatform",[200],{"title":201,"path":202,"stem":203},"useAPI","\u002Fengineering\u002Fplatform\u002Fuse-api","3.engineering\u002Fplatform\u002Fuse-api",{"title":205,"path":206,"stem":207},"Project Management","\u002Fengineering\u002Fproject-management","3.engineering\u002Fproject-management",{"title":209,"path":210,"stem":211},"Releases","\u002Fengineering\u002Frelease","3.engineering\u002Frelease",{"title":213,"path":214,"stem":215},"Tools","\u002Fengineering\u002Ftools","3.engineering\u002Ftools",{"title":217,"icon":218,"path":219,"stem":220,"children":221,"page":63},"Design","i-lucide-palette","\u002Fdesign","4.design",[222,226,230,234,238,241],{"title":223,"path":224,"stem":225},"Design Thinking","\u002Fdesign\u002Fdesign-thinking","4.design\u002F1.design-thinking",{"title":227,"path":228,"stem":229},"Figma","\u002Fdesign\u002Ffigma-structure","4.design\u002F2.figma-structure",{"title":231,"path":232,"stem":233},"Design & Development","\u002Fdesign\u002Fdesign-and-development","4.design\u002F3.design-and-development",{"title":235,"path":236,"stem":237},"Branding","\u002Fdesign\u002Fbranding","4.design\u002F4.branding",{"title":213,"path":239,"stem":240},"\u002Fdesign\u002Ftools","4.design\u002F5.tools",{"title":242,"path":243,"stem":244},"Customer Success","\u002Fdesign\u002Fworking-with-customers","4.design\u002F6.working-with-customers",{"title":246,"icon":247,"path":248,"stem":249,"children":250,"page":63},"Sales","i-lucide-dollar-sign","\u002Fsales","4.sales",[251,255],{"title":252,"path":253,"stem":254},"Customer Onboarding","\u002Fsales\u002Fonboarding","4.sales\u002Fonboarding",{"title":256,"path":257,"stem":258},"Sales Tools","\u002Fsales\u002Ftools","4.sales\u002Ftools",{"title":260,"icon":261,"path":262,"stem":263,"children":264,"page":63},"Marketing","i-lucide-book-image","\u002Fmarketing","5.marketing",[265,269,273,276],{"title":266,"path":267,"stem":268},"Content","\u002Fmarketing\u002Fcontent","5.marketing\u002Fcontent",{"title":270,"path":271,"stem":272},"Messaging","\u002Fmarketing\u002Fmessaging","5.marketing\u002Fmessaging",{"title":213,"path":274,"stem":275},"\u002Fmarketing\u002Ftools","5.marketing\u002Ftools",{"title":277,"path":278,"stem":279},"Website","\u002Fmarketing\u002Fwebsite","5.marketing\u002Fwebsite",{"title":281,"icon":282,"path":283,"stem":284,"children":285,"page":63},"AI & Data Ops","i-lucide-database","\u002Fdata-ops","6.data-ops",[286,294,298,323,340],{"title":56,"path":287,"stem":288,"children":289,"page":63},"\u002Fdata-ops\u002Fcapability-exchange","6.data-ops\u002FCapability Exchange",[290],{"title":291,"path":292,"stem":293},"Leaderboard Calculation","\u002Fdata-ops\u002Fcapability-exchange\u002Fleaderboard-calculation","6.data-ops\u002FCapability Exchange\u002Fleaderboard-calculation",{"title":295,"path":296,"stem":297},"Account Portal (CAS)","\u002Fdata-ops\u002Faccount-portal","6.data-ops\u002Faccount-portal",{"title":299,"path":300,"stem":301,"children":302,"page":63},"Data Management","\u002Fdata-ops\u002Fdata-management","6.data-ops\u002Fdata-management",[303,307,311,315,319],{"title":304,"path":305,"stem":306},"Adding Products","\u002Fdata-ops\u002Fdata-management\u002Fadding-products","6.data-ops\u002Fdata-management\u002Fadding-products",{"title":308,"path":309,"stem":310},"Adding Vendors","\u002Fdata-ops\u002Fdata-management\u002Fadding-vendors","6.data-ops\u002Fdata-management\u002Fadding-vendors",{"title":312,"path":313,"stem":314},"Framework Mapping","\u002Fdata-ops\u002Fdata-management\u002Fframework-mapping","6.data-ops\u002Fdata-management\u002Fframework-mapping",{"title":316,"path":317,"stem":318},"Refreshing Vendors","\u002Fdata-ops\u002Fdata-management\u002Frefreshing-vendors","6.data-ops\u002Fdata-management\u002Frefreshing-vendors",{"title":320,"path":321,"stem":322},"Reviewing Draft Vendors","\u002Fdata-ops\u002Fdata-management\u002Freviewing-draft-vendors","6.data-ops\u002Fdata-management\u002Freviewing-draft-vendors",{"title":324,"path":325,"stem":326,"children":327,"page":63},"LLM Ops","\u002Fdata-ops\u002Fllm-ops","6.data-ops\u002Fllm-ops",[328,332,336],{"title":329,"path":330,"stem":331},"Agents","\u002Fdata-ops\u002Fllm-ops\u002Fagents","6.data-ops\u002Fllm-ops\u002F1.agents",{"title":333,"path":334,"stem":335},"ESPi Architecture & Query Flow","\u002Fdata-ops\u002Fllm-ops\u002Fespi-architecture","6.data-ops\u002Fllm-ops\u002F2.espi-architecture",{"title":337,"path":338,"stem":339},"Evaluating Agents","\u002Fdata-ops\u002Fllm-ops\u002Fevaluations","6.data-ops\u002Fllm-ops\u002F3.evaluations",{"title":341,"path":342,"stem":343},"Message Queues","\u002Fdata-ops\u002Fmessage-queues","6.data-ops\u002Fmessage-queues",{"title":345,"path":346,"stem":347},"Glossary","\u002Fglossary","glossary",{"id":349,"title":337,"body":350,"description":4382,"extension":4383,"links":4384,"meta":4385,"navigation":2408,"path":338,"seo":4386,"stem":339,"__hash__":4387},"docs\u002F6.data-ops\u002Fllm-ops\u002F3.evaluations.md",{"type":351,"value":352,"toc":4359},"minimark",[353,365,368,412,462,465,492,496,499,504,507,552,557,600,603,607,617,658,667,670,674,699,703,2621,2623,2627,2630,2637,2659,2662,2676,2680,2723,2730,2733,2756,2760,2766,2775,2779,2828,2836,2840,3331,3333,3337,3351,3360,3374,3378,3447,3453,3463,3467,3470,3510,3514,3590,3593,3597,3603,3606,3618,3633,3637,3640,3653,3656,3660,4307,4309,4313,4355],[354,355,356,357,364],"p",{},"We use ",[358,359,363],"a",{"href":360,"rel":361},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development",[362],"nofollow","Langfuse"," to evaluate LLM agents in the ESProfiler ecosystem.",[354,366,367],{},"Evaluations follow three steps:",[369,370,371,394,403],"ol",{},[372,373,374,381,382,386,387,386,390,393],"li",{},[375,376,377],"strong",{},[358,378,380],{"href":379},"#1-creating-datasets","Create datasets"," — test cases (",[383,384,385],"code",{},"input"," \u002F ",[383,388,389],{},"expectedOutput",[383,391,392],{},"metadata",")",[372,395,396,402],{},[375,397,398],{},[358,399,401],{"href":400},"#2-creating-evaluators","Create evaluators"," — scoring definitions (mostly LLM-as-judge)",[372,404,405,411],{},[375,406,407],{},[358,408,410],{"href":409},"#3-running-experiments","Run experiments"," — prompt + dataset + evaluators",[413,414,415,428],"table",{},[416,417,418],"thead",{},[419,420,421,425],"tr",{},[422,423,424],"th",{},"Piece",[422,426,427],{},"Role",[429,430,431,442,452],"tbody",{},[419,432,433,439],{},[434,435,436],"td",{},[375,437,438],{},"Dataset",[434,440,441],{},"Reusable test cases",[419,443,444,449],{},[434,445,446],{},[375,447,448],{},"Evaluator",[434,450,451],{},"Scores one quality dimension of an agent output",[419,453,454,459],{},[434,455,456],{},[375,457,458],{},"Experiment",[434,460,461],{},"Runs a prompt against a dataset and applies evaluators",[354,463,464],{},"Existing resources in ESP Development:",[466,467,468,475,482],"ul",{},[372,469,470],{},[358,471,474],{"href":472,"rel":473},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development\u002Fdatasets?pageIndex=0&pageSize=50",[362],"Datasets",[372,476,477],{},[358,478,481],{"href":479,"rel":480},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development\u002Fevals\u002Ftemplates",[362],"Evaluator templates",[372,483,484],{},[358,485,488,489],{"href":486,"rel":487},"https:\u002F\u002Fesplf.esprofiler.com\u002Fproject\u002Fesp-development\u002Fdatasets\u002Fcmruuouiv0021mk07r7itcz46\u002Fexperiments",[362],"Example experiments: ",[383,490,491],{},"findings\u002Fgate_test",[493,494,495],"note",{},"A walkthrough video of the full Langfuse UI will be recorded soon and added to this page later.",[497,498],"hr",{},[500,501,503],"h2",{"id":502},"_1-creating-datasets","1. Creating Datasets",[354,505,506],{},"An evaluation dataset is a collection of test cases. Each item typically has:",[413,508,509,519],{},[416,510,511],{},[419,512,513,516],{},[422,514,515],{},"Field",[422,517,518],{},"Purpose",[429,520,521,530,543],{},[419,522,523,527],{},[434,524,525],{},[375,526,385],{},[434,528,529],{},"What the agent\u002Fprompt receives at runtime",[419,531,532,536],{},[434,533,534],{},[375,535,389],{},[434,537,538,539,542],{},"Golden answer ",[375,540,541],{},"or"," evaluation guidance",[419,544,545,549],{},[434,546,547],{},[375,548,392],{},[434,550,551],{},"Filtering, debugging, and audit context (not fed to the agent)",[553,554,556],"h3",{"id":555},"_11-create-your-first-dataset-5-steps","1.1 Create your first dataset (5 steps)",[369,558,559,568,574,588,594],{},[372,560,561,564,565,567],{},[375,562,563],{},"Pick one agent"," — see ",[358,566,329],{"href":330},".",[372,569,570,573],{},[375,571,572],{},"Decide expected-output style"," — golden answer for deterministic tasks; judge guidance for open-ended ones.",[372,575,576,579,580,582,583,585,586,567],{},[375,577,578],{},"Write 3–5 items"," — ",[383,581,385],{}," keys must match that agent’s prompt variables; add ",[383,584,389],{}," and optional ",[383,587,392],{},[372,589,590,593],{},[375,591,592],{},"Anonymize if data came from production"," — never upload raw tenant\u002FPII.",[372,595,596,599],{},[375,597,598],{},"Create and upload"," — UI for tiny flat cases; Python SDK for nested JSON.",[354,601,602],{},"Then spot-check 2–3 items in Langfuse before attaching evaluators.",[553,604,606],{"id":605},"_12-dataset-naming","1.2 Dataset naming",[608,609,615],"pre",{"className":610,"code":612,"language":613,"meta":614},[611],"language-text","{agentName}\u002F{datasetRole}\n","text","",[383,616,612],{"__ignoreMap":614},[413,618,619,630],{},[416,620,621],{},[419,622,623,625,627],{},[422,624,438],{},[422,626,427],{},[422,628,629],{},"Size guidance",[429,631,632,645],{},[419,633,634,639,642],{},[434,635,636],{},[383,637,638],{},"{agent}\u002Fsmoke_test",[434,640,641],{},"Quick sanity checks after prompt or wiring changes",[434,643,644],{},"~5–15 items",[419,646,647,652,655],{},[434,648,649],{},[383,650,651],{},"{agent}\u002Fgate_test",[434,653,654],{},"Broader frozen set for prompt comparison and release decisions",[434,656,657],{},"~20–50 items",[354,659,660,661,664,665,567],{},"Examples: ",[383,662,663],{},"conversation-namer\u002Fsmoke_test",", ",[383,666,491],{},[354,668,669],{},"Keep both sets reviewed and anonymized. Update them deliberately — do not treat one as a staging queue for the other.",[553,671,673],{"id":672},"_13-upload-overview","1.3 Upload overview",[466,675,676,682,688],{},[372,677,678,681],{},[375,679,680],{},"UI"," — Datasets → New dataset → add items (or CSV for flat strings). Best for small \u002F simple cases.",[372,683,684,687],{},[375,685,686],{},"Python SDK"," — preferred for nested JSON (Findings transcripts, structured guidance). Keys: Langfuse UI → Settings → API Keys.",[372,689,690,693,694],{},[375,691,692],{},"Reference:"," ",[358,695,698],{"href":696,"rel":697},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fexperiments\u002Fdatasets",[362],"Langfuse Datasets",[553,700,702],{"id":701},"_14-dataset-reference-expand-as-needed","1.4 Dataset reference (expand as needed)",[704,705,706,776,829,861,1049,1251,2342,2552],"accordion",{},[707,708,711,715,722,737,741,747,756,766,769],"accordion-item",{"icon":709,"label":710},"i-lucide-table","1.4.1 Field details (input \u002F expectedOutput \u002F metadata)",[712,713,714],"h4",{"id":385},"Input",[354,716,717,718,721],{},"What is injected into the prompt or application under test: a user message, nested ",[375,719,720],{},"prompt variables",", or a multi-turn conversation.",[354,723,724,693,727,729,730,733,734,567],{},[375,725,726],{},"Rule:",[383,728,385],{}," keys must match the prompt variables of the agent you are evaluating. Find those in the agent’s ",[383,731,732],{},".st"," prompt \u002F config in ",[383,735,736],{},"platform-api",[712,738,740],{"id":739},"expected-output","Expected output",[354,742,743,746],{},[375,744,745],{},"A) Golden answer (deterministic)"," — use when the correct answer is known and comparable (tags, short titles, exact fields). A code check is often enough. Curate with human review.",[354,748,749,752,753,755],{},[375,750,751],{},"B) Judge guidance (non-deterministic)"," — use when many good outputs exist (summaries, reports, findings). Put criteria in ",[383,754,389],{}," (what must be covered, constraints, grounding rules) — not a full golden dump.",[493,757,758,759,761,762,765],{},"For Findings-style agents: keep guidance in ",[383,760,389],{},"; put full historical report\u002Ffindings in ",[383,763,764],{},"metadata.reference_output"," for human review only.",[712,767,768],{"id":392},"Metadata",[354,770,771,772,775],{},"Anything that should ",[375,773,774],{},"not"," be fed as prompt input: tags, tool expectations, anonymization flags, debug goldens.",[707,777,780],{"icon":778,"label":779},"i-lucide-git-compare","1.4.2 Choose evaluation style",[413,781,782,794],{},[416,783,784],{},[419,785,786,789,792],{},[422,787,788],{},"Use case",[422,790,791],{},"Expected output style",[422,793,448],{},[429,795,796,807,818],{},[419,797,798,801,804],{},[434,799,800],{},"Categorization \u002F tagging \u002F short titles",[434,802,803],{},"Golden labels",[434,805,806],{},"Code check (app or unit tests)",[419,808,809,812,815],{},[434,810,811],{},"Tone, grounding, alignment, finding types",[434,813,814],{},"Guidance + rubric",[434,816,817],{},"LLM-as-judge",[419,819,820,823,826],{},[434,821,822],{},"Mixed",[434,824,825],{},"Guidance + optional reference in metadata",[434,827,828],{},"Code + LLM-as-judge",[707,830,833,836],{"icon":831,"label":832},"i-lucide-list-filter","1.4.3 How to select records",[354,834,835],{},"Start from production failure modes, not random sampling.",[369,837,838,841,844,847,850],{},[372,839,840],{},"Observe where the agent fails (wrong types, invented claims, missed coverage, etc.)",[372,842,843],{},"Pull representative completed cases for those modes",[372,845,846],{},"Prefer diverse scenarios over near-duplicates",[372,848,849],{},"Anonymize before upload",[372,851,852,853,856,857,860],{},"Put stable regression cases in ",[383,854,855],{},"gate_test","; keep a smaller ",[383,858,859],{},"smoke_test"," set for quick checks",[707,862,865,870,944,947],{"icon":863,"label":864},"i-lucide-shield","1.4.4 Data anonymization (mandatory for production data)",[354,866,867],{},[375,868,869],{},"Do not upload raw tenant data to Langfuse.",[413,871,872,882],{},[416,873,874],{},[419,875,876,879],{},[422,877,878],{},"Original",[422,880,881],{},"Anonymized form",[429,883,884,895,903,914,922,933],{},[419,885,886,889],{},[434,887,888],{},"Tenant \u002F org name",[434,890,891,892,393],{},"Synthetic org (e.g. ",[375,893,894],{},"Black Mesa",[419,896,897,900],{},[434,898,899],{},"Real people",[434,901,902],{},"Stable synthetic names",[419,904,905,908],{},[434,906,907],{},"Emails \u002F tenant domains",[434,909,910,913],{},[383,911,912],{},"@blackmesa.example"," placeholders",[419,915,916,919],{},[434,917,918],{},"Real task \u002F product \u002F vendor UUIDs",[434,920,921],{},"Synthetic UUIDs from hashes",[419,923,924,927],{},[434,925,926],{},"Original task id",[434,928,929,932],{},[383,930,931],{},"original_task_id_hash"," only",[419,934,935,938],{},[434,936,937],{},"Images \u002F screenshots",[434,939,940,941,393],{},"Stripped (",[383,942,943],{},"imageStr: null",[354,945,946],{},"Replace tenant strings everywhere, keep people mapping stable across items, keep market product names only when they are not the customer identity, then sanity-check that known customer tokens are gone.",[608,948,952],{"className":949,"code":950,"language":951,"meta":614,"style":614},"language-json shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","{\n  \"anonymization\": {\n    \"tenant_replaced_with\": \"Black Mesa\",\n    \"people_synthetic\": true,\n    \"images_stripped\": true\n  }\n}\n","json",[383,953,954,963,982,1007,1022,1037,1043],{"__ignoreMap":614},[955,956,959],"span",{"class":957,"line":958},"line",1,[955,960,962],{"class":961},"sMK4o","{\n",[955,964,966,969,973,976,979],{"class":957,"line":965},2,[955,967,968],{"class":961},"  \"",[955,970,972],{"class":971},"spNyl","anonymization",[955,974,975],{"class":961},"\"",[955,977,978],{"class":961},":",[955,980,981],{"class":961}," {\n",[955,983,985,988,992,994,996,999,1002,1004],{"class":957,"line":984},3,[955,986,987],{"class":961},"    \"",[955,989,991],{"class":990},"sBMFI","tenant_replaced_with",[955,993,975],{"class":961},[955,995,978],{"class":961},[955,997,998],{"class":961}," \"",[955,1000,894],{"class":1001},"sfazB",[955,1003,975],{"class":961},[955,1005,1006],{"class":961},",\n",[955,1008,1010,1012,1015,1017,1019],{"class":957,"line":1009},4,[955,1011,987],{"class":961},[955,1013,1014],{"class":990},"people_synthetic",[955,1016,975],{"class":961},[955,1018,978],{"class":961},[955,1020,1021],{"class":961}," true,\n",[955,1023,1025,1027,1030,1032,1034],{"class":957,"line":1024},5,[955,1026,987],{"class":961},[955,1028,1029],{"class":990},"images_stripped",[955,1031,975],{"class":961},[955,1033,978],{"class":961},[955,1035,1036],{"class":961}," true\n",[955,1038,1040],{"class":957,"line":1039},6,[955,1041,1042],{"class":961},"  }\n",[955,1044,1046],{"class":957,"line":1045},7,[955,1047,1048],{"class":961},"}\n",[707,1050,1053,1063],{"icon":1051,"label":1052},"i-lucide-file-json","1.4.5 Example: Conversation Namer item",[354,1054,1055,1056,1058,1059,1062],{},"First user message in → short title out. Confirm the real prompt variable name in ",[383,1057,736],{}," before uploading (replace ",[383,1060,1061],{},"message"," if it differs). Synthetic cases need no anonymization.",[608,1064,1066],{"className":949,"code":1065,"language":951,"meta":614,"style":614},"{\n  \"input\": {\n    \"message\": \"Can you compare our CrowdStrike and Wiz renewals for next quarter?\"\n  },\n  \"expectedOutput\": {\n    \"title\": \"CrowdStrike vs Wiz renewals\"\n  },\n  \"metadata\": {\n    \"dataset\": \"conversation-namer\u002Fgate_test\",\n    \"tags\": [\"conversation_namer\", \"synthetic\"],\n    \"constraints\": { \"max_words\": 5 }\n  }\n}\n",[383,1067,1068,1072,1084,1102,1107,1119,1137,1141,1154,1175,1210,1241,1246],{"__ignoreMap":614},[955,1069,1070],{"class":957,"line":958},[955,1071,962],{"class":961},[955,1073,1074,1076,1078,1080,1082],{"class":957,"line":965},[955,1075,968],{"class":961},[955,1077,385],{"class":971},[955,1079,975],{"class":961},[955,1081,978],{"class":961},[955,1083,981],{"class":961},[955,1085,1086,1088,1090,1092,1094,1096,1099],{"class":957,"line":984},[955,1087,987],{"class":961},[955,1089,1061],{"class":990},[955,1091,975],{"class":961},[955,1093,978],{"class":961},[955,1095,998],{"class":961},[955,1097,1098],{"class":1001},"Can you compare our CrowdStrike and Wiz renewals for next quarter?",[955,1100,1101],{"class":961},"\"\n",[955,1103,1104],{"class":957,"line":1009},[955,1105,1106],{"class":961},"  },\n",[955,1108,1109,1111,1113,1115,1117],{"class":957,"line":1024},[955,1110,968],{"class":961},[955,1112,389],{"class":971},[955,1114,975],{"class":961},[955,1116,978],{"class":961},[955,1118,981],{"class":961},[955,1120,1121,1123,1126,1128,1130,1132,1135],{"class":957,"line":1039},[955,1122,987],{"class":961},[955,1124,1125],{"class":990},"title",[955,1127,975],{"class":961},[955,1129,978],{"class":961},[955,1131,998],{"class":961},[955,1133,1134],{"class":1001},"CrowdStrike vs Wiz renewals",[955,1136,1101],{"class":961},[955,1138,1139],{"class":957,"line":1045},[955,1140,1106],{"class":961},[955,1142,1144,1146,1148,1150,1152],{"class":957,"line":1143},8,[955,1145,968],{"class":961},[955,1147,392],{"class":971},[955,1149,975],{"class":961},[955,1151,978],{"class":961},[955,1153,981],{"class":961},[955,1155,1157,1159,1162,1164,1166,1168,1171,1173],{"class":957,"line":1156},9,[955,1158,987],{"class":961},[955,1160,1161],{"class":990},"dataset",[955,1163,975],{"class":961},[955,1165,978],{"class":961},[955,1167,998],{"class":961},[955,1169,1170],{"class":1001},"conversation-namer\u002Fgate_test",[955,1172,975],{"class":961},[955,1174,1006],{"class":961},[955,1176,1178,1180,1183,1185,1187,1190,1192,1195,1197,1200,1202,1205,1207],{"class":957,"line":1177},10,[955,1179,987],{"class":961},[955,1181,1182],{"class":990},"tags",[955,1184,975],{"class":961},[955,1186,978],{"class":961},[955,1188,1189],{"class":961}," [",[955,1191,975],{"class":961},[955,1193,1194],{"class":1001},"conversation_namer",[955,1196,975],{"class":961},[955,1198,1199],{"class":961},",",[955,1201,998],{"class":961},[955,1203,1204],{"class":1001},"synthetic",[955,1206,975],{"class":961},[955,1208,1209],{"class":961},"],\n",[955,1211,1213,1215,1218,1220,1222,1225,1227,1231,1233,1235,1238],{"class":957,"line":1212},11,[955,1214,987],{"class":961},[955,1216,1217],{"class":990},"constraints",[955,1219,975],{"class":961},[955,1221,978],{"class":961},[955,1223,1224],{"class":961}," {",[955,1226,998],{"class":961},[955,1228,1230],{"class":1229},"sbssI","max_words",[955,1232,975],{"class":961},[955,1234,978],{"class":961},[955,1236,1237],{"class":1229}," 5",[955,1239,1240],{"class":961}," }\n",[955,1242,1244],{"class":957,"line":1243},12,[955,1245,1042],{"class":961},[955,1247,1249],{"class":957,"line":1248},13,[955,1250,1048],{"class":961},[707,1252,1254,1257,1276,1279,1286,1744,1751,2078,2085],{"icon":1051,"label":1253},"1.4.6 Example: Findings Agent shapes",[354,1255,1256],{},"Turns interview transcript + directive into summary, report, and structured findings.",[354,1258,1259,1262,1263,1265,1266,1269,1270,1272,1273,1275],{},[375,1260,1261],{},"Process:"," pull production cases → map prompt variables to ",[383,1264,385],{}," → anonymize → put ",[375,1267,1268],{},"judge guidance"," in ",[383,1271,389],{}," → put historical summary\u002Freport\u002Ffindings in ",[383,1274,764],{}," → upload via SDK.",[493,1277,1278],{},"Internal pull\u002Fbuild scripts live outside this handbook. Use the shapes below as the contract; ask Data Ops \u002F LLM Ops for the current scripts if you need a production refresh.",[354,1280,1281],{},[375,1282,1283,1285],{},[383,1284,385],{}," (prompt variables only)",[608,1287,1289],{"className":949,"code":1288,"language":951,"meta":614,"style":614},"{\n  \"user_name\": \"Marcus Silva\",\n  \"user_role\": \"Team Lead and L3 Engineer\",\n  \"organisation_context\": \"Black Mesa is a diversified technology and research organization...\",\n  \"source_type\": \"PROD\",\n  \"source_info\": {\n    \"id\": \"02e4e876-4c33-b27b-9c8e-82cd402f0f85\",\n    \"name\": \"Akamai App & API Protector\",\n    \"type\": \"PROD\",\n    \"vendor\": { \"id\": \"...\", \"name\": \"Akamai\" }\n  },\n  \"task_directive\": \"You are conducting a structured interview to gather information about a product...\",\n  \"interview_transcript\": [\n    {\n      \"id\": \"rufsaa\",\n      \"role\": \"assistant\",\n      \"text\": \"Hi Marcus! I'm ESPi...\",\n      \"isThoughts\": false,\n      \"imageStr\": null\n    },\n    {\n      \"id\": \"oywj8t\",\n      \"role\": \"user\",\n      \"text\": \"Yes\",\n      \"isThoughts\": false,\n      \"imageStr\": null\n    }\n  ]\n}\n",[383,1290,1291,1295,1315,1335,1355,1375,1388,1408,1428,1447,1494,1498,1518,1532,1538,1559,1580,1600,1615,1630,1636,1641,1661,1681,1701,1714,1727,1733,1739],{"__ignoreMap":614},[955,1292,1293],{"class":957,"line":958},[955,1294,962],{"class":961},[955,1296,1297,1299,1302,1304,1306,1308,1311,1313],{"class":957,"line":965},[955,1298,968],{"class":961},[955,1300,1301],{"class":971},"user_name",[955,1303,975],{"class":961},[955,1305,978],{"class":961},[955,1307,998],{"class":961},[955,1309,1310],{"class":1001},"Marcus Silva",[955,1312,975],{"class":961},[955,1314,1006],{"class":961},[955,1316,1317,1319,1322,1324,1326,1328,1331,1333],{"class":957,"line":984},[955,1318,968],{"class":961},[955,1320,1321],{"class":971},"user_role",[955,1323,975],{"class":961},[955,1325,978],{"class":961},[955,1327,998],{"class":961},[955,1329,1330],{"class":1001},"Team Lead and L3 Engineer",[955,1332,975],{"class":961},[955,1334,1006],{"class":961},[955,1336,1337,1339,1342,1344,1346,1348,1351,1353],{"class":957,"line":1009},[955,1338,968],{"class":961},[955,1340,1341],{"class":971},"organisation_context",[955,1343,975],{"class":961},[955,1345,978],{"class":961},[955,1347,998],{"class":961},[955,1349,1350],{"class":1001},"Black Mesa is a diversified technology and research organization...",[955,1352,975],{"class":961},[955,1354,1006],{"class":961},[955,1356,1357,1359,1362,1364,1366,1368,1371,1373],{"class":957,"line":1024},[955,1358,968],{"class":961},[955,1360,1361],{"class":971},"source_type",[955,1363,975],{"class":961},[955,1365,978],{"class":961},[955,1367,998],{"class":961},[955,1369,1370],{"class":1001},"PROD",[955,1372,975],{"class":961},[955,1374,1006],{"class":961},[955,1376,1377,1379,1382,1384,1386],{"class":957,"line":1039},[955,1378,968],{"class":961},[955,1380,1381],{"class":971},"source_info",[955,1383,975],{"class":961},[955,1385,978],{"class":961},[955,1387,981],{"class":961},[955,1389,1390,1392,1395,1397,1399,1401,1404,1406],{"class":957,"line":1045},[955,1391,987],{"class":961},[955,1393,1394],{"class":990},"id",[955,1396,975],{"class":961},[955,1398,978],{"class":961},[955,1400,998],{"class":961},[955,1402,1403],{"class":1001},"02e4e876-4c33-b27b-9c8e-82cd402f0f85",[955,1405,975],{"class":961},[955,1407,1006],{"class":961},[955,1409,1410,1412,1415,1417,1419,1421,1424,1426],{"class":957,"line":1143},[955,1411,987],{"class":961},[955,1413,1414],{"class":990},"name",[955,1416,975],{"class":961},[955,1418,978],{"class":961},[955,1420,998],{"class":961},[955,1422,1423],{"class":1001},"Akamai App & API Protector",[955,1425,975],{"class":961},[955,1427,1006],{"class":961},[955,1429,1430,1432,1435,1437,1439,1441,1443,1445],{"class":957,"line":1156},[955,1431,987],{"class":961},[955,1433,1434],{"class":990},"type",[955,1436,975],{"class":961},[955,1438,978],{"class":961},[955,1440,998],{"class":961},[955,1442,1370],{"class":1001},[955,1444,975],{"class":961},[955,1446,1006],{"class":961},[955,1448,1449,1451,1454,1456,1458,1460,1462,1464,1466,1468,1470,1473,1475,1477,1479,1481,1483,1485,1487,1490,1492],{"class":957,"line":1177},[955,1450,987],{"class":961},[955,1452,1453],{"class":990},"vendor",[955,1455,975],{"class":961},[955,1457,978],{"class":961},[955,1459,1224],{"class":961},[955,1461,998],{"class":961},[955,1463,1394],{"class":1229},[955,1465,975],{"class":961},[955,1467,978],{"class":961},[955,1469,998],{"class":961},[955,1471,1472],{"class":1001},"...",[955,1474,975],{"class":961},[955,1476,1199],{"class":961},[955,1478,998],{"class":961},[955,1480,1414],{"class":1229},[955,1482,975],{"class":961},[955,1484,978],{"class":961},[955,1486,998],{"class":961},[955,1488,1489],{"class":1001},"Akamai",[955,1491,975],{"class":961},[955,1493,1240],{"class":961},[955,1495,1496],{"class":957,"line":1212},[955,1497,1106],{"class":961},[955,1499,1500,1502,1505,1507,1509,1511,1514,1516],{"class":957,"line":1243},[955,1501,968],{"class":961},[955,1503,1504],{"class":971},"task_directive",[955,1506,975],{"class":961},[955,1508,978],{"class":961},[955,1510,998],{"class":961},[955,1512,1513],{"class":1001},"You are conducting a structured interview to gather information about a product...",[955,1515,975],{"class":961},[955,1517,1006],{"class":961},[955,1519,1520,1522,1525,1527,1529],{"class":957,"line":1248},[955,1521,968],{"class":961},[955,1523,1524],{"class":971},"interview_transcript",[955,1526,975],{"class":961},[955,1528,978],{"class":961},[955,1530,1531],{"class":961}," [\n",[955,1533,1535],{"class":957,"line":1534},14,[955,1536,1537],{"class":961},"    {\n",[955,1539,1541,1544,1546,1548,1550,1552,1555,1557],{"class":957,"line":1540},15,[955,1542,1543],{"class":961},"      \"",[955,1545,1394],{"class":990},[955,1547,975],{"class":961},[955,1549,978],{"class":961},[955,1551,998],{"class":961},[955,1553,1554],{"class":1001},"rufsaa",[955,1556,975],{"class":961},[955,1558,1006],{"class":961},[955,1560,1562,1564,1567,1569,1571,1573,1576,1578],{"class":957,"line":1561},16,[955,1563,1543],{"class":961},[955,1565,1566],{"class":990},"role",[955,1568,975],{"class":961},[955,1570,978],{"class":961},[955,1572,998],{"class":961},[955,1574,1575],{"class":1001},"assistant",[955,1577,975],{"class":961},[955,1579,1006],{"class":961},[955,1581,1583,1585,1587,1589,1591,1593,1596,1598],{"class":957,"line":1582},17,[955,1584,1543],{"class":961},[955,1586,613],{"class":990},[955,1588,975],{"class":961},[955,1590,978],{"class":961},[955,1592,998],{"class":961},[955,1594,1595],{"class":1001},"Hi Marcus! I'm ESPi...",[955,1597,975],{"class":961},[955,1599,1006],{"class":961},[955,1601,1603,1605,1608,1610,1612],{"class":957,"line":1602},18,[955,1604,1543],{"class":961},[955,1606,1607],{"class":990},"isThoughts",[955,1609,975],{"class":961},[955,1611,978],{"class":961},[955,1613,1614],{"class":961}," false,\n",[955,1616,1618,1620,1623,1625,1627],{"class":957,"line":1617},19,[955,1619,1543],{"class":961},[955,1621,1622],{"class":990},"imageStr",[955,1624,975],{"class":961},[955,1626,978],{"class":961},[955,1628,1629],{"class":961}," null\n",[955,1631,1633],{"class":957,"line":1632},20,[955,1634,1635],{"class":961},"    },\n",[955,1637,1639],{"class":957,"line":1638},21,[955,1640,1537],{"class":961},[955,1642,1644,1646,1648,1650,1652,1654,1657,1659],{"class":957,"line":1643},22,[955,1645,1543],{"class":961},[955,1647,1394],{"class":990},[955,1649,975],{"class":961},[955,1651,978],{"class":961},[955,1653,998],{"class":961},[955,1655,1656],{"class":1001},"oywj8t",[955,1658,975],{"class":961},[955,1660,1006],{"class":961},[955,1662,1664,1666,1668,1670,1672,1674,1677,1679],{"class":957,"line":1663},23,[955,1665,1543],{"class":961},[955,1667,1566],{"class":990},[955,1669,975],{"class":961},[955,1671,978],{"class":961},[955,1673,998],{"class":961},[955,1675,1676],{"class":1001},"user",[955,1678,975],{"class":961},[955,1680,1006],{"class":961},[955,1682,1684,1686,1688,1690,1692,1694,1697,1699],{"class":957,"line":1683},24,[955,1685,1543],{"class":961},[955,1687,613],{"class":990},[955,1689,975],{"class":961},[955,1691,978],{"class":961},[955,1693,998],{"class":961},[955,1695,1696],{"class":1001},"Yes",[955,1698,975],{"class":961},[955,1700,1006],{"class":961},[955,1702,1704,1706,1708,1710,1712],{"class":957,"line":1703},25,[955,1705,1543],{"class":961},[955,1707,1607],{"class":990},[955,1709,975],{"class":961},[955,1711,978],{"class":961},[955,1713,1614],{"class":961},[955,1715,1717,1719,1721,1723,1725],{"class":957,"line":1716},26,[955,1718,1543],{"class":961},[955,1720,1622],{"class":990},[955,1722,975],{"class":961},[955,1724,978],{"class":961},[955,1726,1629],{"class":961},[955,1728,1730],{"class":957,"line":1729},27,[955,1731,1732],{"class":961},"    }\n",[955,1734,1736],{"class":957,"line":1735},28,[955,1737,1738],{"class":961},"  ]\n",[955,1740,1742],{"class":957,"line":1741},29,[955,1743,1048],{"class":961},[354,1745,1746],{},[375,1747,1748,1750],{},[383,1749,389],{}," (guidance)",[608,1752,1754],{"className":949,"code":1753,"language":951,"meta":614,"style":614},"{\n  \"expected_summary_focus\": \"Summarize strategic state, material risks\u002Finsights, and decision-relevant takeaways. Surface only transcript-grounded points.\",\n  \"expected_min_findings\": 2,\n  \"expected_max_findings\": 11,\n  \"expected_required_finding_types\": [\"insight\", \"risk\", \"sentiment\"],\n  \"expected_report_sections\": [\n    \"Overview\",\n    \"Strategic State\",\n    \"Capabilities and Usage\",\n    \"Coverage and Controls\",\n    \"Sentiment and Operational Experience\",\n    \"Risks, Gaps and Opportunities\",\n    \"Stakeholders\",\n    \"Conclusion\"\n  ],\n  \"must_align_to_directive\": true,\n  \"must_be_grounded_in_transcript\": true,\n  \"schema_notes\": {\n    \"RISK_requires\": [\"severity\"],\n    \"INSIGHT_requires\": [\"category\"],\n    \"SENTIMENT_requires\": [\"category\", \"rating\"]\n  }\n}\n",[383,1755,1756,1760,1780,1796,1812,1852,1865,1876,1887,1898,1909,1920,1931,1942,1951,1956,1969,1982,1995,2017,2039,2070,2074],{"__ignoreMap":614},[955,1757,1758],{"class":957,"line":958},[955,1759,962],{"class":961},[955,1761,1762,1764,1767,1769,1771,1773,1776,1778],{"class":957,"line":965},[955,1763,968],{"class":961},[955,1765,1766],{"class":971},"expected_summary_focus",[955,1768,975],{"class":961},[955,1770,978],{"class":961},[955,1772,998],{"class":961},[955,1774,1775],{"class":1001},"Summarize strategic state, material risks\u002Finsights, and decision-relevant takeaways. Surface only transcript-grounded points.",[955,1777,975],{"class":961},[955,1779,1006],{"class":961},[955,1781,1782,1784,1787,1789,1791,1794],{"class":957,"line":984},[955,1783,968],{"class":961},[955,1785,1786],{"class":971},"expected_min_findings",[955,1788,975],{"class":961},[955,1790,978],{"class":961},[955,1792,1793],{"class":1229}," 2",[955,1795,1006],{"class":961},[955,1797,1798,1800,1803,1805,1807,1810],{"class":957,"line":1009},[955,1799,968],{"class":961},[955,1801,1802],{"class":971},"expected_max_findings",[955,1804,975],{"class":961},[955,1806,978],{"class":961},[955,1808,1809],{"class":1229}," 11",[955,1811,1006],{"class":961},[955,1813,1814,1816,1819,1821,1823,1825,1827,1830,1832,1834,1836,1839,1841,1843,1845,1848,1850],{"class":957,"line":1024},[955,1815,968],{"class":961},[955,1817,1818],{"class":971},"expected_required_finding_types",[955,1820,975],{"class":961},[955,1822,978],{"class":961},[955,1824,1189],{"class":961},[955,1826,975],{"class":961},[955,1828,1829],{"class":1001},"insight",[955,1831,975],{"class":961},[955,1833,1199],{"class":961},[955,1835,998],{"class":961},[955,1837,1838],{"class":1001},"risk",[955,1840,975],{"class":961},[955,1842,1199],{"class":961},[955,1844,998],{"class":961},[955,1846,1847],{"class":1001},"sentiment",[955,1849,975],{"class":961},[955,1851,1209],{"class":961},[955,1853,1854,1856,1859,1861,1863],{"class":957,"line":1039},[955,1855,968],{"class":961},[955,1857,1858],{"class":971},"expected_report_sections",[955,1860,975],{"class":961},[955,1862,978],{"class":961},[955,1864,1531],{"class":961},[955,1866,1867,1869,1872,1874],{"class":957,"line":1045},[955,1868,987],{"class":961},[955,1870,1871],{"class":1001},"Overview",[955,1873,975],{"class":961},[955,1875,1006],{"class":961},[955,1877,1878,1880,1883,1885],{"class":957,"line":1143},[955,1879,987],{"class":961},[955,1881,1882],{"class":1001},"Strategic State",[955,1884,975],{"class":961},[955,1886,1006],{"class":961},[955,1888,1889,1891,1894,1896],{"class":957,"line":1156},[955,1890,987],{"class":961},[955,1892,1893],{"class":1001},"Capabilities and Usage",[955,1895,975],{"class":961},[955,1897,1006],{"class":961},[955,1899,1900,1902,1905,1907],{"class":957,"line":1177},[955,1901,987],{"class":961},[955,1903,1904],{"class":1001},"Coverage and Controls",[955,1906,975],{"class":961},[955,1908,1006],{"class":961},[955,1910,1911,1913,1916,1918],{"class":957,"line":1212},[955,1912,987],{"class":961},[955,1914,1915],{"class":1001},"Sentiment and Operational Experience",[955,1917,975],{"class":961},[955,1919,1006],{"class":961},[955,1921,1922,1924,1927,1929],{"class":957,"line":1243},[955,1923,987],{"class":961},[955,1925,1926],{"class":1001},"Risks, Gaps and Opportunities",[955,1928,975],{"class":961},[955,1930,1006],{"class":961},[955,1932,1933,1935,1938,1940],{"class":957,"line":1248},[955,1934,987],{"class":961},[955,1936,1937],{"class":1001},"Stakeholders",[955,1939,975],{"class":961},[955,1941,1006],{"class":961},[955,1943,1944,1946,1949],{"class":957,"line":1534},[955,1945,987],{"class":961},[955,1947,1948],{"class":1001},"Conclusion",[955,1950,1101],{"class":961},[955,1952,1953],{"class":957,"line":1540},[955,1954,1955],{"class":961},"  ],\n",[955,1957,1958,1960,1963,1965,1967],{"class":957,"line":1561},[955,1959,968],{"class":961},[955,1961,1962],{"class":971},"must_align_to_directive",[955,1964,975],{"class":961},[955,1966,978],{"class":961},[955,1968,1021],{"class":961},[955,1970,1971,1973,1976,1978,1980],{"class":957,"line":1582},[955,1972,968],{"class":961},[955,1974,1975],{"class":971},"must_be_grounded_in_transcript",[955,1977,975],{"class":961},[955,1979,978],{"class":961},[955,1981,1021],{"class":961},[955,1983,1984,1986,1989,1991,1993],{"class":957,"line":1602},[955,1985,968],{"class":961},[955,1987,1988],{"class":971},"schema_notes",[955,1990,975],{"class":961},[955,1992,978],{"class":961},[955,1994,981],{"class":961},[955,1996,1997,1999,2002,2004,2006,2008,2010,2013,2015],{"class":957,"line":1617},[955,1998,987],{"class":961},[955,2000,2001],{"class":990},"RISK_requires",[955,2003,975],{"class":961},[955,2005,978],{"class":961},[955,2007,1189],{"class":961},[955,2009,975],{"class":961},[955,2011,2012],{"class":1001},"severity",[955,2014,975],{"class":961},[955,2016,1209],{"class":961},[955,2018,2019,2021,2024,2026,2028,2030,2032,2035,2037],{"class":957,"line":1632},[955,2020,987],{"class":961},[955,2022,2023],{"class":990},"INSIGHT_requires",[955,2025,975],{"class":961},[955,2027,978],{"class":961},[955,2029,1189],{"class":961},[955,2031,975],{"class":961},[955,2033,2034],{"class":1001},"category",[955,2036,975],{"class":961},[955,2038,1209],{"class":961},[955,2040,2041,2043,2046,2048,2050,2052,2054,2056,2058,2060,2062,2065,2067],{"class":957,"line":1638},[955,2042,987],{"class":961},[955,2044,2045],{"class":990},"SENTIMENT_requires",[955,2047,975],{"class":961},[955,2049,978],{"class":961},[955,2051,1189],{"class":961},[955,2053,975],{"class":961},[955,2055,2034],{"class":1001},[955,2057,975],{"class":961},[955,2059,1199],{"class":961},[955,2061,998],{"class":961},[955,2063,2064],{"class":1001},"rating",[955,2066,975],{"class":961},[955,2068,2069],{"class":961},"]\n",[955,2071,2072],{"class":957,"line":1643},[955,2073,1042],{"class":961},[955,2075,2076],{"class":957,"line":1663},[955,2077,1048],{"class":961},[354,2079,2080],{},[375,2081,2082,2084],{},[383,2083,392],{}," (example)",[608,2086,2088],{"className":949,"code":2087,"language":951,"meta":614,"style":614},"{\n  \"dataset\": \"findings\u002Fgate_test\",\n  \"tags\": [\"findings_agent\", \"black_mesa\", \"anonymized\"],\n  \"original_task_id_hash\": \"4ec6af6be8582669\",\n  \"product_name\": \"Akamai App & API Protector\",\n  \"expected_tools\": [\"platformSearch\"],\n  \"reference_output\": {\n    \"summary\": \"...\",\n    \"report\": \"...\",\n    \"findings\": []\n  },\n  \"anonymization\": {\n    \"tenant_replaced_with\": \"Black Mesa\",\n    \"people_synthetic\": true,\n    \"images_stripped\": true\n  }\n}\n",[383,2089,2090,2094,2112,2151,2170,2189,2211,2224,2243,2262,2276,2280,2292,2310,2322,2334,2338],{"__ignoreMap":614},[955,2091,2092],{"class":957,"line":958},[955,2093,962],{"class":961},[955,2095,2096,2098,2100,2102,2104,2106,2108,2110],{"class":957,"line":965},[955,2097,968],{"class":961},[955,2099,1161],{"class":971},[955,2101,975],{"class":961},[955,2103,978],{"class":961},[955,2105,998],{"class":961},[955,2107,491],{"class":1001},[955,2109,975],{"class":961},[955,2111,1006],{"class":961},[955,2113,2114,2116,2118,2120,2122,2124,2126,2129,2131,2133,2135,2138,2140,2142,2144,2147,2149],{"class":957,"line":984},[955,2115,968],{"class":961},[955,2117,1182],{"class":971},[955,2119,975],{"class":961},[955,2121,978],{"class":961},[955,2123,1189],{"class":961},[955,2125,975],{"class":961},[955,2127,2128],{"class":1001},"findings_agent",[955,2130,975],{"class":961},[955,2132,1199],{"class":961},[955,2134,998],{"class":961},[955,2136,2137],{"class":1001},"black_mesa",[955,2139,975],{"class":961},[955,2141,1199],{"class":961},[955,2143,998],{"class":961},[955,2145,2146],{"class":1001},"anonymized",[955,2148,975],{"class":961},[955,2150,1209],{"class":961},[955,2152,2153,2155,2157,2159,2161,2163,2166,2168],{"class":957,"line":1009},[955,2154,968],{"class":961},[955,2156,931],{"class":971},[955,2158,975],{"class":961},[955,2160,978],{"class":961},[955,2162,998],{"class":961},[955,2164,2165],{"class":1001},"4ec6af6be8582669",[955,2167,975],{"class":961},[955,2169,1006],{"class":961},[955,2171,2172,2174,2177,2179,2181,2183,2185,2187],{"class":957,"line":1024},[955,2173,968],{"class":961},[955,2175,2176],{"class":971},"product_name",[955,2178,975],{"class":961},[955,2180,978],{"class":961},[955,2182,998],{"class":961},[955,2184,1423],{"class":1001},[955,2186,975],{"class":961},[955,2188,1006],{"class":961},[955,2190,2191,2193,2196,2198,2200,2202,2204,2207,2209],{"class":957,"line":1039},[955,2192,968],{"class":961},[955,2194,2195],{"class":971},"expected_tools",[955,2197,975],{"class":961},[955,2199,978],{"class":961},[955,2201,1189],{"class":961},[955,2203,975],{"class":961},[955,2205,2206],{"class":1001},"platformSearch",[955,2208,975],{"class":961},[955,2210,1209],{"class":961},[955,2212,2213,2215,2218,2220,2222],{"class":957,"line":1045},[955,2214,968],{"class":961},[955,2216,2217],{"class":971},"reference_output",[955,2219,975],{"class":961},[955,2221,978],{"class":961},[955,2223,981],{"class":961},[955,2225,2226,2228,2231,2233,2235,2237,2239,2241],{"class":957,"line":1143},[955,2227,987],{"class":961},[955,2229,2230],{"class":990},"summary",[955,2232,975],{"class":961},[955,2234,978],{"class":961},[955,2236,998],{"class":961},[955,2238,1472],{"class":1001},[955,2240,975],{"class":961},[955,2242,1006],{"class":961},[955,2244,2245,2247,2250,2252,2254,2256,2258,2260],{"class":957,"line":1156},[955,2246,987],{"class":961},[955,2248,2249],{"class":990},"report",[955,2251,975],{"class":961},[955,2253,978],{"class":961},[955,2255,998],{"class":961},[955,2257,1472],{"class":1001},[955,2259,975],{"class":961},[955,2261,1006],{"class":961},[955,2263,2264,2266,2269,2271,2273],{"class":957,"line":1177},[955,2265,987],{"class":961},[955,2267,2268],{"class":990},"findings",[955,2270,975],{"class":961},[955,2272,978],{"class":961},[955,2274,2275],{"class":961}," []\n",[955,2277,2278],{"class":957,"line":1212},[955,2279,1106],{"class":961},[955,2281,2282,2284,2286,2288,2290],{"class":957,"line":1243},[955,2283,968],{"class":961},[955,2285,972],{"class":971},[955,2287,975],{"class":961},[955,2289,978],{"class":961},[955,2291,981],{"class":961},[955,2293,2294,2296,2298,2300,2302,2304,2306,2308],{"class":957,"line":1248},[955,2295,987],{"class":961},[955,2297,991],{"class":990},[955,2299,975],{"class":961},[955,2301,978],{"class":961},[955,2303,998],{"class":961},[955,2305,894],{"class":1001},[955,2307,975],{"class":961},[955,2309,1006],{"class":961},[955,2311,2312,2314,2316,2318,2320],{"class":957,"line":1534},[955,2313,987],{"class":961},[955,2315,1014],{"class":990},[955,2317,975],{"class":961},[955,2319,978],{"class":961},[955,2321,1021],{"class":961},[955,2323,2324,2326,2328,2330,2332],{"class":957,"line":1540},[955,2325,987],{"class":961},[955,2327,1029],{"class":990},[955,2329,975],{"class":961},[955,2331,978],{"class":961},[955,2333,1036],{"class":961},[955,2335,2336],{"class":957,"line":1561},[955,2337,1042],{"class":961},[955,2339,2340],{"class":957,"line":1582},[955,2341,1048],{"class":961},[707,2343,2346,2392,2549],{"icon":2344,"label":2345},"i-lucide-code","1.4.7 Python SDK upload (nested JSON)",[608,2347,2351],{"className":2348,"code":2349,"language":2350,"meta":614,"style":614},"language-bash shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","export LANGFUSE_PUBLIC_KEY=pk-lf-...\nexport LANGFUSE_SECRET_KEY=sk-lf-...\nexport LANGFUSE_HOST=https:\u002F\u002Fcloud.langfuse.com\n","bash",[383,2352,2353,2368,2380],{"__ignoreMap":614},[955,2354,2355,2358,2362,2365],{"class":957,"line":958},[955,2356,2357],{"class":971},"export",[955,2359,2361],{"class":2360},"sTEyZ"," LANGFUSE_PUBLIC_KEY",[955,2363,2364],{"class":961},"=",[955,2366,2367],{"class":2360},"pk-lf-...\n",[955,2369,2370,2372,2375,2377],{"class":957,"line":965},[955,2371,2357],{"class":971},[955,2373,2374],{"class":2360}," LANGFUSE_SECRET_KEY",[955,2376,2364],{"class":961},[955,2378,2379],{"class":2360},"sk-lf-...\n",[955,2381,2382,2384,2387,2389],{"class":957,"line":984},[955,2383,2357],{"class":971},[955,2385,2386],{"class":2360}," LANGFUSE_HOST",[955,2388,2364],{"class":961},[955,2390,2391],{"class":2360},"https:\u002F\u002Fcloud.langfuse.com\n",[608,2393,2397],{"className":2394,"code":2395,"language":2396,"meta":614,"style":614},"language-python shiki shiki-themes material-theme-lighter material-theme material-theme-palenight","from langfuse import get_client\n\nlangfuse = get_client()\n\ndataset_name = \"conversation-namer\u002Fsmoke_test\"\n\nlangfuse.create_dataset(\n    name=dataset_name,\n    description=\"Smoke test set for Conversation Namer\",\n)\n\nitems = [\n    {\n        \"input\": {\n            \"message\": \"Can you compare our CrowdStrike and Wiz renewals for next quarter?\"\n        },\n        \"expected_output\": {\"title\": \"CrowdStrike vs Wiz renewals\"},\n        \"metadata\": {\n            \"dataset\": dataset_name,\n            \"tags\": [\"conversation_namer\", \"synthetic\"],\n        },\n    },\n]\n\nfor item in items:\n    langfuse.create_dataset_item(\n        dataset_name=dataset_name,\n        input=item[\"input\"],\n        expected_output=item[\"expected_output\"],\n        metadata=item.get(\"metadata\"),\n    )\n","python",[383,2398,2399,2404,2410,2415,2419,2424,2428,2433,2438,2443,2448,2452,2457,2461,2466,2471,2476,2481,2486,2491,2496,2500,2504,2508,2512,2517,2522,2527,2532,2537,2543],{"__ignoreMap":614},[955,2400,2401],{"class":957,"line":958},[955,2402,2403],{},"from langfuse import get_client\n",[955,2405,2406],{"class":957,"line":965},[955,2407,2409],{"emptyLinePlaceholder":2408},true,"\n",[955,2411,2412],{"class":957,"line":984},[955,2413,2414],{},"langfuse = get_client()\n",[955,2416,2417],{"class":957,"line":1009},[955,2418,2409],{"emptyLinePlaceholder":2408},[955,2420,2421],{"class":957,"line":1024},[955,2422,2423],{},"dataset_name = \"conversation-namer\u002Fsmoke_test\"\n",[955,2425,2426],{"class":957,"line":1039},[955,2427,2409],{"emptyLinePlaceholder":2408},[955,2429,2430],{"class":957,"line":1045},[955,2431,2432],{},"langfuse.create_dataset(\n",[955,2434,2435],{"class":957,"line":1143},[955,2436,2437],{},"    name=dataset_name,\n",[955,2439,2440],{"class":957,"line":1156},[955,2441,2442],{},"    description=\"Smoke test set for Conversation Namer\",\n",[955,2444,2445],{"class":957,"line":1177},[955,2446,2447],{},")\n",[955,2449,2450],{"class":957,"line":1212},[955,2451,2409],{"emptyLinePlaceholder":2408},[955,2453,2454],{"class":957,"line":1243},[955,2455,2456],{},"items = [\n",[955,2458,2459],{"class":957,"line":1248},[955,2460,1537],{},[955,2462,2463],{"class":957,"line":1534},[955,2464,2465],{},"        \"input\": {\n",[955,2467,2468],{"class":957,"line":1540},[955,2469,2470],{},"            \"message\": \"Can you compare our CrowdStrike and Wiz renewals for next quarter?\"\n",[955,2472,2473],{"class":957,"line":1561},[955,2474,2475],{},"        },\n",[955,2477,2478],{"class":957,"line":1582},[955,2479,2480],{},"        \"expected_output\": {\"title\": \"CrowdStrike vs Wiz renewals\"},\n",[955,2482,2483],{"class":957,"line":1602},[955,2484,2485],{},"        \"metadata\": {\n",[955,2487,2488],{"class":957,"line":1617},[955,2489,2490],{},"            \"dataset\": dataset_name,\n",[955,2492,2493],{"class":957,"line":1632},[955,2494,2495],{},"            \"tags\": [\"conversation_namer\", \"synthetic\"],\n",[955,2497,2498],{"class":957,"line":1638},[955,2499,2475],{},[955,2501,2502],{"class":957,"line":1643},[955,2503,1635],{},[955,2505,2506],{"class":957,"line":1663},[955,2507,2069],{},[955,2509,2510],{"class":957,"line":1683},[955,2511,2409],{"emptyLinePlaceholder":2408},[955,2513,2514],{"class":957,"line":1703},[955,2515,2516],{},"for item in items:\n",[955,2518,2519],{"class":957,"line":1716},[955,2520,2521],{},"    langfuse.create_dataset_item(\n",[955,2523,2524],{"class":957,"line":1729},[955,2525,2526],{},"        dataset_name=dataset_name,\n",[955,2528,2529],{"class":957,"line":1735},[955,2530,2531],{},"        input=item[\"input\"],\n",[955,2533,2534],{"class":957,"line":1741},[955,2535,2536],{},"        expected_output=item[\"expected_output\"],\n",[955,2538,2540],{"class":957,"line":2539},30,[955,2541,2542],{},"        metadata=item.get(\"metadata\"),\n",[955,2544,2546],{"class":957,"line":2545},31,[955,2547,2548],{},"    )\n",[354,2550,2551],{},"Same pattern for Findings — pass the full nested objects; no need to flatten.",[707,2553,2556],{"icon":2554,"label":2555},"i-lucide-square-check","1.4.8 Checklist before upload",[466,2557,2560,2568,2576,2587,2598,2604,2615],{"className":2558},[2559],"contains-task-list",[372,2561,2564,2567],{"className":2562},[2563],"task-list-item",[385,2565],{"disabled":2408,"type":2566},"checkbox"," Agent under test is clear",[372,2569,2571,693,2573,2575],{"className":2570},[2563],[385,2572],{"disabled":2408,"type":2566},[383,2574,385],{}," keys match prompt variables 1:1",[372,2577,2579,693,2581,2583,2584,2586],{"className":2578},[2563],[385,2580],{"disabled":2408,"type":2566},[383,2582,389],{}," is a golden answer ",[375,2585,541],{}," judge guidance (not mixed without intent)",[372,2588,2590,2592,2593,2595,2596],{"className":2589},[2563],[385,2591],{"disabled":2408,"type":2566}," For open-ended agents: guidance in ",[383,2594,389],{},"; full goldens (if kept) in ",[383,2597,764],{},[372,2599,2601,2603],{"className":2600},[2563],[385,2602],{"disabled":2408,"type":2566}," Production data is anonymized",[372,2605,2607,2609,2610,2612,2613],{"className":2606},[2563],[385,2608],{"disabled":2408,"type":2566}," Name is ",[383,2611,638],{}," or ",[383,2614,651],{},[372,2616,2618,2620],{"className":2617},[2563],[385,2619],{"disabled":2408,"type":2566}," Spot-check 2–3 items in the Langfuse UI",[497,2622],{},[500,2624,2626],{"id":2625},"_2-creating-evaluators","2. Creating Evaluators",[354,2628,2629],{},"Evaluators are the scoring definitions you attach when you run prompt experiments.",[354,2631,2632,2633,2636],{},"Today we ",[375,2634,2635],{},"mostly set up LLM-as-judge evaluators"," in Langfuse.",[354,2638,2639,2642,2643,2646,2647,1269,2650,2652,2653,2658],{},[375,2640,2641],{},"Code-based checks"," still matter for deterministic rules, but we typically implement them in the ",[375,2644,2645],{},"application"," and\u002For ",[375,2648,2649],{},"unit\u002Fintegration tests",[383,2651,736],{},", rather than as the primary Langfuse experiment evaluators. Langfuse also supports ",[358,2654,2657],{"href":2655,"rel":2656},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fevaluation-methods\u002Fcode-evaluators",[362],"Code Evaluators"," if needed later.",[354,2660,2661],{},"Rule of thumb:",[466,2663,2664,2670],{},[372,2665,2666,2667,2669],{},"Needs reading comprehension \u002F judgment → ",[375,2668,817],{}," (Langfuse)",[372,2671,2672,2673,2675],{},"A junior engineer could assert it in a test → ",[375,2674,383],{}," (app or unit tests)",[553,2677,2679],{"id":2678},"_21-create-your-first-llm-as-judge-5-steps","2.1 Create your first LLM-as-judge (5 steps)",[369,2681,2682,2688,2694,2700,2717],{},[372,2683,2684,2687],{},[375,2685,2686],{},"Pick one dimension"," — e.g. “grounded in transcript” or “aligns with directive”.",[372,2689,2690,2693],{},[375,2691,2692],{},"Choose score shape"," — boolean \u002F categorical \u002F numeric.",[372,2695,2696,2699],{},[375,2697,2698],{},"Write the judge prompt"," — explicit pass\u002Ffail or category rules; one job only.",[372,2701,2702,2705,2706,386,2709,2712,2713,2716],{},[375,2703,2704],{},"Create it in Langfuse"," — map ",[383,2707,2708],{},"{{input}}",[383,2710,2711],{},"{{output}}"," (and ",[383,2714,2715],{},"{{expected_output}}"," only if needed).",[372,2718,2719,2722],{},[375,2720,2721],{},"Verify mapping"," — use Prompt Preview; spot-check that variables populate as expected.",[354,2724,2725,2726,2729],{},"Start with ",[375,2727,2728],{},"2–3 focused judges"," per agent. Add more only when debugging a specific failure class.",[354,2731,2732],{},"Findings starter pack:",[369,2734,2735,2741],{},[372,2736,2737,2740],{},[383,2738,2739],{},"findings_agent.grounding"," — Boolean",[372,2742,2743,2746,2747,386,2750,386,2753,393],{},[383,2744,2745],{},"findings_agent.directive_alignment"," — Categorical (",[383,2748,2749],{},"fail",[383,2751,2752],{},"partial",[383,2754,2755],{},"pass",[553,2757,2759],{"id":2758},"_22-naming","2.2 Naming",[608,2761,2764],{"className":2762,"code":2763,"language":613,"meta":614},[611],"{agentName}.{dimension}\n",[383,2765,2763],{"__ignoreMap":614},[354,2767,660,2768,664,2771,664,2773,567],{},[383,2769,2770],{},"conversation_namer.title_quality",[383,2772,2739],{},[383,2774,2745],{},[553,2776,2778],{"id":2777},"_23-create-in-langfuse-ui","2.3 Create in Langfuse (UI)",[369,2780,2781,2792,2802,2812,2819,2822],{},[372,2782,2783,2784,2787,2788,2791],{},"Ensure an ",[375,2785,2786],{},"LLM Connection"," exists (",[375,2789,2790],{},"Settings → LLM Connections","). The judge model must support structured output.",[372,2793,2794,2795,2798,2799,567],{},"Open ",[375,2796,2797],{},"Evaluators"," → ",[375,2800,2801],{},"+ Set up Evaluator",[372,2803,2804,2805,2808,2809,567],{},"Pick a managed template, or ",[375,2806,2807],{},"Custom"," and paste your judge prompt with ",[383,2810,2811],{},"{{variables}}",[372,2813,2814,2815,2818],{},"Choose ",[375,2816,2817],{},"score type"," (boolean \u002F categorical \u002F numeric). For categorical, define labels and numeric mapping.",[372,2820,2821],{},"Map variables to Input \u002F Output \u002F Expected output (add JSONPath if needed).",[372,2823,2824,2825,567],{},"Save. Attach these evaluators when ",[358,2826,2827],{"href":409},"running experiments",[354,2829,2830,2831,567],{},"Official guide: ",[358,2832,2835],{"href":2833,"rel":2834},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fevaluation-methods\u002Fllm-as-a-judge",[362],"LLM-as-a-Judge",[553,2837,2839],{"id":2838},"_24-evaluator-reference-expand-as-needed","2.4 Evaluator reference (expand as needed)",[704,2841,2842,2899,2987,3033,3110,3191,3223,3273],{},[707,2843,2846],{"icon":2844,"label":2845},"i-lucide-layout-list","2.4.1 Where evaluators fit in Langfuse",[413,2847,2848,2858],{},[416,2849,2850],{},[419,2851,2852,2855],{},[422,2853,2854],{},"If you want to...",[422,2856,2857],{},"Langfuse \u002F approach we use",[429,2859,2860,2869,2879,2888],{},[419,2861,2862,2865],{},[434,2863,2864],{},"Build a reusable set of test cases",[434,2866,2867],{},[358,2868,474],{"href":379},[419,2870,2871,2874],{},[434,2872,2873],{},"Compare prompt or model changes",[434,2875,2876],{},[358,2877,2878],{"href":409},"Experiments",[419,2880,2881,2884],{},[434,2882,2883],{},"Automatically score quality (grounding, alignment, tone, …)",[434,2885,2886],{},[375,2887,2835],{},[419,2889,2890,2893],{},[434,2891,2892],{},"Run deterministic checks (schema, length, exact match)",[434,2894,2895,2898],{},[375,2896,2897],{},"Code checks"," in the app or unit\u002Fintegration tests",[707,2900,2903,2965,2984],{"icon":2901,"label":2902},"i-lucide-gauge","2.4.2 Score types",[413,2904,2905,2918],{},[416,2906,2907],{},[419,2908,2909,2912,2915],{},[422,2910,2911],{},"Score type",[422,2913,2914],{},"Use when",[422,2916,2917],{},"Typical use",[429,2919,2920,2933,2952],{},[419,2921,2922,2927,2930],{},[434,2923,2924],{},[375,2925,2926],{},"Boolean",[434,2928,2929],{},"Clear pass\u002Ffail",[434,2931,2932],{},"Hard checks (e.g. grounding)",[419,2934,2935,2940,2949],{},[434,2936,2937],{},[375,2938,2939],{},"Categorical",[434,2941,2942,2943,386,2945,386,2947,393],{},"Small fixed tiers (",[383,2944,2749],{},[383,2946,2752],{},[383,2948,2755],{},[434,2950,2951],{},"Soft quality bands",[419,2953,2954,2959,2962],{},[434,2955,2956],{},[375,2957,2958],{},"Numeric",[434,2960,2961],{},"Fine-grained ranking",[434,2963,2964],{},"When tiers are too coarse",[354,2966,2967,2968,2798,2970,664,2973,2798,2975,664,2978,2798,2980,2983],{},"Map categorical labels to numbers when useful (e.g. ",[383,2969,2749],{},[383,2971,2972],{},"0",[383,2974,2752],{},[383,2976,2977],{},"0.5",[383,2979,2755],{},[383,2981,2982],{},"1",").",[354,2985,2986],{},"Prefer boolean for hard correctness; categorical (3-tier) for softer quality.",[707,2988,2991],{"icon":2989,"label":2990},"i-lucide-pencil","2.4.3 Design principles",[369,2992,2993,2999,3005,3011,3017,3023],{},[372,2994,2995,2998],{},[375,2996,2997],{},"One job per evaluator"," — do not mix grounding + writing quality in one score.",[372,3000,3001,3004],{},[375,3002,3003],{},"Say what to check"," — avoid long “do not score X” lists.",[372,3006,3007,3010],{},[375,3008,3009],{},"Define pass\u002Ffail or categories explicitly"," with short examples.",[372,3012,3013,3016],{},[375,3014,3015],{},"Derive criteria from the use case"," — do not hard-code one product’s interview branch unless it is universal.",[372,3018,3019,3022],{},[375,3020,3021],{},"Map only the data needed"," — use JSONPath when you only need a nested field.",[372,3024,3025,3028,3029,3032],{},[375,3026,3027],{},"Name and version stably"," — keep ",[383,3030,3031],{},"{agent}.{dimension}"," names unchanged so later experiment comparisons stay readable.",[707,3034,3037,3078,3092],{"icon":3035,"label":3036},"i-lucide-link","2.4.4 Variable mapping",[413,3038,3039,3049],{},[416,3040,3041],{},[419,3042,3043,3046],{},[422,3044,3045],{},"Common variable",[422,3047,3048],{},"Typical source",[429,3050,3051,3060,3069],{},[419,3052,3053,3057],{},[434,3054,3055],{},[383,3056,2708],{},[434,3058,3059],{},"Dataset item input",[419,3061,3062,3066],{},[434,3063,3064],{},[383,3065,2711],{},[434,3067,3068],{},"Run output (filled when an experiment executes)",[419,3070,3071,3075],{},[434,3072,3073],{},[383,3074,2715],{},[434,3076,3077],{},"Dataset expected output (optional)",[354,3079,3080,3081,3084,3085,3088,3089,567],{},"Use ",[375,3082,3083],{},"JSONPath"," when you only need a nested field (e.g. Output → ",[383,3086,3087],{},"$.findings","). Confirm in Langfuse ",[375,3090,3091],{},"Prompt Preview",[354,3093,3094,3100,3101,2983,3103,3106,3109],{},[375,3095,3096,3097],{},"Skip ",[383,3098,3099],{},"expected_output"," when the rubric lives fully in the judge prompt, or the check is reference-free (e.g. grounding against transcript in ",[383,3102,385],{},[3104,3105],"br",{},[375,3107,3108],{},"Use it"," when item-specific guidance lives in the dataset.",[707,3111,3114,3117,3165,3171,3180],{"icon":3112,"label":3113},"i-lucide-sparkles","2.4.5 Simple example: Conversation Namer title quality",[354,3115,3116],{},"A light judge — good first custom evaluator to practise the shape.",[413,3118,3119,3128],{},[416,3120,3121],{},[419,3122,3123,3125],{},[422,3124,515],{},[422,3126,3127],{},"Value",[429,3129,3130,3141,3150],{},[419,3131,3132,3137],{},[434,3133,3134],{},[375,3135,3136],{},"Name",[434,3138,3139],{},[383,3140,2770],{},[419,3142,3143,3148],{},[434,3144,3145],{},[375,3146,3147],{},"Score",[434,3149,2926],{},[419,3151,3152,3157],{},[434,3153,3154],{},[375,3155,3156],{},"Maps",[434,3158,3159,3161,3162,3164],{},[383,3160,2708],{}," → Input, ",[383,3163,2711],{}," → Output",[608,3166,3169],{"className":3167,"code":3168,"language":613,"meta":614},[611],"You evaluate whether the OUTPUT title is a good short label for the INPUT user message.\n\nPass (true) if ALL are true:\n- Title is non-empty\n- Title is at most 5 words\n- Title reflects the main topic of the user message (no unrelated subject)\n\nFail (false) otherwise.\nIf OUTPUT is empty or malformed, return false.\n\nINPUT:\n{{input}}\n\nOUTPUT:\n{{output}}\n",[383,3170,3168],{"__ignoreMap":614},[354,3172,3173,3174,2612,3177,567],{},"Score output: return ONLY ",[383,3175,3176],{},"true",[383,3178,3179],{},"false",[354,3181,3182,3183,3186,3187,3190],{},"For a pure length rule (",[383,3184,3185],{},"≤ 5 words","), prefer a ",[375,3188,3189],{},"unit test"," or in-app validation instead of a judge.",[707,3192,3195,3207,3213],{"icon":3193,"label":3194},"i-lucide-shield-check","2.4.6 Findings: grounding judge (Boolean)",[354,3196,3197,693,3200,3161,3202,3204,3205,393],{},[375,3198,3199],{},"Maps:",[383,3201,2708],{},[383,3203,2711],{}," → Output (no ",[383,3206,3099],{},[608,3208,3211],{"className":3209,"code":3210,"language":613,"meta":614},[611],"You evaluate whether the agent OUTPUT is grounded in the interview transcript from INPUT.\n\nTask:\nDecide if material claims in the summary, report, and findings are supported by the transcript.\n\nRules:\n- Every material claim must be supported by the transcript\n- Paraphrase is allowed; invention is not\n- Minor omissions are OK; fabrication is not\n\nFail (false) if any material claim is invented, over-precise beyond the transcript, contradicts the transcript without uncertainty, or invents rationale\u002Fabbreviation expansions not stated.\n\nPass (true) only if all material claims are transcript-supported.\nIf OUTPUT is empty or malformed, return false.\n\nList any ungrounded claims briefly before deciding.\n\nINPUT:\n{{input}}\n\nOUTPUT:\n{{output}}\n",[383,3212,3210],{"__ignoreMap":614},[354,3214,3215,3216,3218,3219,2612,3221,567],{},"Score reasoning: 1–3 sentences; if false, name the worst ungrounded claim(s).",[3104,3217],{},"\nScore output: return ONLY ",[383,3220,3176],{},[383,3222,3179],{},[707,3224,3227,3255,3261],{"icon":3225,"label":3226},"i-lucide-list-checks","2.4.7 Findings: directive alignment judge (Categorical)",[354,3228,3229,693,3232,2798,3234,664,3236,2798,3238,664,3240,2798,3242,3244,3246,693,3248,3161,3250,3252,3253,393],{},[375,3230,3231],{},"Categories:",[383,3233,2749],{},[383,3235,2972],{},[383,3237,2752],{},[383,3239,2977],{},[383,3241,2755],{},[383,3243,2982],{},[3104,3245],{},[375,3247,3199],{},[383,3249,2708],{},[383,3251,2711],{}," → Output (directive usually inside ",[383,3254,385],{},[608,3256,3259],{"className":3257,"code":3258,"language":613,"meta":614},[611],"You evaluate whether the Findings\u002FReport agent OUTPUT aligns with the interview DIRECTIVE in INPUT.\n\nTask:\nJudge how well the REPORT and FINDINGS deliver the directive’s information goals, based on what the transcript actually captured.\n\nDo not assume a fixed interview structure. Derive success criteria from the directive itself.\n\nCheck only:\n1) Main directive objective(s) appear in report and\u002For findings\n2) Key requested topics are covered when the transcript has answers\n3) Findings are decision-useful for the directive\n4) Important directive-relevant transcript content is not systematically missing\n\nIf the transcript lacked answers for a topic, do not penalize missing content for that topic.\n\nCategories:\n- fail: largely misses the directive, or major goals missing\n- partial: some alignment, important gaps remain\n- pass: strong alignment; minor gaps only\n\nINPUT:\n{{input}}\n\nOUTPUT:\n{{output}}\n",[383,3260,3258],{"__ignoreMap":614},[354,3262,3263,3264,3218,3266,664,3268,3270,3271,567],{},"Score reasoning: 2–4 sentences with strongest coverage and most important gaps.",[3104,3265],{},[383,3267,2749],{},[383,3269,2752],{},", or ",[383,3272,2755],{},[707,3274,3276],{"icon":2554,"label":3275},"2.4.8 Checklist before saving an evaluator",[466,3277,3279,3285,3291,3297,3303,3309,3315,3323],{"className":3278},[2559],[372,3280,3282,3284],{"className":3281},[2563],[385,3283],{"disabled":2408,"type":2566}," One clear dimension",[372,3286,3288,3290],{"className":3287},[2563],[385,3289],{"disabled":2408,"type":2566}," Score type matches the check (boolean \u002F categorical \u002F numeric)",[372,3292,3294,3296],{"className":3293},[2563],[385,3295],{"disabled":2408,"type":2566}," Categorical labels + numeric mapping defined (if used)",[372,3298,3300,3302],{"className":3299},[2563],[385,3301],{"disabled":2408,"type":2566}," Judge prompt states positive checks",[372,3304,3306,3308],{"className":3305},[2563],[385,3307],{"disabled":2408,"type":2566}," Variable mapping verified in Prompt Preview",[372,3310,3312,3314],{"className":3311},[2563],[385,3313],{"disabled":2408,"type":2566}," JSONPath used when only a nested field is needed",[372,3316,3318,693,3320,3322],{"className":3317},[2563],[385,3319],{"disabled":2408,"type":2566},[383,3321,3099],{}," only when item-specific guidance is required",[372,3324,3326,3328,3329],{"className":3325},[2563],[385,3327],{"disabled":2408,"type":2566}," Name follows ",[383,3330,3031],{},[497,3332],{},[500,3334,3336],{"id":3335},"_3-running-experiments","3. Running Experiments",[354,3338,3339,3340,3342,3343,3346,3347,3350],{},"Use a ",[375,3341,1161],{}," and ",[375,3344,3345],{},"evaluators"," together in a Langfuse ",[375,3348,3349],{},"Prompt Experiment"," to compare prompt versions and decide whether to ship a change.",[354,3352,3353,3354,567],{},"Example completed runs: ",[358,3355,3357,3359],{"href":486,"rel":3356},[362],[383,3358,491],{}," experiments",[354,3361,3362,3363,3368,3369],{},"Official docs: ",[358,3364,3367],{"href":3365,"rel":3366},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fexperiments\u002Fexperiments-via-ui",[362],"Experiments via UI"," · ",[358,3370,3373],{"href":3371,"rel":3372},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fexperiments\u002Fdata-model",[362],"Experiments data model",[553,3375,3377],{"id":3376},"_31-what-a-langfuse-experiment-is","3.1 What a Langfuse experiment is",[413,3379,3380,3390],{},[416,3381,3382],{},[419,3383,3384,3387],{},[422,3385,3386],{},"Concept",[422,3388,3389],{},"Meaning",[429,3391,3392,3408,3418,3428,3437],{},[419,3393,3394,3398],{},[434,3395,3396],{},[375,3397,438],{},[434,3399,3400,3401,3403,3404,664,3406,393],{},"Frozen test cases (",[383,3402,385],{},", optional ",[383,3405,389],{},[383,3407,392],{},[419,3409,3410,3415],{},[434,3411,3412],{},[375,3413,3414],{},"Prompt",[434,3416,3417],{},"Versioned prompt from Prompt Management",[419,3419,3420,3425],{},[434,3421,3422],{},[375,3423,3424],{},"Experiment (Prompt Experiment)",[434,3426,3427],{},"Runs the selected prompt on each dataset item",[419,3429,3430,3434],{},[434,3431,3432],{},[375,3433,448],{},[434,3435,3436],{},"Scores each experiment item output (LLM-as-judge and\u002For code)",[419,3438,3439,3444],{},[434,3440,3441],{},[375,3442,3443],{},"Experiment comparison",[434,3445,3446],{},"Side-by-side aggregate + item-level score comparison across runs",[608,3448,3451],{"className":3449,"code":3450,"language":613,"meta":614},[611],"Dataset item input\n        │\n        ▼\nPrompt version (variables filled from input)\n        │\n        ▼\nModel output\n        │\n        ▼\nEvaluators attach scores\n        │\n        ▼\nCompare runs → promote or reject prompt\n",[383,3452,3450],{"__ignoreMap":614},[354,3454,3455,3458,3459,3462],{},[375,3456,3457],{},"Important:"," one experiment can attach ",[375,3460,3461],{},"multiple evaluators",". Do not create one experiment per score.",[553,3464,3466],{"id":3465},"_32-prerequisites","3.2 Prerequisites",[354,3468,3469],{},"Before running an experiment, confirm:",[369,3471,3472,3483,3496,3502],{},[372,3473,3474,3476,3477,3479,3480,3482],{},[375,3475,3414],{}," in Prompt Management with ",[383,3478,2811],{}," matching dataset ",[383,3481,385],{}," keys",[372,3484,3485,3487,3488,2612,3490,3492,3493],{},[375,3486,438],{}," uploaded (",[383,3489,859],{},[383,3491,855],{},") — see ",[358,3494,3495],{"href":379},"§1",[372,3497,3498,3501],{},[375,3499,3500],{},"LLM connection"," configured; default evaluation model supports structured output for judges",[372,3503,3504,3506,3507],{},[375,3505,2797],{}," created and able to target Experiments — see ",[358,3508,3509],{"href":400},"§2",[553,3511,3513],{"id":3512},"_33-run-a-prompt-experiment-ui","3.3 Run a Prompt Experiment (UI)",[369,3515,3516,3525,3539,3586],{},[372,3517,3518,3519,3521,3522,3524],{},"Go to ",[375,3520,474],{}," → open the dataset (e.g. ",[383,3523,491],{},") → spot-check 1–2 items",[372,3526,3527,3528,386,3531,2798,3534,2798,3536],{},"Click ",[375,3529,3530],{},"Start Experiment",[375,3532,3533],{},"Run Experiment",[375,3535,3349],{},[375,3537,3538],{},"Create",[372,3540,3541,3542],{},"Configure:\n",[466,3543,3544,3550,3558,3563,3568,3581],{},[372,3545,3546,3549],{},[375,3547,3548],{},"Experiment name"," (see naming below)",[372,3551,3552,3554,3555],{},[375,3553,3414],{}," + ",[375,3556,3557],{},"prompt version",[372,3559,3560,3562],{},[375,3561,3500],{}," \u002F model settings",[372,3564,3565,3567],{},[375,3566,438],{}," (usually already selected)",[372,3569,3570,3571,3574,3575,664,3577,664,3579,393],{},"Optional: ",[375,3572,3573],{},"structured output"," schema (recommended for Findings: ",[383,3576,2230],{},[383,3578,2249],{},[383,3580,2268],{},[372,3582,3583,3585],{},[375,3584,2797],{}," to attach (all gate evaluators)",[372,3587,3527,3588],{},[375,3589,3538],{},[354,3591,3592],{},"Langfuse runs the prompt per item, stores outputs, runs evaluators asynchronously, and shows aggregate scores. Runtime depends on dataset size, prompt length, and judge count.",[553,3594,3596],{"id":3595},"_34-experiment-naming","3.4 Experiment naming",[608,3598,3601],{"className":3599,"code":3600,"language":613,"meta":614},[611],"{agent}-{role}-{promptVersion}-{yyyymmdd}\n",[383,3602,3600],{"__ignoreMap":614},[354,3604,3605],{},"Examples:",[466,3607,3608,3613],{},[372,3609,3610],{},[383,3611,3612],{},"findings-baseline-v12-20260729",[372,3614,3615],{},[383,3616,3617],{},"findings-candidate-v13-20260729",[354,3619,3620,3621,3624,3625,3628,3629,3632],{},"For prompt gates: ",[375,3622,3623],{},"baseline"," = current production prompt; ",[375,3626,3627],{},"candidate"," = proposed version; ",[375,3630,3631],{},"same dataset + same evaluators"," for both.",[553,3634,3636],{"id":3635},"_35-compare-experiments","3.5 Compare experiments",[354,3638,3639],{},"After runs complete:",[369,3641,3642,3647,3650],{},[372,3643,2794,3644,3646],{},[375,3645,2878],{}," (or the dataset’s Experiments tab)",[372,3648,3649],{},"Select baseline and candidate runs",[372,3651,3652],{},"Compare aggregate scores, item-level regressions (especially boolean fails), and judge comments on failures",[354,3654,3655],{},"Always spot-check a few failed items manually before promoting.",[553,3657,3659],{"id":3658},"_36-experiment-reference-expand-as-needed","3.6 Experiment reference (expand as needed)",[704,3661,3662,3771,3908,3958,3999,4033,4123,4195],{},[707,3663,3665,3674,3768],{"icon":3035,"label":3664},"3.6.1 Prompt ↔ dataset variable mapping (Findings)",[354,3666,3667,3668,3670,3671,567],{},"A prompt is usable for Prompt Experiments when its ",[383,3669,2811],{}," match dataset item ",[375,3672,3673],{},"input keys",[413,3675,3676,3689],{},[416,3677,3678],{},[419,3679,3680,3683],{},[422,3681,3682],{},"Prompt variable",[422,3684,3685,3686,3688],{},"Dataset ",[383,3687,385],{}," key",[429,3690,3691,3702,3713,3724,3735,3746,3757],{},[419,3692,3693,3698],{},[434,3694,3695],{},[383,3696,3697],{},"{{user_name}}",[434,3699,3700],{},[383,3701,1301],{},[419,3703,3704,3709],{},[434,3705,3706],{},[383,3707,3708],{},"{{user_role}}",[434,3710,3711],{},[383,3712,1321],{},[419,3714,3715,3720],{},[434,3716,3717],{},[383,3718,3719],{},"{{organisation_context}}",[434,3721,3722],{},[383,3723,1341],{},[419,3725,3726,3731],{},[434,3727,3728],{},[383,3729,3730],{},"{{source_type}}",[434,3732,3733],{},[383,3734,1361],{},[419,3736,3737,3742],{},[434,3738,3739],{},[383,3740,3741],{},"{{source_info}}",[434,3743,3744],{},[383,3745,1381],{},[419,3747,3748,3753],{},[434,3749,3750],{},[383,3751,3752],{},"{{task_directive}}",[434,3754,3755],{},[383,3756,1504],{},[419,3758,3759,3764],{},[434,3760,3761],{},[383,3762,3763],{},"{{interview_transcript}}",[434,3765,3766],{},[383,3767,1524],{},[354,3769,3770],{},"If variables and input keys do not match, the experiment will fail or run with empty fields.",[707,3772,3775,3780,3811,3816,3845,3850,3853,3856,3862,3867],{"icon":3773,"label":3774},"i-lucide-git-branch","3.6.2 Findings Agent workflow (baseline → candidate → decide)",[354,3776,3777],{},[375,3778,3779],{},"A) Baseline (current production prompt)",[369,3781,3782,3786,3789,3795,3802,3808],{},[372,3783,2794,3784],{},[383,3785,491],{},[372,3787,3788],{},"Start Prompt Experiment",[372,3790,3791,3792],{},"Select ",[375,3793,3794],{},"current production prompt version",[372,3796,3797,3798,664,3800,393],{},"Attach gate evaluators (",[383,3799,2739],{},[383,3801,2745],{},[372,3803,3804,3805],{},"Name: ",[383,3806,3807],{},"findings-baseline-\u003Cprod-version>",[372,3809,3810],{},"Wait for scores",[354,3812,3813],{},[375,3814,3815],{},"B) Candidate (new prompt)",[369,3817,3818,3825,3834,3840],{},[372,3819,3820,3821,3824],{},"Save prompt edits as a ",[375,3822,3823],{},"new prompt version"," in Prompt Management",[372,3826,3827,3828,693,3831,3833],{},"Run another Prompt Experiment on the ",[375,3829,3830],{},"same",[383,3832,491],{}," dataset",[372,3835,3836,3837,3839],{},"Attach the ",[375,3838,3830],{}," evaluators",[372,3841,3804,3842],{},[383,3843,3844],{},"findings-candidate-\u003Cnew-version>",[354,3846,3847],{},[375,3848,3849],{},"C) Decide",[354,3851,3852],{},"Promote only if grounding and directive alignment do not regress materially, and no new systemic failure pattern appears in item review.",[354,3854,3855],{},"If it fails: revise the candidate prompt and rerun candidate only (keep baseline fixed).",[608,3857,3860],{"className":3858,"code":3859,"language":613,"meta":614},[611],"Prod prompt vN\n   │\n   ▼\nBaseline experiment on findings\u002Fgate_test + gate evaluators\n   │\n   ▼\nEdit prompt → save vN+1\n   │\n   ▼\nCandidate experiment on SAME dataset + SAME evaluators\n   │\n   ▼\nCompare in Langfuse Experiments\n   │\n   ├─ Pass → promote vN+1\n   └─ Fail → revise prompt, rerun candidate\n",[383,3861,3859],{"__ignoreMap":614},[354,3863,3864],{},[375,3865,3866],{},"What to look for",[413,3868,3869,3878],{},[416,3870,3871],{},[419,3872,3873,3875],{},[422,3874,448],{},[422,3876,3877],{},"Prefer",[429,3879,3880,3890],{},[419,3881,3882,3887],{},[434,3883,3884],{},[383,3885,3886],{},"grounding",[434,3888,3889],{},"pass-rate ≥ baseline",[419,3891,3892,3897],{},[434,3893,3894],{},[383,3895,3896],{},"directive_alignment",[434,3898,3899,3900,3903,3904,3907],{},"higher ",[383,3901,3902],{},"% pass",", lower ",[383,3905,3906],{},"% fail",", mean mapped score ≥ baseline − tolerance",[707,3909,3912,3945,3952],{"icon":3910,"label":3911},"i-lucide-app-window","3.6.3 UI vs SDK experiments",[413,3913,3914,3923],{},[416,3915,3916],{},[419,3917,3918,3921],{},[422,3919,3920],{},"Approach",[422,3922,2914],{},[429,3924,3925,3935],{},[419,3926,3927,3932],{},[434,3928,3929],{},[375,3930,3931],{},"Experiments via UI (Prompt Experiments)",[434,3933,3934],{},"Prompt-only changes; variables map cleanly from dataset input",[419,3936,3937,3942],{},[434,3938,3939],{},[375,3940,3941],{},"Experiments via SDK",[434,3943,3944],{},"Full app\u002Fagent logic, tools, retrieval, custom runtime config",[354,3946,3947,3948,3951],{},"Findings Agent ",[375,3949,3950],{},"prompt iteration"," fits UI Prompt Experiments well when structured output is enforced.",[354,3953,3954,3955,3957],{},"If the flow depends heavily on tool calls \u002F multi-step orchestration that Prompt Experiments cannot reproduce, use ",[375,3956,3941],{}," (or hybrid: UI for prompt drafts, SDK for full-agent realism).",[707,3959,3962,3965,3990,3996],{"icon":3960,"label":3961},"i-lucide-snowflake","3.6.4 Dataset freeze rules during experiments",[354,3963,3964],{},"To keep comparisons fair:",[369,3966,3967,3976,3984],{},[372,3968,3969,3970,3972,3973,3975],{},"Do ",[375,3971,774],{}," edit\u002Fadd\u002Fdelete ",[383,3974,855],{}," items while comparing prompts",[372,3977,3978,3979,3981,3982],{},"Put fresh cases into ",[383,3980,859],{}," (or a scratch set), not into ",[383,3983,855],{},[372,3985,3986,3987,3989],{},"Only expand ",[383,3988,855],{}," with reviewed items after the current comparison cycle",[354,3991,3992,3993,3995],{},"Langfuse experiments run against the dataset state at experiment time. Treat ",[383,3994,855],{}," as frozen for the duration of a promotion decision.",[354,3997,3998],{},"Optional: use dataset versioning (Items tab → version view) when available, so you can re-run against a historical snapshot.",[707,4000,4003,4009,4025,4030],{"icon":4001,"label":4002},"i-lucide-braces","3.6.5 Structured output tip (Findings)",[354,4004,4005,4006,4008],{},"For Findings experiments, enable ",[375,4007,3573],{}," with a schema requiring:",[466,4010,4011,4016,4020],{},[372,4012,4013,4015],{},[383,4014,2230],{}," (string)",[372,4017,4018,4015],{},[383,4019,2249],{},[372,4021,4022,4024],{},[383,4023,2268],{}," (array)",[354,4026,4027,4028,2983],{},"This improves parseability for judges, consistency across items, and JSONPath mapping (e.g. ",[383,4029,3087],{},[354,4031,4032],{},"Schemas can be created\u002Fsaved in Langfuse Playground and reused in experiments.",[707,4034,4037],{"icon":4035,"label":4036},"i-lucide-bug","3.6.6 Debugging failed or empty scores",[413,4038,4039,4052],{},[416,4040,4041],{},[419,4042,4043,4046,4049],{},[422,4044,4045],{},"Symptom",[422,4047,4048],{},"Likely cause",[422,4050,4051],{},"Fix",[429,4053,4054,4065,4076,4087,4098,4109],{},[419,4055,4056,4059,4062],{},[434,4057,4058],{},"Experiment fails immediately",[434,4060,4061],{},"Prompt variables ≠ dataset input keys",[434,4063,4064],{},"Align names exactly",[419,4066,4067,4070,4073],{},[434,4068,4069],{},"Empty outputs",[434,4071,4072],{},"LLM connection \u002F model issue",[434,4074,4075],{},"Check project LLM connection + logs",[419,4077,4078,4081,4084],{},[434,4079,4080],{},"No evaluator scores",[434,4082,4083],{},"Evaluator not attached \u002F wrong target",[434,4085,4086],{},"Attach evaluators; target Experiments",[419,4088,4089,4092,4095],{},[434,4090,4091],{},"Judge mapping empty",[434,4093,4094],{},"Wrong source or JSONPath",[434,4096,4097],{},"Fix mapping; use Prompt Preview",[419,4099,4100,4103,4106],{},[434,4101,4102],{},"Noisy scores",[434,4104,4105],{},"Judge prompt too broad",[434,4107,4108],{},"Split into one-dimension evaluators",[419,4110,4111,4114,4117],{},[434,4112,4113],{},"Need judge internals",[434,4115,4116],{},"—",[434,4118,4119,4120],{},"Filter traces by environment ",[383,4121,4122],{},"langfuse-llm-as-a-judge",[707,4124,4126],{"icon":2554,"label":4125},"3.6.7 Checklist before promoting a prompt",[466,4127,4129,4135,4141,4152,4159,4165,4171,4177,4183,4189],{"className":4128},[2559],[372,4130,4132,4134],{"className":4131},[2563],[385,4133],{"disabled":2408,"type":2566}," Prompt version is saved in Prompt Management (not an unsaved playground edit)",[372,4136,4138,4140],{"className":4137},[2563],[385,4139],{"disabled":2408,"type":2566}," Baseline experiment exists for current production prompt",[372,4142,4144,4146,4147,4149,4150,393],{"className":4143},[2563],[385,4145],{"disabled":2408,"type":2566}," Candidate experiment used the ",[375,4148,3830],{}," dataset (e.g. ",[383,4151,491],{},[372,4153,4155,4146,4157,3839],{"className":4154},[2563],[385,4156],{"disabled":2408,"type":2566},[375,4158,3830],{},[372,4160,4162,4164],{"className":4161},[2563],[385,4163],{"disabled":2408,"type":2566}," Structured output schema enabled (if required by agent)",[372,4166,4168,4170],{"className":4167},[2563],[385,4169],{"disabled":2408,"type":2566}," Aggregate scores reviewed",[372,4172,4174,4176],{"className":4173},[2563],[385,4175],{"disabled":2408,"type":2566}," Item-level failures reviewed (especially grounding fails)",[372,4178,4180,4182],{"className":4179},[2563],[385,4181],{"disabled":2408,"type":2566}," No dataset edits happened between baseline and candidate",[372,4184,4186,4188],{"className":4185},[2563],[385,4187],{"disabled":2408,"type":2566}," Promotion decision documented (pass \u002F fail + reason)",[372,4190,4192,4194],{"className":4191},[2563],[385,4193],{"disabled":2408,"type":2566}," Production prompt pointer updated only after pass",[707,4196,4199,4276,4281],{"icon":4197,"label":4198},"i-lucide-bookmark","3.6.8 Quick reference — Findings Agent",[413,4200,4201,4210],{},[416,4202,4203],{},[419,4204,4205,4207],{},[422,4206,424],{},[422,4208,4209],{},"Langfuse object",[429,4211,4212,4221,4231,4238,4249,4257,4268],{},[419,4213,4214,4217],{},[434,4215,4216],{},"Gate dataset",[434,4218,4219],{},[383,4220,491],{},[419,4222,4223,4226],{},[434,4224,4225],{},"Smoke dataset",[434,4227,4228],{},[383,4229,4230],{},"findings\u002Fsmoke_test",[419,4232,4233,4235],{},[434,4234,3414],{},[434,4236,4237],{},"Findings Agent prompt in Prompt Management",[419,4239,4240,4243],{},[434,4241,4242],{},"Gate evaluators",[434,4244,4245,664,4247],{},[383,4246,2739],{},[383,4248,2745],{},[419,4250,4251,4254],{},[434,4252,4253],{},"Experiment type",[434,4255,4256],{},"Prompt Experiment (UI)",[419,4258,4259,4262],{},[434,4260,4261],{},"Example runs",[434,4263,4264],{},[358,4265,4267],{"href":486,"rel":4266},[362],"findings\u002Fgate_test experiments",[419,4269,4270,4273],{},[434,4271,4272],{},"Decision",[434,4274,4275],{},"Compare baseline vs candidate → promote only on gate pass",[354,4277,4278],{},[375,4279,4280],{},"Tips",[369,4282,4283,4291,4297,4300],{},[372,4284,4285,4286,386,4288,4290],{},"Keep experiment names searchable (",[383,4287,3623],{},[383,4289,3627],{}," + prompt version).",[372,4292,4293,4294,4296],{},"Prefer foldered datasets (",[383,4295,491],{},") for clarity in the Datasets UI.",[372,4298,4299],{},"Attach all gate evaluators on every promotion run — do not compare incomplete score sets.",[372,4301,3080,4302,4304,4305,567],{},[383,4303,859],{}," for quick checks; promote prompts using ",[383,4306,855],{},[497,4308],{},[500,4310,4312],{"id":4311},"_4-related-docs","4. Related docs",[466,4314,4315,4319,4326,4332,4338,4343,4348],{},[372,4316,4317],{},[358,4318,329],{"href":330},[372,4320,4321],{},[358,4322,4325],{"href":4323,"rel":4324},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Foverview",[362],"Langfuse evaluation overview",[372,4327,4328],{},[358,4329,4331],{"href":2833,"rel":4330},[362],"Langfuse LLM-as-a-Judge",[372,4333,4334],{},[358,4335,4337],{"href":2655,"rel":4336},[362],"Langfuse Code Evaluators",[372,4339,4340],{},[358,4341,698],{"href":696,"rel":4342},[362],[372,4344,4345],{},[358,4346,3367],{"href":3365,"rel":4347},[362],[372,4349,4350],{},[358,4351,4354],{"href":4352,"rel":4353},"https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fevaluation\u002Fscores\u002Fdata-model",[362],"Scores data model",[4356,4357,4358],"style",{},"html pre.shiki code .sMK4o, html code.shiki .sMK4o{--shiki-light:#39ADB5;--shiki-default:#89DDFF;--shiki-dark:#89DDFF}html pre.shiki code .spNyl, html code.shiki .spNyl{--shiki-light:#9C3EDA;--shiki-default:#C792EA;--shiki-dark:#C792EA}html pre.shiki code .sBMFI, html code.shiki .sBMFI{--shiki-light:#E2931D;--shiki-default:#FFCB6B;--shiki-dark:#FFCB6B}html pre.shiki code .sfazB, html code.shiki .sfazB{--shiki-light:#91B859;--shiki-default:#C3E88D;--shiki-dark:#C3E88D}html .light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html.light .shiki span {color: var(--shiki-light);background: var(--shiki-light-bg);font-style: var(--shiki-light-font-style);font-weight: var(--shiki-light-font-weight);text-decoration: var(--shiki-light-text-decoration);}html .default .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .shiki span {color: var(--shiki-default);background: var(--shiki-default-bg);font-style: var(--shiki-default-font-style);font-weight: var(--shiki-default-font-weight);text-decoration: var(--shiki-default-text-decoration);}html .dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html.dark .shiki span {color: var(--shiki-dark);background: var(--shiki-dark-bg);font-style: var(--shiki-dark-font-style);font-weight: var(--shiki-dark-font-weight);text-decoration: var(--shiki-dark-text-decoration);}html pre.shiki code .sbssI, html code.shiki .sbssI{--shiki-light:#F76D47;--shiki-default:#F78C6C;--shiki-dark:#F78C6C}html pre.shiki code .sTEyZ, html code.shiki .sTEyZ{--shiki-light:#90A4AE;--shiki-default:#EEFFFF;--shiki-dark:#BABED8}",{"title":614,"searchDepth":965,"depth":965,"links":4360},[4361,4367,4373,4381],{"id":502,"depth":965,"text":503,"children":4362},[4363,4364,4365,4366],{"id":555,"depth":984,"text":556},{"id":605,"depth":984,"text":606},{"id":672,"depth":984,"text":673},{"id":701,"depth":984,"text":702},{"id":2625,"depth":965,"text":2626,"children":4368},[4369,4370,4371,4372],{"id":2678,"depth":984,"text":2679},{"id":2758,"depth":984,"text":2759},{"id":2777,"depth":984,"text":2778},{"id":2838,"depth":984,"text":2839},{"id":3335,"depth":965,"text":3336,"children":4374},[4375,4376,4377,4378,4379,4380],{"id":3376,"depth":984,"text":3377},{"id":3465,"depth":984,"text":3466},{"id":3512,"depth":984,"text":3513},{"id":3595,"depth":984,"text":3596},{"id":3635,"depth":984,"text":3636},{"id":3658,"depth":984,"text":3659},{"id":4311,"depth":965,"text":4312},"How to evaluate LLM agents with Langfuse — datasets, evaluators, and experiments.","md",null,{},{"title":337,"description":4382},"0EnGvDfgEMon3yvnwyp4Na6BavnAN1Xc_jwDy73oPok",[4389,4391],{"title":333,"path":334,"stem":335,"description":4390,"children":-1},"This document provides a high-level overview, a component breakdown, and detailed diagrams of how ESPi (ESPROFILER Intelligence), the AI co-pilot and agent orchestrator built into ESPROFILER, handles and processes user queries.",{"title":341,"path":342,"stem":343,"description":614,"children":-1},1789726611341]