[{"data":1,"prerenderedAt":819},["ShallowReactive",2],{"blog-pii-redaction-llm-pipelines-en":3},{"slug":4,"published":5,"minutes":6,"category":7,"tags":8,"keywords":14,"about":25,"sources":35,"cover":96,"og":97,"expertise":98,"locales":99,"lang":100,"title":103,"description":104,"coverAlt":105,"metaTitle":106,"takeaways":107,"faq":113,"toc":132,"blocks":163,"others":533},"pii-redaction-llm-pipelines","2026-10-02",12,"security",[9,10,11,12,13],"PII redaction","GDPR","Microsoft Presidio","LLM security","Pseudonymisation",[15,16,17,18,19,20,21,22,23,24],"PII redaction LLM","how to remove PII before sending to LLM","Microsoft Presidio tutorial","PII detection German Hungarian","pseudonymisation vs anonymisation GDPR","reversible tokenization LLM prompts","redact PII in logs and traces","LLM data masking","PII guardrail LiteLLM","test PII detection recall",[26,29,32],{"name":27,"url":28},"Personal data","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FPersonal_data",{"name":30,"url":31},"General Data Protection Regulation","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FGeneral_Data_Protection_Regulation",{"name":33,"url":34},"Data anonymization","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FData_anonymization",[36,39,42,45,48,51,54,57,60,63,66,69,72,75,78,81,84,87,90,93],{"title":37,"url":38},"Microsoft Presidio: documentation (limitations, methods)","https:\u002F\u002Fpresidio.dataprivacystack.org\u002F",{"title":40,"url":41},"Presidio: supported entities and country-specific recognizers","https:\u002F\u002Fpresidio.dataprivacystack.org\u002Fsupported_entities\u002F",{"title":43,"url":44},"Presidio: supporting additional languages","https:\u002F\u002Fpresidio.dataprivacystack.org\u002Fanalyzer\u002Flanguages\u002F",{"title":46,"url":47},"Presidio: anonymizer operators","https:\u002F\u002Fpresidio.dataprivacystack.org\u002Fanonymizer\u002F",{"title":49,"url":50},"Google Cloud: infoTypes reference","https:\u002F\u002Fdocs.cloud.google.com\u002Fsensitive-data-protection\u002Fdocs\u002Finfotypes-reference",{"title":52,"url":53},"Google Cloud: pseudonymization in Sensitive Data Protection","https:\u002F\u002Fdocs.cloud.google.com\u002Fsensitive-data-protection\u002Fdocs\u002Fpseudonymization",{"title":55,"url":56},"Microsoft Learn: Azure Language PII detection language support","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fazure\u002Fai-services\u002Flanguage-service\u002Fpersonally-identifiable-information\u002Flanguage-support",{"title":58,"url":59},"AWS: Detecting PII entities with Amazon Comprehend","https:\u002F\u002Fdocs.aws.amazon.com\u002Fcomprehend\u002Flatest\u002Fdg\u002Fhow-pii.html",{"title":61,"url":62},"spaCy: models and languages","https:\u002F\u002Fspacy.io\u002Fusage\u002Fmodels",{"title":64,"url":65},"Hugging Face: novakat\u002Fnerkor-hubert (Hungarian NER)","https:\u002F\u002Fhuggingface.co\u002Fnovakat\u002Fnerkor-hubert",{"title":67,"url":68},"arXiv: An Evaluation Study of Hybrid Methods for Multilingual PII Detection","https:\u002F\u002Farxiv.org\u002Fabs\u002F2510.07551",{"title":70,"url":71},"LiteLLM: Presidio PII masking guardrail","https:\u002F\u002Fdocs.litellm.ai\u002Fdocs\u002Fproxy\u002Fguardrails\u002Fpii_masking_v2",{"title":73,"url":74},"Langfuse: masking","https:\u002F\u002Flangfuse.com\u002Fdocs\u002Fobservability\u002Ffeatures\u002Fmasking",{"title":76,"url":77},"GDPR Article 4: definitions (pseudonymisation, 4(5))","https:\u002F\u002Fgdpr-info.eu\u002Fart-4-gdpr\u002F",{"title":79,"url":80},"EDPB: Guidelines 01\u002F2025 on Pseudonymisation","https:\u002F\u002Fwww.edpb.europa.eu\u002Four-work-tools\u002Fdocuments\u002Fpublic-consultations\u002F2025\u002Fguidelines-012025-pseudonymisation_en",{"title":82,"url":83},"IAPP: EDPB publishes draft guidelines on pseudonymization","https:\u002F\u002Fiapp.org\u002Fnews\u002Fa\u002F-what-s-in-a-name-edpb-publishes-draft-guidelines-on-pseudonymization",{"title":85,"url":86},"Taylor Wessing: Analysis of the CJEU judgment in C-413\u002F23 P (EDPS v SRB)","https:\u002F\u002Fwww.taylorwessing.com\u002Fen\u002Finsights-and-events\u002Finsights\u002F2025\u002F09\u002Fanalysis-of-the-cjeu-judgment",{"title":88,"url":89},"Jones Day: CJEU clarifies scope of personal data in EDPS v SRB","https:\u002F\u002Fwww.jonesday.com\u002Fen\u002Finsights\u002F2025\u002F09\u002Fcjeu-clarifies-scope-of-personal-data-in-edps-v-srb-decision",{"title":91,"url":92},"IAPP: leaked Council Digital Omnibus compromise drops the revised personal data definition","https:\u002F\u002Fiapp.org\u002Fnews\u002Fa\u002Feu-member-states-leaked-digital-omnibus-compromise-proposal-eliminates-revised-gdpr-definition-of-personal-data",{"title":94,"url":95},"Law Health Tech: Pseudonymisation under the GDPR and the Digital Omnibus (May 2026)","https:\u002F\u002Flawhealthtech.com\u002F2026\u002F05\u002F04\u002Fpseudonymisation-under-the-gdpr-where-we-are-what-may-change-under-the-digital-omnibus-and-what-regulators-think\u002F","\u002Fimages\u002Fblog\u002Fpii-redaction-llm-pipelines\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fpii-redaction-llm-pipelines\u002Fog.jpg","ai-engineer",[100,101,102],"en","de","hu","PII redaction in LLM pipelines: where to redact, how, and what GDPR says","Where to redact PII in an LLM pipeline, reversible tokens vs masking, Presidio and cloud DLP, German and Hungarian gaps, GDPR on pseudonymised data, and tests.","Diagram: user input passes a redaction gate before the LLM, a token vault restores real values after the output check, and logs and traces only ever see redacted text.","PII redaction in LLM pipelines · Balázs Csorba",[108,109,110,111,112],"Redact at every boundary, not once: ingest, prompt, output, logs and traces each leak differently, and traces and logs are the boundary teams forget most often.","Reversible tokens (a vault that maps placeholders back to real values) keep answers useful, but the vault itself becomes a personal-data store that needs keys, access control and a short retention.","No detector finds everything. Presidio says so itself, and coverage for German is far better than for Hungarian, so measure recall per language on your own data instead of trusting a vendor list.","Under the GDPR, pseudonymised data stays personal data for whoever holds the key. The CJEU ruling of 4 September 2025 adds that it may not be personal data for a recipient who cannot re-identify, but that has to be shown, not assumed.","Treat redaction as defence in depth next to contracts, EU hosting and access control, and test it like any other feature: a labelled multilingual set, recall targets and a regression gate in CI.",[114,117,120,123,126,129],{"q":115,"a":116},"How do I remove PII before sending data to an LLM?","Put a redaction gate between your application and the model API. Detect entities with a mix of patterns, checksums and an NER model (for example Microsoft Presidio), replace each one with a typed placeholder such as PERSON_1, send the redacted text, and optionally map the placeholders back in the answer. Apply the same gate to documents before indexing, and to logs and traces.",{"q":118,"a":119},"What is the difference between redaction, masking and tokenisation?","Redaction removes the value, masking replaces characters with a symbol, and tokenisation replaces the value with a stand-in that can be mapped back through a separate vault. Only tokenisation (or encryption) is reversible. Hashing is one-way but can be guessed for low-entropy values such as phone numbers.",{"q":121,"a":122},"Is pseudonymised data personal data under the GDPR?","For the party that holds the additional information needed to re-identify people, yes. Article 4(5) defines pseudonymisation as a safeguard, not as anonymisation. The CJEU held on 4 September 2025 (C-413\u002F23 P) that for a recipient who has no reasonable means to re-identify, the same data may not be personal data, so the assessment depends on perspective.",{"q":124,"a":125},"Does Microsoft Presidio support German and Hungarian?","Presidio can run in other languages through its NLP engine configuration, and its documentation lists German-specific recognizers such as tax IDs and ID cards. I found no Hungarian-specific recognizers in the list, and spaCy has no trained Hungarian pipeline, so for Hungarian you need your own recognizers or a transformer model and your own evaluation.",{"q":127,"a":128},"Can PII detection guarantee that nothing leaks?","No. Presidio states that because it uses automated detection there is no guarantee that it finds all sensitive information. Names, free text, typos and context-dependent identifiers produce false negatives, so redaction should be one layer next to access control, contracts and EU data residency.",{"q":130,"a":131},"How do I test a PII redaction pipeline?","Build a labelled set in every language you serve, with realistic noise, and measure recall per entity type, because a missed name matters more than a false alarm. Add canary values that must never appear in logs or provider requests, run the suite in CI, and re-run it whenever the model, the language models or the recognizers change.",[133,136,139,142,145,148,151,154,157,160],{"id":134,"title":135},"where-to-redact","Where to redact: five boundaries",{"id":137,"title":138},"techniques","Redaction, masking, tokenisation: choosing the technique",{"id":140,"title":141},"reversible-tokens","Reversible tokenisation: useful, with a vault attached",{"id":143,"title":144},"tools","Tools: Presidio, cloud services and NER models",{"id":146,"title":147},"german-hungarian","German and Hungarian: the coverage gap",{"id":149,"title":150},"false-negatives","False negatives: the failure that matters",{"id":152,"title":153},"gdpr","The GDPR view: pseudonymised is not anonymous",{"id":155,"title":156},"testing","Testing redaction like a feature",{"id":158,"title":159},"first-steps","What I would do first",{"id":161,"title":162},"sources","Sources",[164,168,177,180,183,192,226,229,230,233,315,318,319,322,339,351,352,355,371,374,375,378,395,398,399,402,424,427,428,431,434,437,443,444,452,465,466,469,470],{"type":165,"content":166},"paragraph",[167],"Most teams add PII handling to an LLM feature the same way: one regex for e-mail addresses in front of the API call, and a note in the backlog. It works in the demo and fails in production, because personal data does not enter an LLM system at one point. It arrives in the user message, in the documents you index, in the tool results an agent reads, and then it is copied into the model output, the application log, the trace backend and the evaluation set.",{"type":165,"content":169},[170,171,176],"This article is how I would design redaction for a European company: where the gates go, which technique to use at each one, what the tools can and cannot do (including for German and Hungarian), how to read the current GDPR position on pseudonymised data, and how to test the whole thing. It is engineering advice, not legal advice, and it complements the hosting and contract questions in ",{"tag":172,"to":173,"children":174},"link","\u002Fblog\u002Fgdpr-llm-api-eu-data-residency",[175],"GDPR and LLM APIs: EU data residency",".",{"type":178,"level":179,"id":134,"text":135},"heading",2,{"type":165,"content":181},[182],"Think of the pipeline as five boundaries where text crosses into a system you do not fully control or that lives longer than the request. Each one needs its own decision.",{"type":184,"attrs":185,"inner":189,"caption":190},"diagram",{"viewBox":186,"role":187,"aria-labelledby":188},"0 0 764 290","img","d1-pii-t d1-pii-d","\u003Ctitle id=\"d1-pii-t\">Redaction gates in an LLM pipeline\u003C\u002Ftitle>\u003Cdesc id=\"d1-pii-d\">Flow from input through a redaction gate to the LLM, then an output check and the user. A token vault connects the redaction gate and the output check. Logs and traces below receive only redacted text.\u003C\u002Fdesc>\u003Ctext x=\"20\" y=\"28\" class=\"d-title\">Redaction gates in an LLM pipeline\u003C\u002Ftext>\u003Crect x=\"20\" y=\"50\" width=\"124\" height=\"70\" rx=\"10\" class=\"d-sky\" \u002F>\u003Ctext x=\"82\" y=\"81\" text-anchor=\"middle\" class=\"d-text\">1 Input\u003C\u002Ftext>\u003Ctext x=\"82\" y=\"102\" text-anchor=\"middle\" class=\"d-small\">chat, files, RAG\u003C\u002Ftext>\u003Cpath d=\"M144 85 H162\" class=\"d-line\" \u002F>\u003Cpath d=\"M170 85 l-9 -5 v10 z\" class=\"d-head\" \u002F>\u003Crect x=\"170\" y=\"50\" width=\"124\" height=\"70\" rx=\"10\" class=\"d-accent\" \u002F>\u003Ctext x=\"232\" y=\"81\" text-anchor=\"middle\" class=\"d-text\">2 Redact\u003C\u002Ftext>\u003Ctext x=\"232\" y=\"102\" text-anchor=\"middle\" class=\"d-small\">detect, tokenise\u003C\u002Ftext>\u003Cpath d=\"M294 85 H312\" class=\"d-line\" \u002F>\u003Cpath d=\"M320 85 l-9 -5 v10 z\" class=\"d-head\" \u002F>\u003Crect x=\"320\" y=\"50\" width=\"124\" height=\"70\" rx=\"10\" class=\"d-box\" \u002F>\u003Ctext x=\"382\" y=\"81\" text-anchor=\"middle\" class=\"d-text\">3 LLM\u003C\u002Ftext>\u003Ctext x=\"382\" y=\"102\" text-anchor=\"middle\" class=\"d-small\">external API\u003C\u002Ftext>\u003Cpath d=\"M444 85 H462\" class=\"d-line\" \u002F>\u003Cpath d=\"M470 85 l-9 -5 v10 z\" class=\"d-head\" \u002F>\u003Crect x=\"470\" y=\"50\" width=\"124\" height=\"70\" rx=\"10\" class=\"d-accent\" \u002F>\u003Ctext x=\"532\" y=\"81\" text-anchor=\"middle\" class=\"d-text\">4 Check\u003C\u002Ftext>\u003Ctext x=\"532\" y=\"102\" text-anchor=\"middle\" class=\"d-small\">output, restore\u003C\u002Ftext>\u003Cpath d=\"M594 85 H612\" class=\"d-line\" \u002F>\u003Cpath d=\"M620 85 l-9 -5 v10 z\" class=\"d-head\" \u002F>\u003Crect x=\"620\" y=\"50\" width=\"124\" height=\"70\" rx=\"10\" class=\"d-sky\" \u002F>\u003Ctext x=\"682\" y=\"81\" text-anchor=\"middle\" class=\"d-text\">5 User\u003C\u002Ftext>\u003Ctext x=\"682\" y=\"102\" text-anchor=\"middle\" class=\"d-small\">real values\u003C\u002Ftext>\u003Cpath d=\"M232 150 V129\" class=\"d-line\" \u002F>\u003Cpath d=\"M232 120 l-5 9 h10 z\" class=\"d-head\" \u002F>\u003Cpath d=\"M532 150 V129\" class=\"d-line\" \u002F>\u003Cpath d=\"M532 120 l-5 9 h10 z\" class=\"d-head\" \u002F>\u003Crect x=\"170\" y=\"150\" width=\"424\" height=\"44\" rx=\"10\" class=\"d-gold\" \u002F>\u003Ctext x=\"382\" y=\"177\" text-anchor=\"middle\" class=\"d-small\">Token vault: placeholder to real value, encrypted, short TTL\u003C\u002Ftext>\u003Crect x=\"20\" y=\"222\" width=\"724\" height=\"44\" rx=\"10\" class=\"d-box d-dash\" \u002F>\u003Ctext x=\"382\" y=\"249\" text-anchor=\"middle\" class=\"d-small\">Logs, traces, caches, eval sets: redacted text only\u003C\u002Ftext>",[191],"The model only ever sees placeholders. The vault restores real values for the user and never leaves your trust boundary; logs and traces never see real values.",{"type":193,"ordered":194,"items":195},"list",false,[196,206,211,216,221],[197,201,202,176],{"tag":198,"children":199},"strong",[200],"Ingest."," Redact documents before chunking and embedding. A vector store full of raw personal data is hard to delete from and widens every later leak. Decide per source whether you need the real value at all. See the chunking and indexing trade-offs in ",{"tag":172,"to":203,"children":204},"\u002Fblog\u002Frag-pipeline-chunking-hybrid-search-reranking",[205],"RAG pipeline: chunking, hybrid search, reranking",[207,210],{"tag":198,"children":208},[209],"Prompt."," The user message, retrieved context and tool results all go through the gate right before the API call. This is the most important boundary, because it is the one that controls what the provider receives.",[212,215],{"tag":198,"children":213},[214],"Output."," The model can repeat, infer or invent personal data. Check the answer before it is shown or stored, and restore tokens only where the viewer is allowed to see the real value.",[217,220],{"tag":198,"children":218},[219],"Logs."," Application and gateway logs are the classic leak. Log the redacted prompt, or a hash and a size, never the raw message.",[222,225],{"tag":198,"children":223},[224],"Traces and evals."," Observability tools store full prompts and completions by design. Langfuse, for example, offers masking hooks that run before data is exported, and plain OpenTelemetry setups can mask in the application or in a collector. Evaluation datasets built from production traffic inherit whatever the traces contain.",{"type":165,"content":227},[228],"A gateway such as LiteLLM can host the prompt-side gate. Its Presidio guardrail can run before the call, after the response, or only for logging, and it can parse model output to replace masked tokens with the original values. That is a convenient place to start, but note the limits: it handles the request and response, not your ingest job or your trace backend.",{"type":178,"level":179,"id":137,"text":138},{"type":165,"content":231},[232],"Detection finds the spans; the technique decides what replaces them. The choice is a trade-off between utility for the model, reversibility and the damage if the output leaks. The table is my assessment, built on the operators that Presidio and Google Cloud Sensitive Data Protection document.",{"type":234,"head":235,"rows":246},"table",[236,238,240,242,244],[237],"Technique",[239],"Reversible",[241],"Model utility",[243],"Main risk",[245],"Good for",[247,258,269,284,294,305],[248,250,252,254,256],[249],"Removal (empty or REDACTED)",[251],"No",[253],"Low: sentence structure breaks",[255],"Information loss",[257],"Logs, analytics, anything that never needs the value",[259,261,263,265,267],[260],"Typed placeholder (PERSON_1)",[262],"Only with a vault",[264],"High: the model still sees roles and relations",[266],"Vault becomes a data store",[268],"Prompts, summaries, support tickets",[270,277,278,280,282],[271,272,276],"Character masking (",{"tag":273,"children":274},"em",[275],"*","*1234)",[251],[279],"Low to medium",[281],"Leaks partial values and length",[283],"Display of card or phone tails",[285,287,288,290,292],[286],"Salted or keyed hash",[251],[289],"Medium: stable joins, unreadable text",[291],"Guessable for low-entropy values",[293],"Deduplication, join keys in analytics",[295,297,299,301,303],[296],"Deterministic or format-preserving encryption",[298],"Yes, with the key",[300],"Medium to high: same value gives same token",[302],"Key management; equality leaks",[304],"Structured fields, cross-document consistency",[306,308,309,311,313],[307],"Realistic surrogate (fake name)",[262],[310],"High, reads naturally",[312],"Fake value may collide with a real person",[314],"Demos, test data, evals",{"type":165,"content":316},[317],"Presidio ships operators for replace, redact, hash, mask, encrypt and custom functions, with decrypt as the built-in reverse. Google documents deterministic encryption (AES-SIV), format-preserving encryption (FPE-FFX) and HMAC-SHA-256 hashing, the first two reversible, and recommends keys wrapped by Cloud KMS. My default for prompts is typed, numbered placeholders: the model can still reason that PERSON_1 wrote to PERSON_2, and you decide at the output boundary who may see what.",{"type":178,"level":179,"id":140,"text":141},{"type":165,"content":320},[321],"Reversible tokens solve the usability problem: the user asks for a reply to a customer, the model drafts it around PERSON_1, and your gate restores the real name before display. Three design rules keep it safe.",{"type":193,"ordered":194,"items":323},[324,329,334],[325,328],{"tag":198,"children":326},[327],"Scope tokens to a session or request."," A fresh mapping per conversation avoids a global lookup table and stops tokens from becoming cross-conversation identifiers.",[330,333],{"tag":198,"children":331},[332],"Encrypt and expire the vault."," Keep it in your own infrastructure, encrypt it with a managed key, and delete mappings when the conversation ends or after a short TTL. It is personal data and needs the same deletion path as the rest.",[335,338],{"tag":198,"children":336},[337],"Restore only at the edge."," Do the substitution in the layer that renders to an authorised user, not inside the agent loop. Otherwise a tool call can carry real values back to places the model should not reach.",{"type":340,"variant":341,"title":342,"body":343},"callout","warn","Placeholders can be attacked",[344],[345,346,350],"If the model sees PERSON_1 and the output goes to a tool, an injected instruction can still ask for the real value to be restored or exfiltrated. Treat the vault as a privileged capability and keep it out of the model's reach. The patterns in ",{"tag":172,"to":347,"children":348},"\u002Fblog\u002Fprompt-injection-lethal-trifecta-patterns",[349],"prompt injection and the lethal trifecta"," apply directly.",{"type":178,"level":179,"id":143,"text":144},{"type":165,"content":353},[354],"There is no single answer; there are three families, and many production setups combine them.",{"type":193,"ordered":194,"items":356},[357,361,366],[358,360],{"tag":198,"children":359},[11]," (open source, self-hosted). It combines named-entity recognition, regular expressions, rule-based logic, checksums and context words. It is the usual starting point because you control where the text goes and you can add recognizers. Its documentation is candid: because detection is automated, there is no guarantee that it finds all sensitive information.",[362,365],{"tag":198,"children":363},[364],"Cloud services."," Google Cloud Sensitive Data Protection offers a long list of infoTypes with a location per type, including German ones such as passport, identity card, driver's licence, taxpayer ID and SCHUFA ID. Azure Language lists German and Hungarian for text PII, and its conversation PII is documented for English, French, German and Spanish only. Amazon Comprehend documents PII detection for English or Spanish text. These are managed and easy to start with, but sending raw text to a third-party detector is itself a transfer you must justify.",[367,370],{"tag":198,"children":368},[369],"NER models and hybrids."," Fine-tuned transformers can be run locally. One research paper on hybrid detection (regular expressions plus LLMs, tested on 13 low-resource languages) reports clearly better weighted F1 than fine-tuned NER models and zero-shot LLMs; treat it as a pointer to combine deterministic patterns with context-aware models, not as a ready product.",{"type":165,"content":372},[373],"My rule: deterministic recognizers with validation (IBAN checksums, tax-ID formats) for structured identifiers, an NER model for names and places, and an LLM-based pass only where recall matters more than cost and the model runs inside your boundary.",{"type":178,"level":179,"id":146,"text":147},{"type":165,"content":376},[377],"Most detectors are strongest in English. For a company in Austria or Hungary that is the practical risk, because the text is German or Hungarian, often mixed with English.",{"type":193,"ordered":194,"items":379},[380,385,390],[381,384],{"tag":198,"children":382},[383],"German."," Presidio documents German recognizers for tax IDs, passports, national ID cards, health insurance numbers and vehicle plates, spaCy ships trained German pipelines, and LiteLLM lists German as a supported guardrail language. Names also have to be told apart from the many capitalised common nouns, so test name recall on German text separately.",[386,389],{"tag":198,"children":387},[388],"Hungarian."," I found no Hungarian-specific recognizers in Presidio's entity list or in Google's infoType reference, and spaCy shows no trained Hungarian pipeline. Azure lists Hungarian for text PII. Open models exist, for example a huBERT-based NER model fine-tuned on the NerKor corpus with PER, ORG, LOC and MISC labels, but it is GPL-licensed and limited to 448 tokens of input, which matters for long documents.",[391,394],{"tag":198,"children":392},[393],"Language-specific context."," Presidio recognizers support one language each, and its documentation notes that while patterns such as regular expressions are language agnostic, the context words that raise confidence are not. A German recognizer needs words like \"Steuernummer\"; a Hungarian one needs \"adószám\" or \"TAJ-szám\", and Hungarian suffixes make names change form (Péter, Péternek, Péterrel).",{"type":165,"content":396},[397],"Hungarian-specific identifiers such as the tax number or the social security (TAJ) number are easy to add as custom pattern recognizers with checksums, and that is where I would start. Names are the hard part and need a model plus your own evaluation.",{"type":178,"level":179,"id":149,"text":150},{"type":165,"content":400},[401],"A false positive replaces a harmless word and costs a little quality. A false negative sends a real name to a provider and writes it to a log. Optimise for recall on the entities that matter, and accept noisy precision.",{"type":193,"ordered":194,"items":403},[404,409,414,419],[405,408],{"tag":198,"children":406},[407],"Names in free text:"," nicknames, lower-case typing in chat, names that are also common words, and inflected forms.",[410,413],{"tag":198,"children":411},[412],"Context-dependent identifiers:"," a job title plus a small town plus a date can identify a person without any single obvious entity.",[415,418],{"tag":198,"children":416},[417],"Format variants:"," phone numbers with odd spacing, IBANs split across lines, identifiers inside URLs or code blocks.",[420,423],{"tag":198,"children":421},[422],"Non-text inputs:"," OCR output from scanned documents, tool results in JSON, and file names.",{"type":165,"content":425},[426],"Because of these gaps, do not rely on redaction alone. Add provider-side controls (EU region, no training on data, zero retention where offered), least-privilege retrieval, and a rule that special-category data (health, for example) is not sent to a general model at all unless a documented basis exists.",{"type":178,"level":179,"id":152,"text":153},{"type":165,"content":429},[430],"Article 4(5) GDPR defines pseudonymisation as processing so that data can no longer be attributed to a specific person without additional information, provided that information is kept separately under technical and organisational measures. It is a safeguard, not an exit from the regulation. The EDPB adopted its Guidelines 01\u002F2025 on pseudonymisation on 16 January 2025 and consulted on them in early 2025; I could not confirm a final version, and the EDPB held a stakeholder event on the topic in December 2025, so treat the guidelines as still evolving.",{"type":165,"content":432},[433],"The CJEU added a nuance on 4 September 2025 in EDPS v SRB (C-413\u002F23 P). The data in that case had been pseudonymised by the Single Resolution Board, which kept the key, before being sent to Deloitte. The court confirmed that such data can be personal data for the original controller but not necessarily for a recipient who has no reasonable means to re-identify the people, and that the controller's duty to inform data subjects exists independently of the recipient's view. Commentators advise documenting why a recipient cannot re-identify and reassessing when technology or datasets change.",{"type":165,"content":435},[436],"What this means for an LLM pipeline, in my reading: your own systems that hold the vault or key still process personal data. Whether the model provider receives personal data depends on whether it has reasonable means to re-identify, which is a factual question about the text you send. Free-text prompts that survive redaction with rare combinations of details are weak evidence. Document the assessment, keep your privacy notice accurate about the transfer, and do not call placeholder text anonymous.",{"type":340,"variant":438,"title":439,"body":440},"note","The law may still move",[441],[442],"The Commission's November 2025 Digital Omnibus proposed a relative definition of personal data and an Article 41a letting the Commission specify when pseudonymised data is not personal data. A leaked Council compromise of February 2026 dropped the redefinition, and the EDPB and EDPS recommended deleting Article 41a. I could not verify the final outcome, so design for today's rules.",{"type":178,"level":179,"id":155,"text":156},{"type":165,"content":445},[446,447,451],"Redaction is a classifier, so test it like one, and wire it into the same practices as other LLM features (see ",{"tag":172,"to":448,"children":449},"\u002Fblog\u002Fllm-evals-for-product-features",[450],"LLM evals for product features",").",{"type":193,"ordered":453,"items":454},true,[455,457,459,461,463],[456],"Build a labelled set per language you serve, with realistic noise: typos, lower-case names, mixed German or Hungarian and English, tables and code blocks.",[458],"Report recall and precision per entity type, and set recall targets for names, contact data and identifiers separately. Track the rate of missed entities, not only the average.",[460],"Add canary values (fake but valid-looking names, IBANs and tax IDs) to test traffic and assert in CI that they never appear in provider requests, logs, traces or caches.",[462],"Test the round trip: tokenise, call the model, restore. Check that tokens survive paraphrasing, that unknown tokens are not restored, and that restoring is impossible without the right session.",[464],"Re-run the suite whenever the NLP model, a recognizer or the language configuration changes, and sample live traffic with human review to find drift.",{"type":178,"level":179,"id":158,"text":159},{"type":165,"content":467},[468],"Start with a prompt-side gate using Presidio or an equivalent, typed placeholders with a per-session vault, redaction before indexing, and masking in your trace backend. Add custom recognizers for German and Hungarian identifiers, measure recall on your own text, and pair all of it with EU hosting and contracts. The goal is not perfect detection, which no tool promises, but a pipeline in which one missed name does not end up in five different systems.",{"type":178,"level":179,"id":161,"text":162},{"type":193,"ordered":453,"items":471},[472,476,479,482,485,488,491,494,497,500,503,506,509,512,515,518,521,524,527,530],[473],{"tag":474,"href":38,"children":475},"a",[37],[477],{"tag":474,"href":41,"children":478},[40],[480],{"tag":474,"href":44,"children":481},[43],[483],{"tag":474,"href":47,"children":484},[46],[486],{"tag":474,"href":50,"children":487},[49],[489],{"tag":474,"href":53,"children":490},[52],[492],{"tag":474,"href":56,"children":493},[55],[495],{"tag":474,"href":59,"children":496},[58],[498],{"tag":474,"href":62,"children":499},[61],[501],{"tag":474,"href":65,"children":502},[64],[504],{"tag":474,"href":68,"children":505},[67],[507],{"tag":474,"href":71,"children":508},[70],[510],{"tag":474,"href":74,"children":511},[73],[513],{"tag":474,"href":77,"children":514},[76],[516],{"tag":474,"href":80,"children":517},[79],[519],{"tag":474,"href":83,"children":520},[82],[522],{"tag":474,"href":86,"children":523},[85],[525],{"tag":474,"href":89,"children":526},[88],[528],{"tag":474,"href":92,"children":529},[91],[531],{"tag":474,"href":95,"children":532},[94],[534,611,697,763],{"slug":535,"published":5,"minutes":6,"category":7,"tags":536,"keywords":541,"about":552,"sources":562,"cover":605,"og":606,"expertise":98,"locales":607,"lang":100,"title":608,"description":609,"coverAlt":610},"ai-agent-identity-least-privilege",[537,538,539,540],"AI agent identity","Least privilege","OAuth token exchange","Non-human identity",[542,543,544,545,546,547,548,549,550,551],"AI agent identity management","non-human identity AI agents","AI agent least privilege","OAuth token exchange for AI agents","on-behalf-of flow AI agent","Microsoft Entra Agent ID","Okta Agent SSO Cross App Access","AI agent credentials and secrets","offboarding AI agents","delegated vs autonomous agent access",[553,556,559],{"name":554,"url":555},"Identity management","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FIdentity_management",{"name":557,"url":558},"Principle of least privilege","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FPrinciple_of_least_privilege",{"name":560,"url":561},"OAuth","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FOAuth",[563,566,569,572,575,578,581,584,587,590,593,596,599,602],{"title":564,"url":565},"Microsoft Learn: What is Microsoft Entra Agent ID?","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fentra\u002Fagent-id\u002Fwhat-is-microsoft-entra-agent-id",{"title":567,"url":568},"Microsoft Learn: What are agent identities?","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fentra\u002Fagent-id\u002Fwhat-are-agent-identities",{"title":570,"url":571},"Microsoft Learn: Best practices for Microsoft Entra Agent ID","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fentra\u002Fagent-id\u002Fbest-practices-agent-id",{"title":573,"url":574},"Microsoft Learn: Microsoft Entra Agent ID logs","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fentra\u002Fagent-id\u002Fsign-in-audit-logs-agents",{"title":576,"url":577},"Microsoft Learn: How agent identity deletion works","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fentra\u002Fagent-id\u002Fconcept-agent-identity-deletion",{"title":579,"url":580},"Microsoft Learn: What's new in Microsoft Entra Agent ID","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fentra\u002Fagent-id\u002Fwhats-new-agent-id",{"title":582,"url":583},"Microsoft Learn: Microsoft Agent 365 overview","https:\u002F\u002Flearn.microsoft.com\u002Fen-us\u002Fmicrosoft-agent-365\u002Foverview",{"title":585,"url":586},"Okta: Okta brings first-class identity to AI agents with Agent SSO (24 August 2026)","https:\u002F\u002Fwww.okta.com\u002Fnewsroom\u002Fpress-releases\u002Fokta-brings-first-class-identity-to-ai-agents-with-agent-sso\u002F",{"title":588,"url":589},"Okta: Auth0 gives developers the identity layer to securely ship agentic apps (May 2026)","https:\u002F\u002Fwww.okta.com\u002Fnewsroom\u002Farticles\u002Fauth0-may-2026-product-innovations\u002F",{"title":591,"url":592},"IETF: RFC 8693, OAuth 2.0 Token Exchange","https:\u002F\u002Fwww.rfc-editor.org\u002Frfc\u002Frfc8693",{"title":594,"url":595},"Model Context Protocol blog: Enterprise-Managed Authorization, zero-touch OAuth for MCP (18 June 2026)","https:\u002F\u002Fblog.modelcontextprotocol.io\u002Fposts\u002Fenterprise-managed-auth\u002F",{"title":597,"url":598},"Model Context Protocol: Security best practices","https:\u002F\u002Fmodelcontextprotocol.io\u002Fspecification\u002Fdraft\u002Fbasic\u002Fsecurity_best_practices",{"title":600,"url":601},"WorkOS: AI agents and the multi-hop delegation problem","https:\u002F\u002Fworkos.com\u002Fblog\u002Foauth-multi-hop-delegation-ai-agents",{"title":603,"url":604},"TechCrunch: OpenAI launches Dots, its bubbly agentic avatar","https:\u002F\u002Ftechcrunch.com\u002F2026\u002F09\u002F29\u002Fopenai-launches-dots-its-bubbly-agentic-avatar\u002F","\u002Fimages\u002Fblog\u002Fai-agent-identity-least-privilege\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fai-agent-identity-least-privilege\u002Fog.jpg",[100,101,102],"AI agents are identities: least privilege for non-human users","Give every AI agent its own identity, delegated tokens and an off switch. Token exchange, secrets, audit and offboarding, plus what Entra, Okta and Auth0 ship.","Diagram: a user, an agent with its own identity and an identity provider that issues a short-lived, narrowly scoped token for an API.",{"slug":612,"published":5,"minutes":6,"category":7,"tags":613,"keywords":618,"about":629,"sources":639,"cover":691,"og":692,"expertise":98,"locales":693,"lang":100,"title":694,"description":695,"coverAlt":696},"eu-ai-act-gpai-high-risk-2026",[614,615,616,617],"EU AI Act","GPAI","High-risk AI","AI compliance",[619,620,621,622,623,624,625,626,627,628],"EU AI Act high-risk deadline","AI Act digital omnibus","AI Act GPAI obligations","AI Act provider vs deployer","EU AI Act 2 December 2027","GPAI code of practice","AI literacy Article 4","AI Act compliance checklist","AI Act OpenAI API provider deployer","AI Act mid-size company",[630,633,636],{"name":631,"url":632},"Artificial Intelligence Act","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FArtificial_Intelligence_Act",{"name":634,"url":635},"General-purpose artificial intelligence","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FFoundation_model",{"name":637,"url":638},"European Commission","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FEuropean_Commission",[640,643,646,649,652,655,658,661,664,667,670,673,676,679,682,685,688],{"title":641,"url":642},"Regulation (EU) 2024\u002F1689 (AI Act), EUR-Lex","https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2024\u002F1689\u002Foj",{"title":644,"url":645},"Regulation (EU) 2026\u002F1744 (Digital Omnibus on AI), EUR-Lex","https:\u002F\u002Feur-lex.europa.eu\u002Feli\u002Freg\u002F2026\u002F1744\u002Foj",{"title":647,"url":648},"AI Act Explorer: Digital Omnibus on AI, full amending text","https:\u002F\u002Fartificialintelligenceact.eu\u002Fai-act-explorer\u002Fdigital-omnibus\u002F",{"title":650,"url":651},"European Commission: AI Act regulatory framework and timeline","https:\u002F\u002Fdigital-strategy.ec.europa.eu\u002Fen\u002Fpolicies\u002Fregulatory-framework-ai",{"title":653,"url":654},"European Commission: Guidelines for providers of general-purpose AI models","https:\u002F\u002Fdigital-strategy.ec.europa.eu\u002Fen\u002Fpolicies\u002Fguidelines-gpai-providers",{"title":656,"url":657},"European Commission: Q&A on the guidelines for GPAI providers","https:\u002F\u002Fdigital-strategy.ec.europa.eu\u002Fen\u002Ffaqs\u002Fguidelines-obligations-general-purpose-ai-providers",{"title":659,"url":660},"European Commission: The General-Purpose AI Code of Practice","https:\u002F\u002Fdigital-strategy.ec.europa.eu\u002Fen\u002Fpolicies\u002Fcontents-code-gpai",{"title":662,"url":663},"European Commission: AI literacy Questions and Answers","https:\u002F\u002Fdigital-strategy.ec.europa.eu\u002Fen\u002Ffaqs\u002Fai-literacy-questions-answers",{"title":665,"url":666},"AI Act Article 25: Responsibilities along the AI value chain","https:\u002F\u002Fartificialintelligenceact.eu\u002Farticle\u002F25\u002F",{"title":668,"url":669},"AI Act Article 26: Obligations of deployers of high-risk AI systems","https:\u002F\u002Fartificialintelligenceact.eu\u002Farticle\u002F26\u002F",{"title":671,"url":672},"AI Act Article 27: Fundamental rights impact assessment","https:\u002F\u002Fartificialintelligenceact.eu\u002Farticle\u002F27\u002F",{"title":674,"url":675},"AI Act Article 53: Obligations for providers of general-purpose AI models","https:\u002F\u002Fartificialintelligenceact.eu\u002Farticle\u002F53\u002F",{"title":677,"url":678},"AI Act Article 99: Penalties","https:\u002F\u002Fartificialintelligenceact.eu\u002Farticle\u002F99\u002F",{"title":680,"url":681},"AI Act Article 101: Fines for providers of general-purpose AI models","https:\u002F\u002Fartificialintelligenceact.eu\u002Farticle\u002F101\u002F",{"title":683,"url":684},"Gibson Dunn: EU AI Act Omnibus Agreement, postponed high-risk deadlines (27 May 2026)","https:\u002F\u002Fwww.gibsondunn.com\u002Feu-ai-act-omnibus-agreement-postponed-high-risk-deadlines-and-other-key-changes\u002F",{"title":686,"url":687},"Orrick: EU AI Act Update, Digital Omnibus finalizes 8 compliance changes (29 July 2026)","https:\u002F\u002Fwww.orrick.com\u002Fen\u002FInsights\u002F2026\u002F07\u002FEU-AI-Act-Update-Digital-Omnibus-Finalizes-8-Compliance-Changes",{"title":689,"url":690},"K&L Gates: EU Digital Omnibus on AI enters into force (31 July 2026)","https:\u002F\u002Fwww.klgates.com\u002FEU-Digital-Omnibus-on-AI-Enters-Into-Force-7-31-2026","\u002Fimages\u002Fblog\u002Feu-ai-act-gpai-high-risk-2026\u002Fcover.webp","\u002Fimages\u002Fblog\u002Feu-ai-act-gpai-high-risk-2026\u002Fog.jpg",[100,101,102],"EU AI Act beyond Article 50: GPAI, high-risk dates and what to do now","The AI Act after the Digital Omnibus: GPAI duties, high-risk dates (2 Dec 2027 and 2 Aug 2028), provider vs deployer on OpenAI and Anthropic APIs, AI literacy.","Diagram: the AI Act timeline from February 2025 to August 2028, fanning out into GPAI duties, high-risk systems, provider and deployer roles and AI literacy.",{"slug":698,"published":699,"minutes":700,"category":7,"tags":701,"keywords":707,"about":717,"sources":726,"cover":757,"og":758,"expertise":98,"locales":759,"lang":100,"title":760,"description":761,"coverAlt":762},"prompt-injection-lethal-trifecta-patterns","2026-09-27",9,[702,703,704,705,706],"Prompt injection","AI agent security","Lethal trifecta","Design patterns","Red teaming",[708,709,703,710,711,712,713,714,715,716],"prompt injection","lethal trifecta","indirect prompt injection","dual LLM pattern","prompt injection design patterns","how to prevent prompt injection in AI agents","agents rule of two","prompt injection egress allowlist","OWASP agentic top 10 goal hijack",[718,720,723],{"name":702,"url":719},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FPrompt_injection",{"name":721,"url":722},"Large language model","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FLarge_language_model",{"name":724,"url":725},"OWASP","https:\u002F\u002Fowasp.org\u002F",[727,730,733,736,739,742,745,748,751,754],{"title":728,"url":729},"Simon Willison: The lethal trifecta for AI agents","https:\u002F\u002Fsimonwillison.net\u002F2025\u002FJun\u002F16\u002Fthe-lethal-trifecta\u002F",{"title":731,"url":732},"Design Patterns for Securing LLM Agents against Prompt Injections","https:\u002F\u002Farxiv.org\u002Fabs\u002F2506.08837",{"title":734,"url":735},"Simon Willison: Design patterns for securing LLM agents (summary)","https:\u002F\u002Fsimonwillison.net\u002F2025\u002FJun\u002F13\u002Fprompt-injection-design-patterns\u002F",{"title":737,"url":738},"Simon Willison: The Dual LLM pattern","https:\u002F\u002Fsimonwillison.net\u002F2023\u002FApr\u002F25\u002Fdual-llm-pattern\u002F",{"title":740,"url":741},"Defeating Prompt Injections by Design (CaMeL)","https:\u002F\u002Farxiv.org\u002Fabs\u002F2503.18813",{"title":743,"url":744},"The Attacker Moves Second","https:\u002F\u002Farxiv.org\u002Fabs\u002F2510.09023",{"title":746,"url":747},"Meta: Agents Rule of Two","https:\u002F\u002Fai.meta.com\u002Fblog\u002Fpractical-ai-agent-security\u002F",{"title":749,"url":750},"Anthropic: How we contain Claude","https:\u002F\u002Fwww.anthropic.com\u002Fengineering\u002Fhow-we-contain-claude",{"title":752,"url":753},"Claude Code documentation: Sandboxing","https:\u002F\u002Fcode.claude.com\u002Fdocs\u002Fen\u002Fsandboxing",{"title":755,"url":756},"promptfoo: OWASP Top 10 for Agentic Applications","https:\u002F\u002Fwww.promptfoo.dev\u002Fdocs\u002Fred-team\u002Fowasp-agentic-ai\u002F","\u002Fimages\u002Fblog\u002Fprompt-injection-lethal-trifecta-patterns\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fprompt-injection-lethal-trifecta-patterns\u002Fog.jpg",[100,101,102],"Prompt injection defense: the lethal trifecta and six design patterns","Why prompt injection can't be filtered away: the lethal trifecta, six design patterns that contain it, egress rules and a red-team checklist for AI agents.","Shield diagram with rings for egress control, data scope and pattern choice around a core labelled trifecta, broken",{"slug":764,"published":699,"minutes":765,"category":7,"tags":766,"keywords":772,"about":781,"sources":792,"cover":813,"og":814,"expertise":98,"locales":815,"lang":100,"title":816,"description":817,"coverAlt":818},"mcp-server-security-checklist",10,[767,768,769,770,771],"MCP security","Tool poisoning","MCP OAuth","Supply chain","Audit logs",[767,773,774,775,769,776,777,778,779,780],"MCP server security","MCP tool poisoning","MCP rug pull","MCP vulnerabilities","how to secure an MCP server","MCP authorization RFC 9207","NSA MCP guidance","MCP audit logging",[782,785,787,790],{"name":783,"url":784},"Model Context Protocol","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FModel_Context_Protocol",{"name":786,"url":561},"OAuth 2.0",{"name":788,"url":789},"OpenTelemetry","https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FOpenTelemetry",{"name":724,"url":791},"https:\u002F\u002Fen.wikipedia.org\u002Fwiki\u002FOWASP",[793,796,799,802,805,808,811],{"title":794,"url":795},"Invariant Labs: MCP Security Notification – Tool Poisoning Attacks (1 April 2025)","https:\u002F\u002Finvariantlabs.ai\u002Fblog\u002Fmcp-security-notification-tool-poisoning-attacks",{"title":797,"url":798},"OWASP: Top 10 for Agentic Applications for 2026 (9 December 2025)","https:\u002F\u002Fgenai.owasp.org\u002Fresource\u002Fowasp-top-10-for-agentic-applications-for-2026\u002F",{"title":800,"url":801},"MCP specification 2026-07-28: changelog (SEP-2468, SEP-2352, SEP-414)","https:\u002F\u002Fgithub.com\u002Fmodelcontextprotocol\u002Fmodelcontextprotocol\u002Fblob\u002Fmain\u002Fdocs\u002Fspecification\u002F2026-07-28\u002Fchangelog.mdx",{"title":803,"url":804},"RFC 9207: OAuth 2.0 Authorization Server Issuer Identification","https:\u002F\u002Fwww.rfc-editor.org\u002Frfc\u002Frfc9207",{"title":806,"url":807},"NSA: Model Context Protocol (MCP) – Security Design Considerations for AI-Driven Automation (May 2026)","https:\u002F\u002Fmedia.defense.gov\u002F2026\u002FJun\u002F02\u002F2003943289\u002F-1\u002F-1\u002F0\u002FCSI_MCP_SECURITY.PDF",{"title":809,"url":810},"Reed Smith: NSA publishes security guidance on designing AI systems with MCP (4 June 2026)","https:\u002F\u002Fwww.reedsmith.com\u002Four-insights\u002Fblogs\u002Fviewpoints\u002F102mvg9\u002Fnsa-publishes-security-guidance-on-designing-ai-systems-with-model-context-protoc\u002F",{"title":812,"url":750},"Anthropic: How we contain Claude across products (25 May 2026)","\u002Fimages\u002Fblog\u002Fmcp-server-security-checklist\u002Fcover.webp","\u002Fimages\u002Fblog\u002Fmcp-server-security-checklist\u002Fog.jpg",[100,101,102],"MCP security checklist: tool poisoning, rug pulls and OAuth","MCP security checklist: the threat model, tool poisoning, rug pulls, RFC 9207 issuer checks, per-issuer credentials, scoped tokens and audit logs.","Concentric rings from outside in: NSA guidance, OWASP agentic risks, issuer-bound credentials, pinned tool definitions and a scoped core token.",1791009037143]