dolibarr 25.0.0-alpha
privacy_guard.class.php
Go to the documentation of this file.
1<?php
2/* Copyright (C) 2026 Laurent Destailleur <eldy@users.sourceforge.net>
3 * Copyright (C) 2026 Nick Fragoulis
4 * Copyright (C) 2026 MDW <mdeweerd@users.noreply.github.com>
5 *
6 * This program is free software; you can redistribute it and/or modify
7 * it under the terms of the GNU General Public License as published by
8 * the Free Software Foundation; either version 3 of the License, or
9 * (at your option) any later version.
10 *
11 * This program is distributed in the hope that it will be useful,
12 * but WITHOUT ANY WARRANTY; without even the implied warranty of
13 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
14 * GNU General Public License for more details.
15 *
16 * You should have received a copy of the GNU General Public License
17 * along with this program. If not, see <https://www.gnu.org/licenses/>.
18 */
19
30{
34 public $db;
35
39 private $map = [];
40
44 private $salt = '';
45
49 private $index = 0;
50
59 public function mask($text)
60 {
61 $this->startSession();
62
63 // References / IDs (e.g. FA24-001, CUS-999)
64 // Must contain letters and numbers and separators
65 $text = preg_replace_callback(
66 '/\b(?=[A-Z0-9]*[0-9])(?=[A-Z0-9]*[A-Z])[A-Z0-9-_]{4,}\b/i',
71 function (array $m) {
72 return $this->createToken($m[0], 'REF');
73 },
74 $text
75 );
76
77 // Credit cards (13-19 digits, various separators)
78 // Use the Luhn algorithm to validate credit cards before masking
79 $text = preg_replace_callback(
80 '/\b(?:\d[ -]*?){13,19}\b/',
81 [$this, 'maskCreditCardCallback'],
82 $text
83 );
84
85 // IBAN (International Bank Account Number)
86 $text = preg_replace_callback(
87 '/\b[A-Z]{2}[0-9]{2}[a-zA-Z0-9]{4,30}\b/',
92 function (array $m) {
93 return $this->createToken($m[0], 'IBAN');
94 },
95 $text
96 );
97
98 // SWIFT / BIC Codes (8 or 11 characters)
99 $text = preg_replace_callback(
100 '/\b[A-Z]{6}[A-Z0-9]{2}([A-Z0-9]{3})?\b/',
105 function (array $m) {
106 return $this->createToken($m[0], 'SWIFT');
107 },
108 $text
109 );
110
111 // Generic bank account numbers (Context-aware)
112 // This looks for numbers preceded by keywords to reduce false positives.
113 $text = preg_replace_callback(
114 '/(?i)(?:account\s+num(?:ber)?|bank\s+acct|acct\s*#)[:\s#]*\b(\d{8,17})\b/',
119 function (array $m) {
120 return $this->createToken($m[0], 'BANKACCT');
121 },
122 $text
123 );
124
125 // Emails
126 $text = preg_replace_callback(
127 '/[a-zA-Z0-9._%+-]+@[a-zA-Z0-9.-]+\.[a-zA-Z]{2,}/',
132 function (array $m) {
133 return $this->createToken($m[0], 'EMAIL');
134 },
135 $text
136 );
137
138 // Vat numbers (EU: 2 letters + 2-12 chars)
139 $taxPatterns = [
140 [
141 'name' => 'EU VAT Numbers',
142 'regex' => '/\b(AT|BE|BG|CY|CZ|DE|DK|EE|EL|ES|FI|FR|GB|GR|HR|HU|IE|IT|LT|LU|LV|MT|NL|PL|PT|RO|SE|SI|SK)(?![a-z])[0-9A-Z]{2,12}\b/i',
143 'token' => 'VAT'
144 ],
145 [
146 'name' => 'Canadian GST/HST Numbers',
147 'regex' => '/\b\d{9}\s*RT\s*\d{4}\b/i',
148 'token' => 'TAXID'
149 ],
150 [
151 'name' => 'Australian ABN (Australian Business Number)',
152 'regex' => '/\b\d{2}\s*\d{3}\s*\d{3}\s*\d{3}\b/',
153 'token' => 'TAXID'
154 ],
155 [
156 'name' => 'Norwegian MVA (VAT) Numbers',
157 'regex' => '/\b\d{9}\s*MVA\b/i',
158 'token' => 'TAXID'
159 ],
160 [
161 'name' => 'Swiss VAT Numbers (MWST/TVA/IVA)',
162 'regex' => '/\bCHE-?\d{3}\.?\d{3}\.?\d{3}\s*(MWST|TVA|IVA)\b/i',
163 'token' => 'TAXID'
164 ],
165 ];
166
167 foreach ($taxPatterns as $pattern) {
168 $text = preg_replace_callback(
169 $pattern['regex'],
174 function (array $m) use ($pattern) {
175 return $this->createToken($m[0], $pattern['token']);
176 },
177 $text
178 );
179 }
180
181 // Phone numbers
182 $phonePatterns = [
183 [
184 'name' => 'Greek International Numbers',
185 // +30 / 0030 followed by exactly 10 national digits with the valid
186 // prefixes (2x landline, 69 mobile) — Greek numbers are 10 digits,
187 // so this is checked strictly; other countries fall through to the
188 // length-flexible generic pattern below.
189 'regex' => '/(?<![\w+])(?:\+30|0030)[\s.\-]?(?:2\d|69)(?:[\s.\-()]?\d){8}(?!\d)/',
190 'token' => 'PHONE'
191 ],
192 [
193 'name' => 'Generic International Numbers',
194 // Counts DIGITS (9-14 after the prefix) with optional single separators
195 // between them; the old char-counting class broke on spaced formats.
196 // No \b before '+': space->'+' is not a word boundary, so that \b
197 // could never match and the pattern was dead for '+30 ...' numbers.
198 'regex' => '/(?<![\w+])(?:\+|00)\d{1,3}(?:[\s.\-()]?\d){8,13}(?!\d)/',
199 'token' => 'PHONE'
200 ],
201 [
202 'name' => 'Greek National Numbers',
203 // This pattern is specific to Greece.
204 // Landlines: 10 digits starting with '2' (e.g., 210 123 4567).
205 // Mobiles: 10 digits starting with '69' (e.g., 698 123 4567).
206 // It matches numbers with optional separators like spaces, hyphens, or dots.
207 // Exactly 10 digits (landline 2x..., mobile 69...), separators optional
208 // BETWEEN digits. The old pattern counted characters: '210 2461234'
209 // (11 chars) overflowed {9} and slipped through unmasked, while
210 // '2026-09-07' (9 chars after the 2) was masked as a phone. The
211 // lookarounds forbid digit-adjacency, so it never fires inside longer
212 // digit runs (EAN barcodes, references).
213 'regex' => '/(?<!\d)(?:2\d|69)(?:[\s.\-()]?\d){8}(?!\d)/',
214 'token' => 'PHONE'
215 ],
216 [
217 'name' => 'French National Numbers',
218 // This pattern matches standard 10-digit French numbers.
219 // It handles common formats like "01 23 45 67 89", "01.23.45.67.89", or "0123456789".
220 // It covers all geographic prefixes (01-05) and mobiles (06, 07).
221 'regex' => '/\b0[1-9](?:[\s.-]?\d){8}\b/',
222 'token' => 'PHONE'
223 ],
224 ];
225
226 foreach ($phonePatterns as $pattern) {
227 $text = preg_replace_callback(
228 $pattern['regex'],
233 function (array $m) use ($pattern) {
234 return $this->createToken($m[0], $pattern['token']);
235 },
236 $text
237 );
238 }
239
240 // Address patterns
241 // Initialize translation and exclusions
242 // We need the global $langs object to get dynamic month names.
243 global $langs;
244
245 // Hardcoded Fallbacks (English, French, Spanish, German, Greek)
246 // We keep these hardcoded because an invoice might be in English even if the ERP is in Greek.
247 $hardcoded_excludes = [
248 // English
249 'January',
250 'February',
251 'March',
252 'April',
253 'May',
254 'June',
255 'July',
256 'August',
257 'September',
258 'October',
259 'November',
260 'December',
261 'Jan',
262 'Feb',
263 'Mar',
264 'Apr',
265 'Jun',
266 'Jul',
267 'Aug',
268 'Sep',
269 'Oct',
270 'Nov',
271 'Dec',
272 // Greek
273 'Ιανουάριος',
274 'Φεβρουάριος',
275 'Μάρτιος',
276 'Απρίλιος',
277 'Μάιος',
278 'Ιούνιος',
279 'Ιούλιος',
280 'Αύγουστος',
281 'Σεπτέμβριος',
282 'Οκτώβριος',
283 'Νοέμβριος',
284 'Δεκέμβριος',
285 'Ιαν',
286 'Φεβ',
287 'Μαρ',
288 'Απρ',
289 'Μαι',
290 'Ιουν',
291 'Ιουλ',
292 'Αυγ',
293 'Σεπ',
294 'Οκτ',
295 'Νοε',
296 'Δεκ',
297 // French
298 'Janvier',
299 'Février',
300 'Mars',
301 'Avril',
302 'Mai',
303 'Juin',
304 'Juillet',
305 'Août',
306 'Septembre',
307 'Octobre',
308 'Novembre',
309 'Décembre',
310 // German
311 'Januar',
312 'Februar',
313 'März',
314 'Juni',
315 'Juli',
316 'Oktober',
317 'Dezember',
318 // Spanish
319 'Enero',
320 'Febrero',
321 'Marzo',
322 'Abril',
323 'Mayo',
324 'Junio',
325 'Julio',
326 'Agosto',
327 'Septiembre',
328 'Octubre',
329 'Noviembre',
330 'Diciembre',
331 // ERP Noise (Common False Positives)
332 'Page',
333 'Pag',
334 'Vol',
335 'Volume',
336 'Inv',
337 'Invoice',
338 'Tel',
339 'Fax',
340 'Mob',
341 'Email',
342 'Vat',
343 'Tax',
344 'Sarl',
345 'Gmbh',
346 'Inc',
347 'Ltd',
348 'Total',
349 'Subtotal'
350 ];
351
352 // Dynamic Dolibarr Translations
353 // If $langs is available, we add the months in the current user's language.
354 $dynamic_excludes = [];
355 if (is_object($langs)) {
356 for ($i = 1; $i <= 12; $i++) {
357 $key = sprintf("%02d", $i); // 01, 02...
358 $dynamic_excludes[] = $langs->trans('Month' . $key); // Full name
359 $dynamic_excludes[] = $langs->trans('MonthShort' . $key); // Short name
360 }
361 }
362
363 // Merge and Deduplicate
364 // We combine hardcoded list + dynamic list + noise words
365 $all_excludes_array = array_unique(array_merge($hardcoded_excludes, $dynamic_excludes));
366
367 // Remove empty entries just in case
368 $all_excludes_array = array_filter($all_excludes_array);
369
370 // Create the Regex string: "January|Feb|Μάρτιος|Page..."
371 // We use preg_quote to ensure no special characters break the regex (though rare in months).
372 $excluded_words_regex = implode('|', array_map(
373 function (string $word): string {
374 return preg_quote($word, '/');
375 },
376 $all_excludes_array
377 ));
378
379
380 // Define address keywords
381 $address_keywords = 'Street|St|Road|Rd|Avenue|Ave|Lane|Ln|Boulevard|Blvd|Rue|Via|Strasse|Platz|Drive|Dr|Court|Ct|Way|Plaza|Square|Sq|Οδός|Λεωφόρος|Διεύθυνση|Piazza|Avenida';
382
383
384 // Define patterns
385 $addressPatterns = [
386 [
387 'name' => 'Number First (e.g., 123 Main St)',
388 // 123 Main St
389 'regex' => '/\b\d{1,5}\s+(?:[\p{L}\p{N}\.\'\-]+\s+){1,6}(?:' . $address_keywords . ')\b/ui',
390 'token' => 'ADDR'
391 ],
392 [
393 'name' => 'Keyword First (e.g., Rue de la Paix 12)',
394 // Rue de la Paix 12
395 'regex' => '/\b(?:' . $address_keywords . ')\s+(?:[\p{L}\p{N}\.\'\-]+\s+){1,6}\d{1,5}\b/ui',
396 'token' => 'ADDR'
397 ],
398 [
399 'name' => 'Name First, Keyword Middle (e.g., Main St 12)',
400 // Main St 12
401 'regex' => '/\b(?:[\p{L}\p{N}\.\'\-]+\s+){1,4}(?:' . $address_keywords . ')\s+\d{1,5}\b/ui',
402 'token' => 'ADDR'
403 ],
404 [
405 'name' => 'Name First, No Keyword (Strict)',
406 // Matches: "ΦΟΡΜΙΩΝΟΣ 101" or "Musterway 12"
407 // Ignores: "January 2024", "Page 1", "Invoice 2023"
408 // Logic:
409 // 1. Negative Lookahead (?!(?:...)\b): If next word is in exclusion list, STOP.
410 // 2. \p{Lu}: Must start with Uppercase Letter (Unicode safe).
411 'regex' => '/\b(?!(?:' . $excluded_words_regex . ')\b)\p{Lu}[\p{L}\p{N}\.\'\-]+\s+\d{1,5}\b/u',
412 'token' => 'ADDR'
413 ],
414 ];
415
416 foreach ($addressPatterns as $pattern) {
417 // We use a callback to replace the found address with a token
418 $text = preg_replace_callback(
419 $pattern['regex'],
424 function (array $m) use ($pattern) {
425 return $this->createToken($m[0], $pattern['token']);
426 },
427 $text
428 );
429 }
430
431 // Zip codes
432 $zipCodePatterns = [
433 [
434 'name' => 'UK Postal Codes',
435 // Matches UK postcodes like SW1A 0AA, M1 1AA, B33 8TH.
436 'regex' => '/\b[A-Z]{1,2}\d[A-Z\d]? ?\d[A-Z]{2}\b/i',
437 'token' => 'ZIP'
438 ],
439 [
440 'name' => 'Canadian Postal Codes',
441 // Matches Canadian codes like K1A 0B1 or V6A 1H1.
442 'regex' => '/\b[A-CEGHJ-NPR-STV-Z]\d[A-CEGHJ-NPR-STV-Z][ -]?\d[A-CEGHJ-NPR-STV-Z]\d\b/i',
443 'token' => 'ZIP'
444 ],
445 [
446 'name' => 'French Postal Codes',
447 // Matches 5-digit French codes. It's more specific than a generic \d{5}
448 // by checking for valid department numbers (01-95) and Corsica (2A, 2B).
449 'regex' => '/\b(0[1-9]\d{3}|[1-8]\d{4}|9[0-5]\d{2}|2[AB]\d{3})\b/',
450 'token' => 'ZIP'
451 ],
452 [
453 'name' => 'Greek Postal Codes',
454 // Matches 5-digit Greek codes (e.g., 115 28). Note: This is a generic
455 // 5-digit pattern and may have false positives, but is standard for Greece.
456 'regex' => '/\b\d{3}\s?\d{2}\b/',
457 'token' => 'ZIP'
458 ],
459 [
460 'name' => 'US ZIP Codes',
461 // Matches 5-digit US ZIP codes and ZIP+4 format.
462 'regex' => '/\b\d{5}(?:-\d{4})?\b/',
463 'token' => 'ZIP'
464 ],
465 ];
466
467 foreach ($zipCodePatterns as $pattern) {
468 $text = preg_replace_callback(
469 $pattern['regex'],
474 function (array $m) use ($pattern) {
475 return $this->createToken($m[0], $pattern['token']);
476 },
477 $text
478 );
479 }
480
481 return $text;
482 }
483
491 public function unmask($jsonString)
492 {
493 if (empty($this->map)) {
494 return $jsonString;
495 }
496
497 // Handle case where AI might return token inside quotes or escaped
498 $search = array_keys($this->map);
499 $replace = array_values($this->map);
500
501 return str_replace($search, $replace, $jsonString);
502 }
503
511 private function maskCreditCardCallback(array $matches)
512 {
513 $potentialCc = $matches[0];
514 if ($this->passesLuhnCheck($potentialCc)) {
515 return $this->createToken($potentialCc, 'CC');
516 }
517
518 // If it doesn't pass the Luhn check, return the original string unmodified.
519 return $potentialCc;
520 }
521
528 private function passesLuhnCheck($number)
529 {
530 // Clean the string to contain only digits.
531 $digits = preg_replace('/\D/', '', $number);
532
533 // Check if the cleaned string is within a valid length range.
534 if (strlen($digits) < 13 || strlen($digits) > 19) {
535 return false;
536 }
537
538 // Perform the Luhn algorithm.
539 $sum = 0;
540 $isEvenDigit = false;
541
542 // Iterate from right to left
543 for ($i = strlen($digits) - 1; $i >= 0; $i--) {
544 $digit = (int) $digits[$i];
545
546 if ($isEvenDigit) {
547 $digit *= 2;
548 // If the result is two digits, sum them (or subtract 9)
549 if ($digit > 9) {
550 $digit -= 9;
551 }
552 }
553
554 $sum += $digit;
555 $isEvenDigit = !$isEvenDigit; // Flip the flag for the next digit
556 }
557
558 // The number is valid if the sum is a multiple of 10.
559 return ($sum % 10) === 0;
560 }
561
569 public function unmaskAiResponse($text)
570 {
571 if (empty($this->map)) {
572 return $text;
573 }
574
575 // Standard unmasking
576 $text = $this->unmask($text);
577
578 // Next, find and replace any tokens that were stripped by the AI.
579 // We iterate through our map and check for the stripped version of each token.
580 foreach ($this->map as $fullToken => $originalValue) {
581 // The stripped token is the full token without the brackets.
582 // e.g., '[[REF_1]]' becomes 'REF_1'
583 $strippedToken = substr($fullToken, 2, -2);
584
585 if (strpos($text, $strippedToken) !== false) {
586 // We found a stripped token in the text, so we replace it.
587 $text = str_replace($strippedToken, $originalValue, $text);
588 }
589 }
590
591 return $text;
592 }
593
599 private function startSession()
600 {
601 // One masking session per guard instance: mask() and maskNames() may
602 // be called several times for one request (query, context line), and
603 // every token they issue must survive in the same map until unmask.
604 if ($this->salt !== '') {
605 return;
606 }
607 $this->map = [];
608 $this->index = 0;
609 $this->salt = dol_substr(dol_hash(uniqid((string) mt_rand(), true), 'md5'), 0, 4);
610 }
611
619 private function createToken($value, $type)
620 {
621 // Reuse the token already issued for this value in this request, so
622 // the same entity reads as the same placeholder throughout the prompt.
623 $existing = array_search($value, $this->map, true);
624 if ($existing !== false) {
625 return (string) $existing;
626 }
627
628 $this->index++;
629 // Format: [[EMAIL_1a2b3c]]. The per-request salt keeps placeholders
630 // unpredictable (a provider cannot correlate "entity 1" across
631 // requests) and prevents collisions with literal [[TYPE_n]] text.
632 $token = "[[{$type}_{$this->index}{$this->salt}]]";
633 $this->map[$token] = $value;
634 return $token;
635 }
636
649 public function maskNames($text, $names)
650 {
651 $this->startSession();
652
653 $clean = array();
654 foreach ($names as $name) {
655 $name = trim((string) $name);
656 // Very short names would shred unrelated words.
657 if (dol_strlen($name) >= 4) {
658 $clean[] = $name;
659 }
660 }
661 if (empty($clean)) {
662 return $text;
663 }
664 $clean = array_unique($clean);
665 usort($clean, function (string $a, string $b) {
666 return dol_strlen($b) - dol_strlen($a);
667 });
668 foreach ($clean as $name) {
669 if (stripos($text, $name) === false) {
670 continue;
671 }
672 $token = $this->createToken($name, 'NAME');
673 $text = str_ireplace($name, $token, $text);
674 }
675
676 return $text;
677 }
678}
Class to manage privacy data masking and unmasking.
maskNames($text, $names)
Mask the names of the objects carried in this payload.
startSession()
Start the masking session for this guard instance, once.
mask($text)
Mask sensitive GDPR data in the query.
maskCreditCardCallback(array $matches)
Callback function for preg_replace_callback to mask credit cards.
unmaskAiResponse($text)
Unmasks a string from an AI response, handling cases where the AI might have stripped the [[ and ]] d...
passesLuhnCheck($number)
Validates a number string using the Luhn algorithm.
unmask($jsonString)
Restore real data from a masked string.
createToken($value, $type)
Create a unique token and store the original value in the map.
dol_strlen($string, $stringencoding='UTF-8')
Make a strlen call.
dol_substr($string, $start, $length=null, $stringencoding='', $trunconbytes=0)
Make a substring.
dol_hash($chain, $type='0', $nosalt=0, $mode=0)
Returns a hash (non reversible encryption) of a string.