001package eu.righettod;
002
003
004import com.auth0.jwt.interfaces.DecodedJWT;
005import org.apache.batik.anim.dom.SAXSVGDocumentFactory;
006import org.apache.batik.util.XMLResourceDescriptor;
007import org.apache.commons.csv.CSVFormat;
008import org.apache.commons.csv.CSVRecord;
009import org.apache.commons.imaging.ImageInfo;
010import org.apache.commons.imaging.Imaging;
011import org.apache.commons.imaging.common.ImageMetadata;
012import org.apache.commons.validator.routines.CreditCardValidator;
013import org.apache.commons.validator.routines.EmailValidator;
014import org.apache.commons.validator.routines.InetAddressValidator;
015import org.apache.pdfbox.Loader;
016import org.apache.pdfbox.pdmodel.PDDocument;
017import org.apache.pdfbox.pdmodel.PDDocumentCatalog;
018import org.apache.pdfbox.pdmodel.PDDocumentInformation;
019import org.apache.pdfbox.pdmodel.PDDocumentNameDictionary;
020import org.apache.pdfbox.pdmodel.common.PDMetadata;
021import org.apache.pdfbox.pdmodel.interactive.action.*;
022import org.apache.pdfbox.pdmodel.interactive.annotation.AnnotationFilter;
023import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation;
024import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationLink;
025import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm;
026import org.apache.poi.poifs.filesystem.DirectoryEntry;
027import org.apache.poi.poifs.filesystem.POIFSFileSystem;
028import org.apache.poi.poifs.macros.VBAMacroReader;
029import org.apache.tika.detect.DefaultDetector;
030import org.apache.tika.detect.Detector;
031import org.apache.tika.io.TemporaryResources;
032import org.apache.tika.io.TikaInputStream;
033import org.apache.tika.metadata.Metadata;
034import org.apache.tika.mime.MediaType;
035import org.apache.tika.mime.MimeTypes;
036import org.apache.tika.parser.ParseContext;
037import org.iban4j.IbanUtil;
038import org.owasp.html.HtmlPolicyBuilder;
039import org.owasp.html.PolicyFactory;
040import org.w3c.dom.Document;
041import org.w3c.dom.svg.SVGDocument;
042import org.xml.sax.EntityResolver;
043import org.xml.sax.InputSource;
044import org.xml.sax.SAXException;
045
046import javax.crypto.Mac;
047import javax.crypto.spec.SecretKeySpec;
048import javax.imageio.ImageIO;
049import javax.json.Json;
050import javax.json.JsonReader;
051import javax.xml.XMLConstants;
052import javax.xml.parsers.DocumentBuilder;
053import javax.xml.parsers.DocumentBuilderFactory;
054import javax.xml.parsers.ParserConfigurationException;
055import javax.xml.stream.XMLInputFactory;
056import javax.xml.stream.XMLStreamReader;
057import javax.xml.stream.events.XMLEvent;
058import javax.xml.validation.Schema;
059import javax.xml.validation.SchemaFactory;
060import java.awt.*;
061import java.awt.image.BufferedImage;
062import java.io.*;
063import java.net.*;
064import java.net.http.HttpClient;
065import java.net.http.HttpRequest;
066import java.net.http.HttpResponse;
067import java.nio.ByteBuffer;
068import java.nio.charset.Charset;
069import java.nio.charset.StandardCharsets;
070import java.nio.file.Files;
071import java.nio.file.Paths;
072import java.security.MessageDigest;
073import java.security.SecureRandom;
074import java.time.Duration;
075import java.time.LocalDate;
076import java.time.YearMonth;
077import java.time.ZoneId;
078import java.util.*;
079import java.util.List;
080import java.util.concurrent.*;
081import java.util.concurrent.atomic.AtomicInteger;
082import java.util.regex.Matcher;
083import java.util.regex.Pattern;
084import java.util.zip.GZIPInputStream;
085import java.util.zip.ZipEntry;
086import java.util.zip.ZipFile;
087
088/**
089 * Provides different utilities methods to apply processing from a security perspective.<br>
090 * These code snippet:
091 * <ul>
092 *     <li>Can be used, as "foundation", to customize the validation to the app context.</li>
093 *     <li>Were implemented in a way to facilitate adding or removal of validations depending on usage context.</li>
094 *     <li>Were centralized on one class to be able to enhance them across time as well as <a href="https://github.com/righettod/code-snippets-security-utils/issues">missing case/bug identification</a>.</li>
095 * </ul>
096 * <br>
097 * <a href="https://github.com/righettod/code-snippets-security-utils">GitHub repository</a>.<br><br>
098 * <a href="https://github.com/righettod/code-snippets-security-utils/blob/main/src/main/java/eu/righettod/SecurityUtils.java">Source code of the class</a>.
099 */
100public class SecurityUtils {
101    /**
102     * Default constructor: Not needed as the class only provides static methods.
103     */
104    private SecurityUtils() {
105    }
106
107    /**
108     * Apply a collection of validation to verify if a provided PIN code is considered weak (easy to guess) or none.<br>
109     * This method consider that format of the PIN code is [0-9]{6,}<br>
110     * Rule to consider a PIN code as weak:
111     * <ul>
112     * <li>Length is inferior to 6 positions.</li>
113     * <li>Contain only the same number or only a sequence of zero.</li>
114     * <li>Contain sequence of following incremental or decremental numbers.</li>
115     * </ul>
116     *
117     * @param pinCode PIN code to verify.
118     * @return True only if the PIN is considered as weak.
119     */
120    public static boolean isWeakPINCode(String pinCode) {
121        boolean isWeak = true;
122        //Length is inferior to 6 positions
123        //Use "Long.parseLong(pinCode)" to cause a NumberFormatException if the PIN is not a numeric one
124        //and to ensure that the PIN is not only a sequence of zero
125        if (pinCode != null && Long.parseLong(pinCode) > 0 && pinCode.trim().length() > 5) {
126            //Contain only the same number
127            String regex = String.format("^[%s]{%s}$", pinCode.charAt(0), pinCode.length());
128            if (!Pattern.matches(regex, pinCode)) {
129                //Contain sequence of following incremental or decremental numbers
130                char previousChar = 'X';
131                boolean containSequence = false;
132                for (char c : pinCode.toCharArray()) {
133                    if (previousChar != 'X') {
134                        int previousNbr = Integer.parseInt(String.valueOf(previousChar));
135                        int currentNbr = Integer.parseInt(String.valueOf(c));
136                        if (currentNbr == (previousNbr - 1) || currentNbr == (previousNbr + 1)) {
137                            containSequence = true;
138                            break;
139                        }
140                    }
141                    previousChar = c;
142                }
143                if (!containSequence) {
144                    isWeak = false;
145                }
146            }
147        }
148        return isWeak;
149    }
150
151    /**
152     * Apply a collection of validations on a Word 97-2003 (binary format) document file provided:
153     * <ul>
154     * <li>Real Microsoft Word 97-2003 document file.</li>
155     * <li>No VBA Macro.<br></li>
156     * <li>No embedded objects.</li>
157     * </ul>
158     *
159     * @param wordFilePath Filename of the Word document file to check.
160     * @return True only if the file pass all validations.
161     * @see "https://poi.apache.org/components/"
162     * @see "https://poi.apache.org/components/document/"
163     * @see "https://poi.apache.org/components/poifs/how-to.html"
164     * @see "https://poi.apache.org/components/poifs/embeded.html"
165     * @see "https://poi.apache.org/"
166     * @see "https://mvnrepository.com/artifact/org.apache.poi/poi"
167     */
168    public static boolean isWord972003DocumentSafe(String wordFilePath) {
169        boolean isSafe = false;
170        try {
171            File wordFile = new File(wordFilePath);
172            if (wordFile.exists() && wordFile.canRead() && wordFile.isFile()) {
173                //Step 1: Try to load the file, if its fail then it imply that is not a valid Word 97-2003 format file
174                try (POIFSFileSystem fs = new POIFSFileSystem(wordFile)) {
175                    //Step 2: Check if the document contains VBA macros, in our case is not allowed
176                    VBAMacroReader macroReader = new VBAMacroReader(fs);
177                    Map<String, String> macros = macroReader.readMacros();
178                    if (macros == null || macros.isEmpty()) {
179                        //Step 3: Check if the document contains any embedded objects, in our case is not allowed
180                        //From POI documentation:
181                        //Word normally stores embedded files in subdirectories of the ObjectPool directory, itself a subdirectory of the filesystem root.
182                        //Typically, these subdirectories and named starting with an underscore, followed by 10 numbers.
183                        final List<String> embeddedObjectFound = new ArrayList<>();
184                        DirectoryEntry root = fs.getRoot();
185                        if (root.getEntryCount() > 0) {
186                            root.iterator().forEachRemaining(entry -> {
187                                if ("ObjectPool".equalsIgnoreCase(entry.getName()) && entry instanceof DirectoryEntry) {
188                                    DirectoryEntry objPoolDirectory = (DirectoryEntry) entry;
189                                    if (objPoolDirectory.getEntryCount() > 0) {
190                                        objPoolDirectory.iterator().forEachRemaining(objPoolDirectoryEntry -> {
191                                            if (objPoolDirectoryEntry instanceof DirectoryEntry) {
192                                                DirectoryEntry objPoolDirectoryEntrySubDirectoryEntry = (DirectoryEntry) objPoolDirectoryEntry;
193                                                if (objPoolDirectoryEntrySubDirectoryEntry.getEntryCount() > 0) {
194                                                    objPoolDirectoryEntrySubDirectoryEntry.forEach(objPoolDirectoryEntrySubDirectoryEntryEntry -> {
195                                                        if (objPoolDirectoryEntrySubDirectoryEntryEntry.isDocumentEntry()) {
196                                                            embeddedObjectFound.add(objPoolDirectoryEntrySubDirectoryEntryEntry.getName());
197                                                        }
198                                                    });
199                                                }
200                                            }
201                                        });
202                                    }
203                                }
204                            });
205                        }
206                        isSafe = embeddedObjectFound.isEmpty();
207                    }
208                }
209            }
210        } catch (Exception e) {
211            isSafe = false;
212        }
213        return isSafe;
214    }
215
216    /**
217     * Ensure that an XML file does not contain any External Entity, DTD or XInclude instructions.
218     *
219     * @param xmlFilePath Filename of the XML file to check.
220     * @return True only if the file pass all validations.
221     * @see "https://portswigger.net/web-security/xxe"
222     * @see "https://cheatsheetseries.owasp.org/cheatsheets/XML_External_Entity_Prevention_Cheat_Sheet.html#java"
223     * @see "https://docs.oracle.com/en/java/javase/13/security/java-api-xml-processing-jaxp-security-guide.html#GUID-82F8C206-F2DF-4204-9544-F96155B1D258"
224     * @see "https://www.w3.org/TR/xinclude-11/"
225     * @see "https://en.wikipedia.org/wiki/XInclude"
226     */
227    public static boolean isXMLSafe(String xmlFilePath) {
228        boolean isSafe = false;
229        try {
230            File xmlFile = new File(xmlFilePath);
231            if (xmlFile.exists() && xmlFile.canRead() && xmlFile.isFile()) {
232                //Step 1a: Verify that the XML file content does not contain any XInclude instructions
233                boolean containXInclude = Files.readAllLines(xmlFile.toPath()).stream().anyMatch(line -> line.toLowerCase(Locale.ROOT).contains(":include "));
234                if (!containXInclude) {
235                    //Step 1b: Parse the XML file, if an exception occur than it's imply that the XML specified is not a valid ones
236                    //Create an XML document builder throwing Exception if a DOCTYPE instruction is present
237                    DocumentBuilderFactory dbfInstance = DocumentBuilderFactory.newInstance();
238                    dbfInstance.setFeature("http://apache.org/xml/features/disallow-doctype-decl", true);
239                    //Xerces 2 only
240                    //dbfInstance.setFeature("http://xerces.apache.org/xerces2-j/features.html#disallow-doctype-decl",true);
241                    dbfInstance.setXIncludeAware(false);
242                    DocumentBuilder builder = dbfInstance.newDocumentBuilder();
243                    //Parse the document
244                    Document doc = builder.parse(xmlFile);
245                    isSafe = (doc != null && doc.getDocumentElement() != null);
246                }
247            }
248        } catch (Exception e) {
249            isSafe = false;
250        }
251        return isSafe;
252    }
253
254
255    /**
256     * Extract all URL links from a PDF file provided.<br>
257     * This can be used to apply validation on a PDF against contained links.
258     *
259     * @param pdfFilePath pdfFilePath Filename of the PDF file to process.
260     * @return A List of URL objects that is empty if no links is found.
261     * @throws Exception If any error occurs during the processing of the PDF file.
262     * @see "https://www.gushiciku.cn/pl/21KQ"
263     * @see "https://pdfbox.apache.org/"
264     * @see "https://mvnrepository.com/artifact/org.apache.pdfbox/pdfbox"
265     */
266    public static List<URL> extractAllPDFLinks(String pdfFilePath) throws Exception {
267        final List<URL> links = new ArrayList<>();
268        File pdfFile = new File(pdfFilePath);
269        try (PDDocument document = Loader.loadPDF(pdfFile)) {
270            PDDocumentCatalog documentCatalog = document.getDocumentCatalog();
271            AnnotationFilter actionURIAnnotationFilter = new AnnotationFilter() {
272                @Override
273                public boolean accept(PDAnnotation annotation) {
274                    boolean keep = false;
275                    if (annotation instanceof PDAnnotationLink) {
276                        keep = (((PDAnnotationLink) annotation).getAction() instanceof PDActionURI);
277                    }
278                    return keep;
279                }
280            };
281            documentCatalog.getPages().forEach(page -> {
282                try {
283                    page.getAnnotations(actionURIAnnotationFilter).forEach(annotation -> {
284                        PDActionURI linkAnnotation = (PDActionURI) ((PDAnnotationLink) annotation).getAction();
285                        try {
286                            URL urlObj = new URL(linkAnnotation.getURI());
287                            if (!links.contains(urlObj)) {
288                                links.add(urlObj);
289                            }
290                        } catch (MalformedURLException e) {
291                            throw new RuntimeException(e);
292                        }
293                    });
294                } catch (Exception e) {
295                    throw new RuntimeException(e);
296                }
297            });
298        }
299        return links;
300    }
301
302    /**
303     * Apply a collection of validations on a PDF file provided:
304     * <ul>
305     * <li>Real PDF file.</li>
306     * <li>No attachments.</li>
307     * <li>No Javascript code.</li>
308     * <li>No links using action of type URI/Launch/RemoteGoTo/ImportData.</li>
309     * <li>No XFA forms in order to prevent exposure to XXE/SSRF like CVE-2025-54988.</li>
310     * </ul>
311     *
312     * @param pdfFilePath Filename of the PDF file to check.
313     * @return True only if the file pass all validations.
314     * @see "https://stackoverflow.com/a/36161267"
315     * @see "https://www.gushiciku.cn/pl/21KQ"
316     * @see "https://github.com/jonaslejon/malicious-pdf"
317     * @see "https://pdfbox.apache.org/"
318     * @see "https://mvnrepository.com/artifact/org.apache.pdfbox/pdfbox"
319     * @see "https://nvd.nist.gov/vuln/detail/CVE-2025-54988"
320     * @see "https://github.com/mgthuramoemyint/POC-CVE-2025-54988"
321     * @see "https://en.wikipedia.org/wiki/XFA"
322     */
323    public static boolean isPDFSafe(String pdfFilePath) {
324        boolean isSafe = false;
325        try {
326            File pdfFile = new File(pdfFilePath);
327            if (pdfFile.exists() && pdfFile.canRead() && pdfFile.isFile()) {
328                //Step 1: Try to load the file, if its fail then it imply that is not a valid PDF file
329                try (PDDocument document = Loader.loadPDF(pdfFile)) {
330                    //Step 2: Check if the file contains attached files, in our case is not allowed
331                    PDDocumentCatalog documentCatalog = document.getDocumentCatalog();
332                    PDDocumentNameDictionary namesDictionary = new PDDocumentNameDictionary(documentCatalog);
333                    if (namesDictionary.getEmbeddedFiles() == null) {
334                        //Step 3: Check if the file contains any XFA forms
335                        PDAcroForm acroForm = documentCatalog.getAcroForm();
336                        boolean hasForm = (acroForm != null && acroForm.getXFA() != null);
337                        if (!hasForm) {
338                            //Step 4: Check if the file contains Javascript code, in our case is not allowed
339                            if (namesDictionary.getJavaScript() == null) {
340                                //Step 5: Check if the file contains links using action of type URI/Launch/RemoteGoTo/ImportData, in our case is not allowed
341                                final List<Integer> notAllowedAnnotationCounterList = new ArrayList<>();
342                                AnnotationFilter notAllowedAnnotationFilter = new AnnotationFilter() {
343                                    @Override
344                                    public boolean accept(PDAnnotation annotation) {
345                                        boolean keep = false;
346                                        if (annotation instanceof PDAnnotationLink) {
347                                            PDAnnotationLink link = (PDAnnotationLink) annotation;
348                                            PDAction action = link.getAction();
349                                            if ((action instanceof PDActionURI) || (action instanceof PDActionLaunch) || (action instanceof PDActionRemoteGoTo) || (action instanceof PDActionImportData)) {
350                                                keep = true;
351                                            }
352                                        }
353                                        return keep;
354                                    }
355                                };
356                                documentCatalog.getPages().forEach(page -> {
357                                    try {
358                                        notAllowedAnnotationCounterList.add(page.getAnnotations(notAllowedAnnotationFilter).size());
359                                    } catch (IOException e) {
360                                        throw new RuntimeException(e);
361                                    }
362                                });
363                                if (notAllowedAnnotationCounterList.stream().reduce(0, Integer::sum) == 0) {
364                                    isSafe = true;
365                                }
366                            }
367                        }
368                    }
369                }
370            }
371        } catch (Exception e) {
372            isSafe = false;
373        }
374        return isSafe;
375    }
376
377    /**
378     * Remove as much as possible metadata from the provided PDF document object.
379     *
380     * @param document PDFBox PDF document object on which metadata must be removed.
381     * @see "https://gist.github.com/righettod/d7e07443c43d393a39de741a0d920069"
382     * @see "https://pdfbox.apache.org/"
383     * @see "https://mvnrepository.com/artifact/org.apache.pdfbox/pdfbox"
384     */
385    public static void clearPDFMetadata(PDDocument document) {
386        if (document != null) {
387            PDDocumentInformation infoEmpty = new PDDocumentInformation();
388            document.setDocumentInformation(infoEmpty);
389            PDMetadata newMetadataEmpty = new PDMetadata(document);
390            document.getDocumentCatalog().setMetadata(newMetadataEmpty);
391        }
392    }
393
394
395    /**
396     * Validate that the URL provided is really a relative URL.
397     *
398     * @param targetUrl URL to validate.
399     * @return True only if the file pass all validations.
400     * @see "https://portswigger.net/web-security/ssrf"
401     * @see "https://stackoverflow.com/q/6785442"
402     */
403    public static boolean isRelativeURL(String targetUrl) {
404        boolean isValid = false;
405        String work = targetUrl;
406        Pattern startingPrefix = Pattern.compile("^[/a-zA-Z0-9\\-_].*");
407        //Reject any URL no starting with a slash, letter, number, dash, or underscore
408        if (startingPrefix.matcher(work).find()) {
409            //Reject any URL encoded content and URL starting with a double slash
410            if (!work.startsWith("//") && !work.contains("%")) {
411                //Try to create en URI object
412                try {
413                    URI u = new URI(work);
414                    //Scheme must be null
415                    if (u.getScheme() == null) {
416                        isValid = (!u.isAbsolute());
417                    }
418                } catch (URISyntaxException mf) {
419                    isValid = false;
420                }
421            }
422        }
423
424        return isValid;
425    }
426
427    /**
428     * Apply a collection of validations on a ZIP file provided:
429     * <ul>
430     * <li>Real ZIP file.</li>
431     * <li>Contain less than a specified level of deepness.</li>
432     * <li>Do not contain Zip-Slip entry path.</li>
433     * </ul>
434     *
435     * @param zipFilePath       Filename of the ZIP file to check.
436     * @param maxLevelDeepness  Threshold of deepness above which a ZIP archive will be rejected.
437     * @param rejectArchiveFile Flag to specify if presence of any archive entry will cause the rejection of the ZIP file.
438     * @return True only if the file pass all validations.
439     * @see "https://rules.sonarsource.com/java/type/Security%20Hotspot/RSPEC-5042"
440     * @see "https://security.snyk.io/research/zip-slip-vulnerability"
441     * @see "https://en.wikipedia.org/wiki/Zip_bomb"
442     * @see "https://github.com/ptoomey3/evilarc"
443     * @see "https://github.com/abdulfatir/ZipBomb"
444     * @see "https://www.baeldung.com/cs/zip-bomb"
445     * @see "https://thesecurityvault.com/attacks-with-zip-files-and-mitigations/"
446     * @see "https://wiki.sei.cmu.edu/confluence/display/java/IDS04-J.+Safely+extract+files+from+ZipInputStream"
447     */
448    public static boolean isZIPSafe(String zipFilePath, int maxLevelDeepness, boolean rejectArchiveFile) {
449        List<String> archiveExtensions = Arrays.asList("zip", "tar", "7z", "gz", "jar", "phar", "bz2", "tgz");
450        boolean isSafe = false;
451        try {
452            File zipFile = new File(zipFilePath);
453            if (zipFile.exists() && zipFile.canRead() && zipFile.isFile() && maxLevelDeepness > 0) {
454                //Step 1: Try to load the file, if its fail then it imply that is not a valid ZIP file
455                try (ZipFile zipArch = new ZipFile(zipFile)) {
456                    //Step 2: Parse entries
457                    long deepness = 0;
458                    ZipEntry zipEntry;
459                    String entryExtension;
460                    String zipEntryName;
461                    boolean validationsFailed = false;
462                    Enumeration<? extends ZipEntry> entries = zipArch.entries();
463                    while (entries.hasMoreElements()) {
464                        zipEntry = entries.nextElement();
465                        zipEntryName = zipEntry.getName();
466                        entryExtension = zipEntryName.substring(zipEntryName.lastIndexOf(".") + 1).toLowerCase(Locale.ROOT).trim();
467                        //Step 2a: Check if the current entry is an archive file
468                        if (rejectArchiveFile && archiveExtensions.contains(entryExtension)) {
469                            validationsFailed = true;
470                            break;
471                        }
472                        //Step 2b: Check that level of deepness is inferior to the threshold specified
473                        if (zipEntryName.contains("/")) {
474                            //Determine deepness by inspecting the entry name.
475                            //Indeed, folder will be represented like this: folder/folder/folder/
476                            //So we can count the number of "/" to identify the deepness of the entry
477                            deepness = zipEntryName.chars().filter(ch -> ch == '/').count();
478                            if (deepness > maxLevelDeepness) {
479                                validationsFailed = true;
480                                break;
481                            }
482                        }
483                        //Step 2c: Check if any entries match pattern of zip slip payload
484                        if (zipEntryName.contains("..\\") || zipEntryName.contains("../")) {
485                            validationsFailed = true;
486                            break;
487                        }
488                    }
489                    if (!validationsFailed) {
490                        isSafe = true;
491                    }
492                }
493            }
494        } catch (Exception e) {
495            isSafe = false;
496        }
497        return isSafe;
498    }
499
500    /**
501     * Identify the mime type of the content specified (array of bytes).<br>
502     * Note that it cannot be fully trusted (see the tweet '1595824709186519041' referenced), so, additional validations are required.
503     *
504     * @param content The content as an array of bytes.
505     * @return The mime type in lower case or null if it cannot be identified.
506     * @see "https://twitter.com/righettod/status/1595824709186519041"
507     * @see "https://tika.apache.org/"
508     * @see "https://mvnrepository.com/artifact/org.apache.tika/tika-core"
509     * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types"
510     * @see "https://www.iana.org/assignments/media-types/media-types.xhtml"
511     */
512    public static String identifyMimeType(byte[] content) {
513        String mimeType = null;
514        if (content != null && content.length > 0) {
515            Detector detector = new DefaultDetector(MimeTypes.getDefaultMimeTypes());
516            Metadata metadata = new Metadata();
517            try {
518                try (TemporaryResources temporaryResources = new TemporaryResources(); TikaInputStream tikaInputStream = TikaInputStream.get(new ByteArrayInputStream(content), temporaryResources, metadata)) {
519                    MediaType mt = detector.detect(tikaInputStream, metadata, new ParseContext());
520                    if (mt != null) {
521                        mimeType = mt.toString().toLowerCase(Locale.ROOT);
522                    }
523                }
524            } catch (IOException ioe) {
525                mimeType = null;
526            }
527        }
528        return mimeType;
529    }
530
531    /**
532     * Apply a collection of validations on a string expected to be an public IP address:
533     * <ul>
534     * <li>Is a valid IP v4 or v6 address.</li>
535     * <li>Is public from an Internet perspective.</li>
536     * </ul>
537     * <br>
538     * <b>Note:</b> I often see missing such validation in the value read from HTTP request headers like "X-Forwarded-For" or "Forwarded".
539     * <br><br>
540     * <b>Note for IPv6:</b> I used documentation found so it is really experimental!
541     *
542     * @param ip String expected to be a valid IP address.
543     * @return True only if the string pass all validations.
544     * @see "https://commons.apache.org/proper/commons-validator/"
545     * @see "https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/InetAddressValidator.html"
546     * @see "https://cheatsheetseries.owasp.org/cheatsheets/Server_Side_Request_Forgery_Prevention_Cheat_Sheet.html"
547     * @see "https://cheatsheetseries.owasp.org/assets/Server_Side_Request_Forgery_Prevention_Cheat_Sheet_Orange_Tsai_Talk.pdf"
548     * @see "https://cheatsheetseries.owasp.org/assets/Server_Side_Request_Forgery_Prevention_Cheat_Sheet_SSRF_Bible.pdf"
549     * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/X-Forwarded-For"
550     * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/Forwarded"
551     * @see "https://ipcisco.com/lesson/ipv6-address/"
552     * @see "https://www.juniper.net/documentation/us/en/software/junos/interfaces-security-devices/topics/topic-map/security-interface-ipv4-ipv6-protocol.html"
553     * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.base/java/net/InetAddress.html#getByName(java.lang.String)"
554     * @see "https://www.arin.net/reference/research/statistics/address_filters/"
555     * @see "https://en.wikipedia.org/wiki/Multicast_address"
556     * @see "https://stackoverflow.com/a/5619409"
557     * @see "https://www.ripe.net/media/documents/ipv6-address-types.pdf"
558     * @see "https://www.iana.org/assignments/ipv6-unicast-address-assignments/ipv6-unicast-address-assignments.xhtml"
559     * @see "https://developer.android.com/reference/java/net/Inet6Address"
560     * @see "https://en.wikipedia.org/wiki/Unique_local_address"
561     */
562    public static boolean isPublicIPAddress(String ip) {
563        boolean isValid = false;
564        try {
565            //Quick validation on the string itself based on characters used to compose an IP v4/v6 address
566            if (Pattern.matches("[0-9a-fA-F:.]+", ip)) {
567                //If OK then use the dedicated InetAddressValidator from Apache Commons Validator
568                if (InetAddressValidator.getInstance().isValid(ip)) {
569                    //If OK then validate that is an public IP address
570                    //From Javadoc for "InetAddress.getByName": If a literal IP address is supplied, only the validity of the address format is checked.
571                    InetAddress addr = InetAddress.getByName(ip);
572                    isValid = (!addr.isAnyLocalAddress() && !addr.isLinkLocalAddress() && !addr.isLoopbackAddress() && !addr.isMulticastAddress() && !addr.isSiteLocalAddress());
573                    //If OK and the IP is an V6 one then make additional validation because the built-in Java API will let pass some V6 IP
574                    //For the prefix map, the start of the key indicates if the value is a regex or a string
575                    if (isValid && (addr instanceof Inet6Address)) {
576                        Map<String, String> prefixes = new HashMap<>();
577                        prefixes.put("REGEX_LOOPBACK", "^(0|:)+1$");
578                        prefixes.put("REGEX_UNIQUE-LOCAL-ADDRESSES", "^f(c|d)[a-f0-9]{2}:.*$");
579                        prefixes.put("STRING_LINK-LOCAL-ADDRESSES", "fe80:");
580                        prefixes.put("REGEX_TEREDO", "^2001:[0]*:.*$");
581                        prefixes.put("REGEX_BENCHMARKING", "^2001:[0]*2:.*$");
582                        prefixes.put("REGEX_ORCHID", "^2001:[0]*10:.*$");
583                        prefixes.put("STRING_DOCUMENTATION", "2001:db8:");
584                        prefixes.put("STRING_GLOBAL-UNICAST", "2000:");
585                        prefixes.put("REGEX_MULTICAST", "^ff[0-9]{2}:.*$");
586                        final List<Boolean> results = new ArrayList<>();
587                        final String ipLower = ip.trim().toLowerCase(Locale.ROOT);
588                        prefixes.forEach((addressType, expr) -> {
589                            String exprLower = expr.trim().toLowerCase();
590                            if (addressType.startsWith("STRING_")) {
591                                results.add(ipLower.startsWith(exprLower));
592                            } else {
593                                results.add(Pattern.matches(exprLower, ipLower));
594                            }
595                        });
596                        isValid = ((results.size() == prefixes.size()) && !results.contains(Boolean.TRUE));
597                    }
598                }
599            }
600        } catch (Exception e) {
601            isValid = false;
602        }
603        return isValid;
604    }
605
606    /**
607     * Compute a SHA256 hash from an input composed of a collection of strings.<br><br>
608     * This method take care to build the source string in a way to prevent this source string to be prone to abuse targeting the different parts composing it.<br><br>
609     * <p>
610     * Example of possible abuse without precautions applied during the hash calculation logic:<br>
611     * Hash of <code>SHA256("Hello", "My", "World!!!")</code> will be equals to the hash of <code>SHA256("Hell", "oMyW", "orld!!!")</code>.<br>
612     * </p>
613     * This method ensure that both hash above will be different.<br><br>
614     *
615     * <b>Note:</b> The character <code>|</code> is used, as separator, of every parts so a part is not allowed to contains this character.
616     *
617     * @param parts Ordered list of strings to use to build the input string for which the hash must be computed on. No null value is accepted on object composing the collection.
618     * @return The hash, as an array of bytes, to allow caller to convert it to the final representation wanted (HEX, Base64, etc.). If the collection passed is null or empty then the method return null.
619     * @throws Exception If any exception occurs
620     * @see "https://github.com/righettod/code-snippets-security-utils/issues/16"
621     * @see "https://pentesterlab.com/badges/codereview"
622     * @see "https://blog.trailofbits.com/2024/08/21/yolo-is-not-a-valid-hash-construction/"
623     * @see "https://www.nist.gov/publications/sha-3-derived-functions-cshake-kmac-tuplehash-and-parallelhash"
624     */
625    public static byte[] computeHashNoProneToAbuseOnParts(List<String> parts) throws Exception {
626        byte[] hash = null;
627        String separator = "|";
628        if (parts != null && !parts.isEmpty()) {
629            //Ensure that not part is null
630            if (parts.stream().anyMatch(Objects::isNull)) {
631                throw new IllegalArgumentException("No part must be null!");
632            }
633            //Ensure that the separator is absent from every part
634            if (parts.stream().anyMatch(part -> part.contains(separator))) {
635                throw new IllegalArgumentException(String.format("The character '%s', used as parts separator, must be absent from every parts!", separator));
636            }
637            MessageDigest digest = MessageDigest.getInstance("SHA-256");
638            final StringBuilder buffer = new StringBuilder(separator);
639            parts.forEach(p -> {
640                buffer.append(p).append(separator);
641            });
642            hash = digest.digest(buffer.toString().getBytes(StandardCharsets.UTF_8));
643        }
644        return hash;
645    }
646
647    /**
648     * Ensure that an XML file only uses DTD/XSD references (called System Identifier) present in the allowed list provided.<br><br>
649     * The code is based on the validation implemented into the OpenJDK 21, by the class <b><a href="https://github.com/openjdk/jdk/blob/jdk-21%2B35/src/java.prefs/share/classes/java/util/prefs/XmlSupport.java">java.util.prefs.XmlSupport</a></b>, in the method <b><a href="https://github.com/openjdk/jdk/blob/jdk-21%2B35/src/java.prefs/share/classes/java/util/prefs/XmlSupport.java#L240">loadPrefsDoc()</a></b>.<br><br>
650     * The method also ensure that no Public Identifier is used to prevent potential bypasses of the validations.
651     *
652     * @param xmlFilePath              Filename of the XML file to check.
653     * @param allowedSystemIdentifiers List of URL allowed for System Identifier specified for any XSD/DTD references.
654     * @return True only if the file pass all validations.
655     * @see "https://www.w3schools.com/xml/prop_documenttype_systemid.asp"
656     * @see "https://www.ibm.com/docs/en/integration-bus/9.0.0?topic=doctypedecl-xml-systemid"
657     * @see "https://www.liquid-technologies.com/Reference/Glossary/XML_DocType.html"
658     * @see "https://www.xml.com/pub/98/08/xmlqna0.html"
659     * @see "https://github.com/openjdk/jdk/blob/jdk-21%2B35/src/java.prefs/share/classes/java/util/prefs/XmlSupport.java#L397"
660     * @see "https://en.wikipedia.org/wiki/Formal_Public_Identifier"
661     */
662    public static boolean isXMLOnlyUseAllowedXSDorDTD(String xmlFilePath, final List<String> allowedSystemIdentifiers) {
663        boolean isSafe = false;
664        final String errorTemplate = "Non allowed %s ID detected!";
665        final String emptyFakeDTD = "<?xml version=\"1.0\" encoding=\"UTF-8\"?><!ELEMENT dummy EMPTY>";
666        final String emptyFakeXSD = "<xs:schema xmlns:xs=\"http://www.w3.org/2001/XMLSchema\"> <xs:element name=\"dummy\"/></xs:schema>";
667
668        if (allowedSystemIdentifiers == null || allowedSystemIdentifiers.isEmpty()) {
669            throw new IllegalArgumentException("At least one SID must be specified!");
670        }
671        File xmlFile = new File(xmlFilePath);
672        if (xmlFile.exists() && xmlFile.canRead() && xmlFile.isFile()) {
673            try {
674                EntityResolver resolverValidator = (publicId, systemId) -> {
675                    if (publicId != null) {
676                        throw new SAXException(String.format(errorTemplate, "PUBLIC"));
677                    }
678                    if (!allowedSystemIdentifiers.contains(systemId)) {
679                        throw new SAXException(String.format(errorTemplate, "SYSTEM"));
680                    }
681                    //If it is OK then return a empty DTD/XSD
682                    return new InputSource(new StringReader(systemId.toLowerCase().endsWith(".dtd") ? emptyFakeDTD : emptyFakeXSD));
683                };
684                DocumentBuilderFactory dbfInstance = DocumentBuilderFactory.newInstance();
685                dbfInstance.setIgnoringElementContentWhitespace(true);
686                dbfInstance.setXIncludeAware(false);
687                dbfInstance.setValidating(false);
688                dbfInstance.setCoalescing(true);
689                dbfInstance.setIgnoringComments(false);
690                DocumentBuilder builder = dbfInstance.newDocumentBuilder();
691                builder.setEntityResolver(resolverValidator);
692                Document doc = builder.parse(xmlFile);
693                isSafe = (doc != null);
694            } catch (SAXException | IOException | ParserConfigurationException e) {
695                isSafe = false;
696            }
697        }
698
699        return isSafe;
700    }
701
702    /**
703     * Apply a collection of validations on a EXCEL CSV file provided (file was expected to be opened in Microsoft EXCEL):
704     * <ul>
705     * <li>Real CSV file.</li>
706     * <li>Do not contains any payload related to a CSV injections.</li>
707     * </ul>
708     * Ensure that, if Apache Commons CSV does not find any record then, the file will be considered as NOT safe (prevent potential bypasses).<br><br>
709     * <b>Note:</b> Record delimiter used is the <code>,</code> (comma) character. See the Apache Commons CSV reference provided for EXCEL.<br>
710     *
711     * @param csvFilePath Filename of the CSV file to check.
712     * @return True only if the file pass all validations.
713     * @see "https://commons.apache.org/proper/commons-csv/"
714     * @see "https://commons.apache.org/proper/commons-csv/apidocs/org/apache/commons/csv/CSVFormat.html#EXCEL"
715     * @see "https://www.we45.com/post/your-excel-sheets-are-not-safe-heres-how-to-beat-csv-injection"
716     * @see "https://www.whiteoaksecurity.com/blog/2020-4-23-csv-injection-whats-the-risk/"
717     * @see "https://book.hacktricks.xyz/pentesting-web/formula-csv-doc-latex-ghostscript-injection"
718     * @see "https://owasp.org/www-community/attacks/CSV_Injection"
719     * @see "https://payatu.com/blog/csv-injection-basic-to-exploit/"
720     * @see "https://cwe.mitre.org/data/definitions/1236.html"
721     */
722    public static boolean isExcelCSVSafe(String csvFilePath) {
723        boolean isSafe;
724        final AtomicInteger recordCount = new AtomicInteger();
725        final List<Character> payloadDetectionCharacters = List.of('=', '+', '@', '-', '\r', '\t');
726
727        try {
728            final List<String> payloadsIdentified = new ArrayList<>();
729            try (Reader in = new FileReader(csvFilePath)) {
730                Iterable<CSVRecord> records = CSVFormat.EXCEL.parse(in);
731                records.forEach(record -> {
732                    record.forEach(recordValue -> {
733                        if (recordValue != null && !recordValue.trim().isEmpty() && payloadDetectionCharacters.contains(recordValue.trim().charAt(0))) {
734                            payloadsIdentified.add(recordValue);
735                        }
736                        recordCount.getAndIncrement();
737                    });
738                });
739            }
740            isSafe = (payloadsIdentified.isEmpty() && recordCount.get() > 0);
741        } catch (Exception e) {
742            isSafe = false;
743        }
744
745        return isSafe;
746    }
747
748    /**
749     * Provide a way to add an integrity marker (<a href="https://en.wikipedia.org/wiki/HMAC">HMAC</a>) to a serialized object serialized using the <a href="https://www.baeldung.com/java-serialization">java native system</a> (binary).<br>
750     * The goal is to provide <b>a temporary workaround</b> to try to prevent deserialization attacks and give time to move to a text-based serialization approach.
751     *
752     * @param processingModeType Define the mode of processing i.e. protect or validate. ({@link ProcessingModeType})
753     * @param input              When the processing mode is "protect" than the expected input (string) is a java serialized object encoded in Base64 otherwise (processing mode is "validate") expected input is the output of this method when the "protect" mode was used.
754     * @param secret             Secret to use to compute the SHA256 HMAC.
755     * @return A map with the following keys: <ul><li><b>PROCESSING_MODE</b>: Processing mode used to compute the result.</li><li><b>STATUS</b>: A boolean indicating if the processing was successful or not.</li><li><b>RESULT</b>: Always contains a string representing the protected serialized object in the format <code>[SERIALIZED_OBJECT_BASE64_ENCODED]:[SERIALIZED_OBJECT_HMAC_BASE64_ENCODED]</code>.</li></ul>
756     * @throws Exception If any exception occurs.
757     * @see "https://cheatsheetseries.owasp.org/cheatsheets/Deserialization_Cheat_Sheet.html"
758     * @see "https://owasp.org/www-project-top-ten/2017/A8_2017-Insecure_Deserialization"
759     * @see "https://portswigger.net/web-security/deserialization"
760     * @see "https://www.baeldung.com/java-serialization-approaches"
761     * @see "https://www.baeldung.com/java-serialization"
762     * @see "https://cryptobook.nakov.com/mac-and-key-derivation/hmac-and-key-derivation"
763     * @see "https://en.wikipedia.org/wiki/HMAC"
764     * @see "https://smattme.com/posts/how-to-generate-hmac-signature-in-java/"
765     */
766    public static Map<String, Object> ensureSerializedObjectIntegrity(ProcessingModeType processingModeType, String input, byte[] secret) throws Exception {
767        Map<String, Object> results;
768        String resultFormatTemplate = "%s:%s";
769        //Verify input provided to be consistent
770        if (processingModeType == null) {
771            throw new IllegalArgumentException("The processing mode is mandatory!");
772        }
773        if (input == null || input.trim().isEmpty()) {
774            throw new IllegalArgumentException("Input data is mandatory!");
775        }
776        if (secret == null || secret.length == 0) {
777            throw new IllegalArgumentException("The HMAC secret is mandatory!");
778        }
779        if (processingModeType.equals(ProcessingModeType.VALIDATE) && input.split(":").length != 2) {
780            throw new IllegalArgumentException("Input data provided is invalid for the processing mode specified!");
781        }
782        //Processing
783        Base64.Decoder b64Decoder = Base64.getDecoder();
784        Base64.Encoder b64Encoder = Base64.getEncoder();
785        String hmacAlgorithm = "HmacSHA256";
786        Mac mac = Mac.getInstance(hmacAlgorithm);
787        SecretKeySpec key = new SecretKeySpec(secret, hmacAlgorithm);
788        mac.init(key);
789        results = new HashMap<>();
790        results.put("PROCESSING_MODE", processingModeType.toString());
791        switch (processingModeType) {
792            case PROTECT -> {
793                byte[] objectBytes = b64Decoder.decode(input);
794                byte[] hmac = mac.doFinal(objectBytes);
795                String encodedHmac = b64Encoder.encodeToString(hmac);
796                results.put("STATUS", Boolean.TRUE);
797                results.put("RESULT", String.format(resultFormatTemplate, input, encodedHmac));
798            }
799            case VALIDATE -> {
800                String[] parts = input.split(":");
801                byte[] objectBytes = b64Decoder.decode(parts[0].trim());
802                byte[] hmacProvided = b64Decoder.decode(parts[1].trim());
803                byte[] hmacComputed = mac.doFinal(objectBytes);
804                String encodedHmacComputed = b64Encoder.encodeToString(hmacComputed);
805                Boolean hmacIsValid = Arrays.equals(hmacProvided, hmacComputed);
806                results.put("STATUS", hmacIsValid);
807                results.put("RESULT", String.format(resultFormatTemplate, parts[0].trim(), encodedHmacComputed));
808            }
809            default -> throw new IllegalArgumentException("Not supported processing mode!");
810        }
811        return results;
812    }
813
814    /**
815     * Apply a collection of validations on a JSON string provided:
816     * <ul>
817     * <li>Real JSON structure.</li>
818     * <li>Contain less than a specified number of deepness for nested objects or arrays.</li>
819     * <li>Contain less than a specified number of items in any arrays.</li>
820     * </ul>
821     * <br>
822     * <b>Note:</b> I decided to use a parsing approach using only string processing to prevent any StackOverFlow or OutOfMemory error that can be abused.<br><br>
823     * I used the following assumption:
824     * <ul>
825     *      <li>The character <code>{</code> identify the beginning of an object.</li>
826     *      <li>The character <code>}</code> identify the end of an object.</li>
827     *      <li>The character <code>[</code> identify the beginning of an array.</li>
828     *      <li>The character <code>]</code> identify the end of an array.</li>
829     *      <li>The character <code>"</code> identify the delimiter of a string.</li>
830     *      <li>The character sequence <code>\"</code> identify the escaping of an double quote.</li>
831     * </ul>
832     *
833     * @param json                  String containing the JSON data to validate.
834     * @param maxItemsByArraysCount Maximum number of items allowed in an array.
835     * @param maxDeepnessAllowed    Maximum number nested objects or arrays allowed.
836     * @return True only if the string pass all validations.
837     * @see "https://javaee.github.io/jsonp/"
838     * @see "https://community.f5.com/discussions/technicalforum/disable-buffer-overflow-in-json-parameters/124306"
839     * @see "https://github.com/InductiveComputerScience/pbJson/issues/2"
840     */
841    public static boolean isJSONSafe(String json, int maxItemsByArraysCount, int maxDeepnessAllowed) {
842        boolean isSafe = false;
843
844        try {
845            //Step 1: Analyse the JSON string
846            int currentDeepness = 0;
847            int currentArrayItemsCount = 0;
848            int maxDeepnessReached = 0;
849            int maxArrayItemsCountReached = 0;
850            boolean currentlyInArray = false;
851            boolean currentlyInString = false;
852            int currentNestedArrayLevel = 0;
853            String jsonEscapedDoubleQuote = "\\\"";//Escaped double quote must not be considered as a string delimiter
854            String work = json.replace(jsonEscapedDoubleQuote, "'");
855            for (char c : work.toCharArray()) {
856                switch (c) {
857                    case '{': {
858                        if (!currentlyInString) {
859                            currentDeepness++;
860                        }
861                        break;
862                    }
863                    case '}': {
864                        if (!currentlyInString) {
865                            currentDeepness--;
866                        }
867                        break;
868                    }
869                    case '[': {
870                        if (!currentlyInString) {
871                            currentDeepness++;
872                            if (currentlyInArray) {
873                                currentNestedArrayLevel++;
874                            }
875                            currentlyInArray = true;
876                        }
877                        break;
878                    }
879                    case ']': {
880                        if (!currentlyInString) {
881                            currentDeepness--;
882                            currentArrayItemsCount = 0;
883                            if (currentNestedArrayLevel > 0) {
884                                currentNestedArrayLevel--;
885                            }
886                            if (currentNestedArrayLevel == 0) {
887                                currentlyInArray = false;
888                            }
889                        }
890                        break;
891                    }
892                    case '"': {
893                        currentlyInString = !currentlyInString;
894                        break;
895                    }
896                    case ',': {
897                        if (!currentlyInString && currentlyInArray) {
898                            currentArrayItemsCount++;
899                        }
900                        break;
901                    }
902                }
903                if (currentDeepness > maxDeepnessReached) {
904                    maxDeepnessReached = currentDeepness;
905                }
906                if (currentArrayItemsCount > maxArrayItemsCountReached) {
907                    maxArrayItemsCountReached = currentArrayItemsCount;
908                }
909            }
910            //Step 2: Apply validation against the value specified as limits
911            isSafe = ((maxItemsByArraysCount > maxArrayItemsCountReached) && (maxDeepnessAllowed > maxDeepnessReached));
912
913            //Step 3: If the content is safe then ensure that it is valid JSON structure using the "Java API for JSON Processing" (JSR 374) parser reference implementation.
914            if (isSafe) {
915                JsonReader reader = Json.createReader(new StringReader(json));
916                isSafe = (reader.read() != null);
917            }
918
919        } catch (Exception e) {
920            isSafe = false;
921        }
922        return isSafe;
923    }
924
925    /**
926     * Apply a collection of validations on a image file provided:
927     * <ul>
928     * <li>Real image file.</li>
929     * <li>Its mime type is into the list of allowed mime types.</li>
930     * <li>Its metadata fields do not contains any characters related to a malicious payloads.</li>
931     * </ul>
932     * <br>
933     * <b>Important note:</b> This implementation is prone to bypass using the "<b>raw insertion</b>" method documented in the <a href="https://www.synacktiv.com/en/publications/persistent-php-payloads-in-pngs-how-to-inject-php-code-in-an-image-and-keep-it-there">blog post</a> from the Synacktiv team.
934     * To handle such case, it is recommended to resize the image to remove any non image-related content, see <a href="https://github.com/righettod/document-upload-protection/blob/master/src/main/java/eu/righettod/poc/sanitizer/ImageDocumentSanitizerImpl.java#L54">here</a> for an example.<br>
935     *
936     * @param imageFilePath         Filename of the image file to check.
937     * @param imageAllowedMimeTypes List of image mime types allowed.
938     * @return True only if the file pass all validations.
939     * @see "https://commons.apache.org/proper/commons-imaging/"
940     * @see "https://commons.apache.org/proper/commons-imaging/formatsupport.html"
941     * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Common_types"
942     * @see "https://www.iana.org/assignments/media-types/media-types.xhtml#image"
943     * @see "https://www.synacktiv.com/en/publications/persistent-php-payloads-in-pngs-how-to-inject-php-code-in-an-image-and-keep-it-there"
944     * @see "https://cheatsheetseries.owasp.org/cheatsheets/File_Upload_Cheat_Sheet.html"
945     * @see "https://github.com/righettod/document-upload-protection/blob/master/src/main/java/eu/righettod/poc/sanitizer/ImageDocumentSanitizerImpl.java"
946     * @see "https://exiftool.org/examples.html"
947     * @see "https://en.wikipedia.org/wiki/List_of_file_signatures"
948     * @see "https://hexed.it/"
949     * @see "https://github.com/sighook/pixload"
950     */
951    public static boolean isImageSafe(String imageFilePath, List<String> imageAllowedMimeTypes) {
952        boolean isSafe = false;
953        Pattern payloadDetectionRegex = Pattern.compile("[<>${}`]+", Pattern.CASE_INSENSITIVE);
954        try {
955            File imgFile = new File(imageFilePath);
956            if (imgFile.exists() && imgFile.canRead() && imgFile.isFile() && !imageAllowedMimeTypes.isEmpty()) {
957                final byte[] imgBytes = Files.readAllBytes(imgFile.toPath());
958                //Step 1: Check the mime type of the file against the allowed ones
959                ImageInfo imgInfo = Imaging.getImageInfo(imgBytes);
960                if (imageAllowedMimeTypes.contains(imgInfo.getMimeType())) {
961                    //Step 2: Load the image into an object using the Image API
962                    BufferedImage imgObject = Imaging.getBufferedImage(imgBytes);
963                    if (imgObject != null && imgObject.getWidth() > 0 && imgObject.getHeight() > 0) {
964                        //Step 3: Check the metadata if the image format support it - Highly experimental
965                        List<String> metadataWithPayloads = new ArrayList<>();
966                        final ImageMetadata imgMetadata = Imaging.getMetadata(imgBytes);
967                        if (imgMetadata != null) {
968                            imgMetadata.getItems().forEach(item -> {
969                                String metadata = item.toString();
970                                if (payloadDetectionRegex.matcher(metadata).find()) {
971                                    metadataWithPayloads.add(metadata);
972                                }
973                            });
974                        }
975                        isSafe = metadataWithPayloads.isEmpty();
976                    }
977                }
978            }
979        } catch (Exception e) {
980            isSafe = false;
981        }
982        return isSafe;
983    }
984
985    /**
986     * Rewrite the input file to remove any embedded files that is not embedded using a methods supported by the official format of the file.<br>
987     * Example: a file can be embedded by adding it to the end of the source file, see the reference provided for details.
988     *
989     * @param inputFilePath Filename of the file to clean up.
990     * @param inputFileType Type of the file provided.
991     * @return A array of bytes with the cleaned file.
992     * @throws IllegalArgumentException If an invalid parameter is passed
993     * @throws Exception                If any technical error during the cleaning processing
994     * @see "https://www.synacktiv.com/en/publications/persistent-php-payloads-in-pngs-how-to-inject-php-code-in-an-image-and-keep-it-there"
995     * @see "https://github.com/righettod/toolbox-pentest-web/tree/master/misc"
996     * @see "https://github.com/righettod/toolbox-pentest-web?tab=readme-ov-file#misc"
997     * @see "https://stackoverflow.com/a/13605411"
998     */
999    public static byte[] sanitizeFile(String inputFilePath, InputFileType inputFileType) throws Exception {
1000        ByteArrayOutputStream sanitizedContent = new ByteArrayOutputStream();
1001        File inputFile = new File(inputFilePath);
1002        if (!inputFile.exists() || !inputFile.canRead() || !inputFile.isFile()) {
1003            throw new IllegalArgumentException("Cannot read the content of the input file!");
1004        }
1005        switch (inputFileType) {
1006            case PDF -> {
1007                try (PDDocument document = Loader.loadPDF(inputFile)) {
1008                    document.save(sanitizedContent);
1009                }
1010            }
1011            case IMAGE -> {
1012                // Load the original image
1013                BufferedImage originalImage = ImageIO.read(inputFile);
1014                String originalFormat = identifyMimeType(Files.readAllBytes(inputFile.toPath())).split("/")[1].trim();
1015                // Check that image has been successfully loaded
1016                if (originalImage == null) {
1017                    throw new IOException("Cannot load the original image !");
1018                }
1019                // Get current Width and Height of the image
1020                int originalWidth = originalImage.getWidth(null);
1021                int originalHeight = originalImage.getHeight(null);
1022                // Resize the image by removing 1px on Width and Height
1023                Image resizedImage = originalImage.getScaledInstance(originalWidth - 1, originalHeight - 1, Image.SCALE_SMOOTH);
1024                // Resize the resized image by adding 1px on Width and Height - In fact set image to is initial size
1025                Image initialSizedImage = resizedImage.getScaledInstance(originalWidth, originalHeight, Image.SCALE_SMOOTH);
1026                // Save image to a bytes buffer
1027                int bufferedImageType = BufferedImage.TYPE_INT_ARGB;//By default use a format supporting transparency
1028                //Sometimes for BMP, the format detected is "bmp; format=compressed"
1029                if ("jpeg".equalsIgnoreCase(originalFormat) || "bmp".equalsIgnoreCase(originalFormat) || originalFormat.startsWith("bmp;")) {
1030                    bufferedImageType = BufferedImage.TYPE_INT_RGB;
1031                }
1032                BufferedImage sanitizedImage = new BufferedImage(initialSizedImage.getWidth(null), initialSizedImage.getHeight(null), bufferedImageType);
1033                Graphics2D drawer = sanitizedImage.createGraphics();
1034                drawer.drawImage(initialSizedImage, 0, 0, null);
1035                drawer.dispose();
1036                //Handle "bmp; format=compressed" case
1037                String formatToUse = originalFormat;
1038                if (formatToUse.startsWith("bmp;")) {
1039                    formatToUse = formatToUse.split(";")[0].trim();
1040                }
1041                ImageIO.write(sanitizedImage, formatToUse, sanitizedContent);
1042            }
1043            default -> throw new IllegalArgumentException("Type of file not supported !");
1044        }
1045        if (sanitizedContent.size() == 0) {
1046            throw new IOException("An error occur during the rewrite operation!");
1047        }
1048        return sanitizedContent.toByteArray();
1049    }
1050
1051    /**
1052     * Apply a collection of validations on a string expected to be an email address:
1053     * <ul>
1054     * <li>Is a valid email address, from a parser perspective, following RFCs on email addresses.</li>
1055     * <li>Is not using "Encoded-word" format.</li>
1056     * <li>Is not using comment format.</li>
1057     * <li>Is not using "Punycode" format.</li>
1058     * <li>Is not using UUCP style addresses.</li>
1059     * <li>Is not using address literals.</li>
1060     * <li>Is not using source routes.</li>
1061     * <li>Is not using the "percent hack".</li>
1062     * <li>Does not contain newline or carriage-return characters (CRLF injection prevention).</li>
1063     * <li>The domain part contains at least one dot (reject single-label domains such as localhost or internal hostnames).</li>
1064     * <li>The local part is not a quoted string (i.e. not wrapped in double quotes).</li>
1065     * <li>Respect the RFC 5321 length limits: local part ≤ 64 characters, domain ≤ 255 characters, total address ≤ 320 characters.</li>
1066     * </ul><br>
1067     * This is based on the research work from <a href="https://portswigger.net/research/gareth-heyes">Gareth Heyes</a> added in references (Portswigger).<br><br>
1068     *
1069     * <b>Note:</b> The notion of valid, here, is to take from a secure usage of the data perspective.
1070     *
1071     * @param addr String expected to be a valid email address.
1072     * @return True only if the string pass all validations.
1073     * @see "https://commons.apache.org/proper/commons-validator/"
1074     * @see "https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/EmailValidator.html"
1075     * @see "https://datatracker.ietf.org/doc/html/rfc2047#section-2"
1076     * @see "https://portswigger.net/research/splitting-the-email-atom"
1077     * @see "https://www.jochentopf.com/email/address.html"
1078     * @see "https://en.wikipedia.org/wiki/Email_address"
1079     */
1080    public static boolean isEmailAddress(String addr) {
1081        boolean isValid = false;
1082        String work = addr.toLowerCase(Locale.ROOT);
1083        Pattern encodedWordRegex = Pattern.compile("[=?]+", Pattern.CASE_INSENSITIVE);
1084        Pattern forbiddenCharacterRegex = Pattern.compile("[():!%\\[\\],;\"\n\r]+", Pattern.CASE_INSENSITIVE);
1085        try {
1086            //Start with the use of the dedicated EmailValidator from Apache Commons Validator
1087            if (EmailValidator.getInstance(true, true).isValid(work)) {
1088                //If OK then validate it does not contains "Encoded-word" patterns using an aggressive approach
1089                if (!encodedWordRegex.matcher(work).find()) {
1090                    //If OK then validate it does not contains punycode
1091                    if (!work.contains("xn--")) {
1092                        //If OK then validate it does not use:
1093                        // UUCP style addresses,
1094                        // Comment format,
1095                        // Address literals,
1096                        // Source routes,
1097                        // The percent hack.
1098                        if (!forbiddenCharacterRegex.matcher(work).find()) {
1099                            //If OK ensure that the domain part contains at least one dot
1100                            long arobaseCount = addr.chars().filter(c -> c == '@').count();
1101                            if (arobaseCount == 1) {
1102                                String[] parts = addr.split("@");
1103                                String localPart = parts[0];
1104                                String domainPart = parts[1];
1105                                if (domainPart.contains(".")) {
1106                                    //If OK the check the respect to the RFC 5321 length limits:
1107                                    // local part ≤ 64 characters, domain ≤ 255 characters, total address ≤ 320 characters.
1108                                    if (localPart.length() <= 64 && domainPart.length() <= 255 && addr.length() <= 320) {
1109                                        isValid = true;
1110                                    }
1111                                }
1112                            }
1113                        }
1114                    }
1115                }
1116
1117            }
1118        } catch (Exception e) {
1119            isValid = false;
1120        }
1121        return isValid;
1122    }
1123
1124    /**
1125     * The <a href="https://www.stet.eu/en/psd2/">PSD2 STET</a> specification require to use <a href="https://datatracker.ietf.org/doc/draft-cavage-http-signatures/">HTTP Signature</a>.
1126     * <br>
1127     * Section <b>3.5.1.2</b> of the document <a href="https://www.stet.eu/assets/files/PSD2/1-6-3/api-dsp2-stet-v1.6.3.1-part-1-framework.pdf">Documentation Framework</a> version <b>1.6.3</b>.
1128     * <br>
1129     * The problem is that, by design, the HTTP Signature specification is prone to blind SSRF.
1130     * <br>
1131     * URL example taken from the STET specification: <code>https://path.to/myQsealCertificate_714f8154ec259ac40b8a9786c9908488b2582b68b17e865fede4636d726b709f</code>.
1132     * <br>
1133     * The objective of this code is to try to decrease the "exploitability/interest" of this SSRF for an attacker.
1134     *
1135     * @param certificateUrl Url pointing to a Qualified Certificate (QSealC) encoded in PEM format and respecting the ETSI/TS119495 technical Specification .
1136     * @return TRUE only if the url point to a Qualified Certificate in PEM format.
1137     * @see "https://www.stet.eu/en/psd2/"
1138     * @see "https://www.stet.eu/assets/files/PSD2/1-6-3/api-dsp2-stet-v1.6.3.1-part-1-framework.pdf"
1139     * @see "https://datatracker.ietf.org/doc/draft-cavage-http-signatures/"
1140     * @see "https://datatracker.ietf.org/doc/rfc9421/"
1141     * @see "https://openjdk.org/groups/net/httpclient/intro.html"
1142     * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.net.http/java/net/http/package-summary.html"
1143     * @see "https://portswigger.net/web-security/ssrf"
1144     * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/Cache-Control"
1145     */
1146    public static boolean isPSD2StetSafeCertificateURL(String certificateUrl) {
1147        boolean isValid = false;
1148        long connectionTimeoutInSeconds = 10;
1149        String userAgent = "PSD2-STET-HTTPSignature-CertificateRequest";
1150        try {
1151            //1. Ensure that the URL end with the SHA-256 fingerprint encoded in HEX of the certificate like requested by STET
1152            if (certificateUrl != null && certificateUrl.lastIndexOf("_") != -1) {
1153                String digestPart = certificateUrl.substring(certificateUrl.lastIndexOf("_") + 1);
1154                if (Pattern.matches("^[0-9a-f]{64}$", digestPart)) {
1155                    //2. Ensure that the URL is a valid url by creating a instance of the class URI
1156                    URI uri = URI.create(certificateUrl);
1157                    //3. Require usage of HTTPS and reject any url containing query parameters
1158                    if ("https".equalsIgnoreCase(uri.getScheme()) && uri.getQuery() == null) {
1159                        //4. Perform a HTTP HEAD request in order to get the content type of the remote resource
1160                        //and limit the interest to use the SSRF because to pass the check the url need to:
1161                        //- Do not having any query parameters.
1162                        //- Use HTTPS protocol.
1163                        //- End with a string having the format "_[0-9a-f]{64}".
1164                        //- Trigger the malicious action that the attacker want but with a HTTP HEAD without any redirect and parameters.
1165                        HttpResponse<String> response;
1166                        try (HttpClient client = HttpClient.newBuilder().followRedirects(HttpClient.Redirect.NEVER).build()) {
1167                            HttpRequest request = HttpRequest.newBuilder().uri(uri).timeout(Duration.ofSeconds(connectionTimeoutInSeconds)).method("HEAD", HttpRequest.BodyPublishers.noBody()).header("User-Agent", userAgent)//To provide an hint to the target about the initiator of the request
1168                                    .header("Cache-Control", "no-store, max-age=0")//To prevent caching issues or abuses
1169                                    .build();
1170                            response = client.send(request, HttpResponse.BodyHandlers.ofString());
1171                            if (response.statusCode() == 200) {
1172                                //5. Ensure that the response content type is "text/plain"
1173                                Optional<String> contentType = response.headers().firstValue("Content-Type");
1174                                isValid = (contentType.isPresent() && contentType.get().trim().toLowerCase(Locale.ENGLISH).startsWith("text/plain"));
1175                            }
1176                        }
1177                    }
1178                }
1179            }
1180        } catch (Exception e) {
1181            isValid = false;
1182        }
1183        return isValid;
1184    }
1185
1186    /**
1187     * Perform sequential URL decoding operations against a URL encoded data until the data is not URL encoded anymore or if the specified threshold is reached.
1188     *
1189     * @param encodedData            URL encoded data.
1190     * @param decodingRoundThreshold Threshold above which decoding will fail.
1191     * @return The decoded data.
1192     * @throws SecurityException If the threshold is reached.
1193     * @see "https://en.wikipedia.org/wiki/Percent-encoding"
1194     * @see "https://owasp.org/www-community/Double_Encoding"
1195     * @see "https://portswigger.net/web-security/essential-skills/obfuscating-attacks-using-encodings"
1196     * @see "https://capec.mitre.org/data/definitions/120.html"
1197     */
1198    public static String applyURLDecoding(String encodedData, int decodingRoundThreshold) throws SecurityException {
1199        if (decodingRoundThreshold < 1) {
1200            throw new IllegalArgumentException("Threshold must be a positive number !");
1201        }
1202        if (encodedData == null) {
1203            throw new IllegalArgumentException("Data provided must not be null !");
1204        }
1205        Charset charset = StandardCharsets.UTF_8;
1206        int currentDecodingRound = 0;
1207        boolean isFinished = false;
1208        String currentRoundData = encodedData;
1209        String previousRoundData = encodedData;
1210        while (!isFinished) {
1211            if (currentDecodingRound > decodingRoundThreshold) {
1212                throw new SecurityException(String.format("Decoding round threshold of %s reached!", decodingRoundThreshold));
1213            }
1214            currentRoundData = URLDecoder.decode(currentRoundData, charset);
1215            isFinished = currentRoundData.equals(previousRoundData);
1216            previousRoundData = currentRoundData;
1217            currentDecodingRound++;
1218        }
1219        return currentRoundData;
1220    }
1221
1222    /**
1223     * Apply a collection of validations on a string expected to be an system file/folder path:
1224     * <ul>
1225     * <li>Does not contains path traversal payload.</li>
1226     * <li>The canonical path is equals to the absolute path.</li>
1227     * </ul><br>
1228     *
1229     * @param path String expected to be a valid system file/folder path.
1230     * @return True only if the string pass all validations.
1231     * @see "https://portswigger.net/web-security/file-path-traversal"
1232     * @see "https://learn.snyk.io/lesson/directory-traversal/"
1233     * @see "https://capec.mitre.org/data/definitions/126.html"
1234     * @see "https://owasp.org/www-community/attacks/Path_Traversal"
1235     */
1236    public static boolean isPathSafe(String path) {
1237        boolean isSafe = false;
1238        int decodingRoundThreshold = 3;
1239        try {
1240            if (path != null && !path.isEmpty()) {
1241                //URL decode the path if case of data coming from a web context
1242                String decodedPath = applyURLDecoding(path, decodingRoundThreshold);
1243                //Ensure that no path traversal expression is present
1244                if (!decodedPath.contains("..")) {
1245                    File f = new File(decodedPath);
1246                    String canonicalPath = f.getCanonicalPath();
1247                    String absolutePath = f.getAbsolutePath();
1248                    isSafe = canonicalPath.equals(absolutePath);
1249                }
1250            }
1251        } catch (Exception e) {
1252            isSafe = false;
1253        }
1254        return isSafe;
1255    }
1256
1257    /**
1258     * Identify if an XML contains any XML comments or have any XSL processing instructions.<br>
1259     * Stream reader based parsing is used to support large XML tree.
1260     *
1261     * @param xmlFilePath Filename of the XML file to check.
1262     * @return True only if XML comments or XSL processing instructions are identified.
1263     * @see "https://www.tutorialspoint.com/xml/xml_processing.htm"
1264     * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.xml/javax/xml/stream/XMLInputFactory.html"
1265     * @see "https://portswigger.net/kb/issues/00400700_xml-entity-expansion"
1266     * @see "https://www.w3.org/Style/styling-XML.en.html"
1267     */
1268    public static boolean isXMLHaveCommentsOrXSLProcessingInstructions(String xmlFilePath) {
1269        boolean itemsDetected = false;
1270        try {
1271            //Ensure that the parser will not be prone XML external entity (XXE) injection or XML entity expansion (XEE) attacks
1272            XMLInputFactory xmlInputFactory = XMLInputFactory.newFactory();
1273            xmlInputFactory.setProperty(XMLInputFactory.SUPPORT_DTD, false);
1274            xmlInputFactory.setProperty(XMLConstants.ACCESS_EXTERNAL_DTD, "");
1275            xmlInputFactory.setProperty(XMLInputFactory.IS_REPLACING_ENTITY_REFERENCES, false);
1276            xmlInputFactory.setProperty(XMLInputFactory.IS_SUPPORTING_EXTERNAL_ENTITIES, false);
1277
1278            //Parse file
1279            try (FileInputStream fis = new FileInputStream(xmlFilePath)) {
1280                XMLStreamReader reader = xmlInputFactory.createXMLStreamReader(fis);
1281                int eventType;
1282                while (reader.hasNext() && !itemsDetected) {
1283                    eventType = reader.next();
1284                    if (eventType == XMLEvent.COMMENT) {
1285                        itemsDetected = true;
1286                    } else if (eventType == XMLEvent.PROCESSING_INSTRUCTION && "xml-stylesheet".equalsIgnoreCase(reader.getPITarget())) {
1287                        itemsDetected = true;
1288                    }
1289                }
1290            }
1291        } catch (Exception e) {
1292            //In case of error then assume that the check failed
1293            itemsDetected = true;
1294        }
1295        return itemsDetected;
1296    }
1297
1298
1299    /**
1300     * Perform a set of additional validations against a JWT token:
1301     * <ul>
1302     *     <li>Do not use the <b>NONE</b> signature algorithm.</li>
1303     *     <li>Have a <a href="https://www.iana.org/assignments/jwt/jwt.xhtml">EXP claim</a> defined.</li>
1304     *     <li>The token identifier (<a href="https://www.iana.org/assignments/jwt/jwt.xhtml">JTI claim</a>) is NOT part of the list of revoked token.</li>
1305     *     <li>Match the expected type of token: ACCESS or ID or REFRESH.</li>
1306     * </ul>
1307     *
1308     * @param token               JWT token for which <b>signature was already validated</b> and on which a set of additional validations will be applied.
1309     * @param expectedTokenType   The type of expected token using the enumeration provided.
1310     * @param revokedTokenJTIList A list of token identifier (<b>JTI</b> claim) referring to tokens that were revoked and to which the JTI claim of the token will be compared to.
1311     * @return True only the token pass all the validations.
1312     * @see "https://www.iana.org/assignments/jwt/jwt.xhtml"
1313     * @see "https://auth0.com/docs/secure/tokens/access-tokens"
1314     * @see "https://auth0.com/docs/secure/tokens/id-tokens"
1315     * @see "https://auth0.com/docs/secure/tokens/refresh-tokens"
1316     * @see "https://auth0.com/blog/id-token-access-token-what-is-the-difference/"
1317     * @see "https://jwt.io/libraries?language=Java"
1318     * @see "https://pentesterlab.com/blog/secure-jwt-library-design"
1319     * @see "https://github.com/auth0/java-jwt"
1320     */
1321    public static boolean applyJWTExtraValidation(DecodedJWT token, TokenType expectedTokenType, List<String> revokedTokenJTIList) {
1322        boolean isValid = false;
1323        TokenType tokenType;
1324        try {
1325            if (!"none".equalsIgnoreCase(token.getAlgorithm().trim())) {
1326                if (!token.getClaim("exp").isMissing() && token.getExpiresAt() != null) {
1327                    String jti = token.getId();
1328                    if (jti != null && !jti.trim().isEmpty()) {
1329                        boolean jtiIsRevoked = revokedTokenJTIList.stream().anyMatch(jti::equalsIgnoreCase);
1330                        if (!jtiIsRevoked) {
1331                            //Determine the token type based on the presence of specifics claims
1332                            if (!token.getClaim("scope").isMissing()) {
1333                                tokenType = TokenType.ACCESS;
1334                            } else if (!token.getClaim("name").isMissing() || !token.getClaim("email").isMissing()) {
1335                                tokenType = TokenType.ID;
1336                            } else {
1337                                tokenType = TokenType.REFRESH;
1338                            }
1339                            isValid = (tokenType.equals(expectedTokenType));
1340                        }
1341                    }
1342                }
1343            }
1344
1345        } catch (Exception e) {
1346            //In case of error then assume that the check failed
1347            isValid = false;
1348        }
1349        return isValid;
1350    }
1351
1352    /**
1353     * Apply a validations on a regular expression to ensure that is not prone to the ReDOS attack.
1354     * <br>If your technology is supported by <a href="https://github.com/doyensec/regexploit">regexploit</a> then <b>use it instead of this method!</b>
1355     * <br>Indeed, the <a href="https://www.doyensec.com/">Doyensec</a> team has made an intensive and amazing work on this topic and created this effective tool.
1356     *
1357     * @param regex                       String expected to be a valid regular expression (regex).
1358     * @param data                        Test data on which the regular expression is executed for the test.
1359     * @param maximumRunningTimeInSeconds Optional parameter to specify a number of seconds above which a regex execution time is considered as not safe (default to 4 seconds when not specified).
1360     * @return True only if the string pass all validations.
1361     * @see "https://github.blog/security/how-to-fix-a-redos/"
1362     * @see "https://learn.snyk.io/lesson/redos"
1363     * @see "https://rules.sonarsource.com/java/RSPEC-2631/"
1364     * @see "https://github.com/doyensec/regexploit"
1365     * @see "https://github.com/makenowjust-labs/recheck"
1366     * @see "https://github.com/tjenkinson/redos-detector"
1367     * @see "https://wiki.owasp.org/images/2/23/OWASP_IL_2009_ReDoS.pdf"
1368     * @see "https://owasp.org/www-community/attacks/Regular_expression_Denial_of_Service_-_ReDoS"
1369     */
1370    public static boolean isRegexSafe(String regex, String data, Optional<Integer> maximumRunningTimeInSeconds) {
1371        Objects.requireNonNull(maximumRunningTimeInSeconds, "Use 'Optional.empty()' to leverage the default value.");
1372        Objects.requireNonNull(data, "A sample data is needed to perform the test.");
1373        Objects.requireNonNull(regex, "A regular expression is needed to perform the test.");
1374        boolean isSafe = false;
1375        int executionTimeout = maximumRunningTimeInSeconds.orElse(4);
1376        ExecutorService executor = Executors.newSingleThreadExecutor();
1377        try {
1378            Callable<Boolean> task = () -> {
1379                Pattern pattern = Pattern.compile(regex);
1380                return pattern.matcher(data).matches();
1381            };
1382            List<Future<Boolean>> tasks = executor.invokeAll(List.of(task), executionTimeout, TimeUnit.SECONDS);
1383            if (!tasks.getFirst().isCancelled()) {
1384                isSafe = true;
1385            }
1386        } catch (Exception e) {
1387            isSafe = false;
1388        } finally {
1389            executor.shutdownNow();
1390        }
1391        return isSafe;
1392    }
1393
1394    /**
1395     * Compute a UUID version 7 without using any external dependency.<br><br>
1396     * <b>Below are my personal point of view and perhaps I'm totally wrong!</b>
1397     * <br><br>
1398     * Why such method?
1399     * <ul>
1400     * <li>Java inferior or equals to 21 does not supports natively the generation of an UUID version 7.</li>
1401     * <li>Import a library just to generate such value is overkill for me.</li>
1402     * <li>Library that I have found, generating such version of an UUID, are not provided by entities commonly used in the java world, such as the SPRING framework provider.</li>
1403     * </ul>
1404     * <br>
1405     * <b>Full credits for this implementation goes to the authors and contributors of the <a href="https://github.com/nalgeon/uuidv7">UUIDv7</a> project.</b>
1406     * <br><br>
1407     * Below are the java libraries that I have found but, for which, I do not trust enough the provider to use them directly:
1408     * <ul>
1409     *     <li><a href="https://github.com/cowtowncoder/java-uuid-generator">java-uuid-generator</a></li>
1410     *     <li><a href="https://github.com/f4b6a3/uuid-creator">uuid-creator</a></li>
1411     * </ul>
1412     *
1413     * @return A UUID object representing the UUID v7.
1414     * @see "https://uuid7.com/"
1415     * @see "https://antonz.org/uuidv7/"
1416     * @see "https://mccue.dev/pages/3-11-25-life-altering-postgresql-patterns"
1417     * @see "https://www.ietf.org/archive/id/draft-peabody-dispatch-new-uuid-format-04.html#name-uuid-version-7"
1418     * @see "https://www.baeldung.com/java-generating-time-based-uuids"
1419     * @see "https://en.wikipedia.org/wiki/Universally_unique_identifier"
1420     * @see "https://buildkite.com/resources/blog/goodbye-integers-hello-uuids/"
1421     */
1422    public static UUID computeUUIDv7() {
1423        SecureRandom secureRandom = new SecureRandom();
1424        // Generate truly random bytes
1425        byte[] value = new byte[16];
1426        secureRandom.nextBytes(value);
1427        // Get current timestamp in milliseconds
1428        ByteBuffer timestamp = ByteBuffer.allocate(Long.BYTES);
1429        timestamp.putLong(System.currentTimeMillis());
1430        // Create the TIMESTAMP part of the UUID
1431        System.arraycopy(timestamp.array(), 2, value, 0, 6);
1432        // Create the VERSION and the VARIANT parts of the UUID
1433        value[6] = (byte) ((value[6] & 0x0F) | 0x70);
1434        value[8] = (byte) ((value[8] & 0x3F) | 0x80);
1435        //Create the HIGH and LOW parts of the UUID
1436        ByteBuffer buf = ByteBuffer.wrap(value);
1437        long high = buf.getLong();
1438        long low = buf.getLong();
1439        //Create and return the UUID object
1440        UUID uuidv7 = new UUID(high, low);
1441        return uuidv7;
1442    }
1443
1444    /**
1445     * Ensure that an XSD file does not contain any include/import/redefine instruction (prevent exposure to SSRF).
1446     *
1447     * @param xsdFilePath Filename of the XSD file to check.
1448     * @return True only if the file pass all validations.
1449     * @see "https://portswigger.net/web-security/ssrf"
1450     * @see "https://www.w3schools.com/Xml/el_import.asp"
1451     * @see "https://www.w3schools.com/xml/el_include.asp"
1452     * @see "https://www.linkedin.com/posts/righettod_appsec-appsecurity-java-activity-7344048434326188053-6Ru9"
1453     * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.xml/javax/xml/validation/SchemaFactory.html#setProperty(java.lang.String,java.lang.Object)"
1454     */
1455    public static boolean isXSDSafe(String xsdFilePath) {
1456        boolean isSafe = false;
1457        try {
1458            File xsdFile = new File(xsdFilePath);
1459            if (xsdFile.exists() && xsdFile.canRead() && xsdFile.isFile()) {
1460                //Parse the XSD file, if an exception occur then it's imply that the XSD specified is not a valid ones
1461                //Create an schema factory throwing Exception if a external schema is specified
1462                SchemaFactory schemaFactory = SchemaFactory.newDefaultInstance();
1463                schemaFactory.setProperty(XMLConstants.ACCESS_EXTERNAL_DTD, "");
1464                schemaFactory.setProperty(XMLConstants.ACCESS_EXTERNAL_SCHEMA, "");
1465                //Parse the schema
1466                Schema schema = schemaFactory.newSchema(xsdFile);
1467                isSafe = (schema != null);
1468            }
1469        } catch (Exception e) {
1470            isSafe = false;
1471        }
1472        return isSafe;
1473    }
1474
1475
1476    /**
1477     * Extract all sensitive information from a string provided.<br>
1478     * This can be used to identify any sensitive information into a <a href="https://cwe.mitre.org/data/definitions/532.html">message expected to be written in a log</a> and then replace every sensitive values by an obfuscated ones.<br><br>
1479     * For the luxembourg national identification number, this method focus on detecting identifiers for a physical entity (people) and not a moral one (company).<br><br>
1480     * I delegated the validation of the IBAN to a dedicated library (<a href="https://github.com/arturmkrtchyan/iban4j">iban4j</a>) to not "reinvent the wheel" and then introduce buggy validation myself. I used <b>iban4j</b> over the <b><a href="https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/IBANValidator.html">IBANValidator</a></b> class from the <a href="https://commons.apache.org/proper/commons-validator/"><b>Apache Commons Validator</b></a> library because <b>iban4j</b> perform a full official IBAN specification validation so its reduce risks of false-positives by ensuring that an IBAN detected is a real IBAN.<br><br>
1481     * Same thing and reason regarding the validation of the bank card PAN using the  class <a href="https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/CreditCardValidator.html">CreditCardValidator</a> from the <b>Apache Commons Validator</b> library.
1482     *
1483     * @param content String in which sensitive information must be searched.
1484     * @return A map with the collection of identified sensitive information gathered by sensitive information type. If nothing is found then the map is empty. A type of sensitive information is only present if there is at least one item found. A set is used to not store duplicates occurrence of the same sensitive information.
1485     * @throws Exception If any error occurs during the processing.
1486     * @see "https://guichet.public.lu/en/citoyens/citoyennete/registre-national/identification/demande-numero-rnpp.html"
1487     * @see "https://cnpd.public.lu/fr/decisions-avis/2009/identifiant-unique.html"
1488     * @see "https://cnpd.public.lu/content/dam/cnpd/fr/decisions-avis/2009/identifiant-unique/48_2009.pdf"
1489     * @see "https://en.wikipedia.org/wiki/International_Bank_Account_Number"
1490     * @see "https://www.iban.com/structure"
1491     * @see "https://github.com/arturmkrtchyan/iban4j"
1492     * @see "https://cwe.mitre.org/data/definitions/532.html"
1493     * @see "https://www.baeldung.com/logback-mask-sensitive-data"
1494     * @see "https://en.wikipedia.org/wiki/Payment_card_number"
1495     * @see "https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/CreditCardValidator.html"
1496     * @see "https://commons.apache.org/proper/commons-validator/"
1497     */
1498    public static Map<SensitiveInformationType, Set<String>> extractAllSensitiveInformation(String content) throws Exception {
1499        CreditCardValidator creditCardValidator = CreditCardValidator.genericCreditCardValidator();
1500        Pattern nationalIdentifierRegex = Pattern.compile("([0-9]{13})");
1501        Pattern ibanNonHumanFormattedRegex = Pattern.compile("([A-Z]{2}[0-9]{2}[A-Z0-9]{11,30})", Pattern.CASE_INSENSITIVE);
1502        Pattern ibanHumanFormattedRegex = Pattern.compile("([A-Z]{2}[0-9]{2}(?:\\s[A-Z0-9]{4}){2,7}\\s[A-Z0-9]{1,4})", Pattern.CASE_INSENSITIVE);
1503        Pattern panRegex = Pattern.compile("((?:\\d[ -]*?){13,19})");
1504        Map<SensitiveInformationType, Set<String>> data = new HashMap<>();
1505        data.put(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER, new HashSet<>());
1506        data.put(SensitiveInformationType.IBAN, new HashSet<>());
1507        data.put(SensitiveInformationType.BANK_CARD_PAN, new HashSet<>());
1508
1509        if (content != null && !content.isBlank()) {
1510            /* Step 1: Search for LU national identifier */
1511            //A national identifier have the following structure: [BIRTHDATE_YEAR_YYYY][BIRTHDATE_MONTH_MM][BIRTHDATE_DAY_DD][FIVE_INTEGER]
1512            //Define minimal and maximal birth year base on current year
1513            //Assume people live less than 120 years
1514            int maxBirthYear = LocalDate.now(ZoneId.of("Europe/Luxembourg")).getYear();
1515            int minBirthYear = maxBirthYear - 120;
1516            Matcher matcher = nationalIdentifierRegex.matcher(content);
1517            String nationalIdentierFull;
1518            int nationalIdentierYear, nationalIdentierMonth, nationalIdentierDay;
1519            while (matcher.find()) {
1520                nationalIdentierFull = matcher.group(1);
1521                //Check that the string is a valid national identifier and if yes then add it
1522                nationalIdentierYear = Integer.parseInt(nationalIdentierFull.substring(0, 4));
1523                nationalIdentierMonth = Integer.parseInt(nationalIdentierFull.substring(4, 6));
1524                nationalIdentierDay = Integer.parseInt(nationalIdentierFull.substring(6, 8));
1525                if (nationalIdentierYear >= minBirthYear && nationalIdentierYear <= maxBirthYear) {
1526                    if (nationalIdentierMonth >= 1 && nationalIdentierMonth <= 12) {
1527                        if (YearMonth.of(nationalIdentierYear, nationalIdentierMonth).isValidDay(nationalIdentierDay)) {
1528                            data.get(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER).add(nationalIdentierFull);
1529                        }
1530                    }
1531                }
1532            }
1533
1534            /* Step 2a: Search for IBAN that are non human formatted */
1535            matcher = ibanNonHumanFormattedRegex.matcher(content);
1536            String iban, ibanUpperCased;
1537            while (matcher.find()) {
1538                iban = matcher.group(1);
1539                ibanUpperCased = iban.toUpperCase(Locale.ROOT);
1540                //Check that the string is a valid IBAN and if yes then add it
1541                if (IbanUtil.isValid(ibanUpperCased)) {
1542                    data.get(SensitiveInformationType.IBAN).add(iban);
1543                }
1544            }
1545
1546            /* Step 2b: Search for IBAN that are human formatted */
1547            matcher = ibanHumanFormattedRegex.matcher(content);
1548            String ibanUpperCasedNoSpace;
1549            while (matcher.find()) {
1550                iban = matcher.group(1);
1551                ibanUpperCasedNoSpace = iban.toUpperCase(Locale.ROOT).replace(" ", "");
1552                //Check that the string is a valid IBAN and if yes then add it
1553                if (IbanUtil.isValid(ibanUpperCasedNoSpace)) {
1554                    data.get(SensitiveInformationType.IBAN).add(iban);
1555                }
1556            }
1557
1558            /* Step 3: Search for bank card PAN */
1559            matcher = panRegex.matcher(content);
1560            String pan, panNoSeparator;
1561            while (matcher.find()) {
1562                pan = matcher.group(1);
1563                panNoSeparator = pan.toUpperCase(Locale.ROOT).replace(" ", "").replace("-", "");
1564                //Check that the string is a valid PAN and if yes then add it
1565                if (creditCardValidator.isValid(panNoSeparator)) {
1566                    data.get(SensitiveInformationType.BANK_CARD_PAN).add(pan);
1567                }
1568            }
1569
1570        }
1571
1572        //Cleanup if a set is empty
1573        if (data.get(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER).isEmpty()) {
1574            data.remove(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER);
1575        }
1576        if (data.get(SensitiveInformationType.IBAN).isEmpty()) {
1577            data.remove(SensitiveInformationType.IBAN);
1578        }
1579        if (data.get(SensitiveInformationType.BANK_CARD_PAN).isEmpty()) {
1580            data.remove(SensitiveInformationType.BANK_CARD_PAN);
1581        }
1582
1583        return data;
1584    }
1585
1586    /**
1587     * Apply a collection of validations on a bytes array provided representing GZIP compressed data:
1588     * <ul>
1589     * <li>Are valid GZIP compressed data.</li>
1590     * <li>The number of bytes once decompressed is under the specified limit.</li>
1591     * </ul>
1592     * <br><b>Note:</b> The value <code>Integer.MAX_VALUE - 8</code> was chosen because during my tests on Java 25 (JDK 64 bits on Windows 11 Pro), it was possible to decompress such amount of data with the default JVM settings without causing an <a href="https://docs.oracle.com/en/java/javase/25/docs/api//java.base/java/lang/OutOfMemoryError.html">Out Of Memory error</a>.
1593     *
1594     * @param compressedBytes                    Array of bytes containing the GZIP compressed data to check.
1595     * @param maxCountOfDecompressedBytesAllowed Maximum number of decompressed bytes allowed. Default to 10 MB if the specified value is inferior to 1 or superior to Integer.MAX_VALUE - 8.
1596     * @return True only if the file pass all validations.
1597     * @see "https://en.wikipedia.org/wiki/Gzip"
1598     * @see "https://www.rapid7.com/db/modules/auxiliary/dos/http/gzip_bomb_dos/"
1599     */
1600    public static boolean isGZIPCompressedDataSafe(byte[] compressedBytes, long maxCountOfDecompressedBytesAllowed) {
1601        boolean isSafe = false;
1602
1603        try {
1604            long limit = maxCountOfDecompressedBytesAllowed;
1605            long totalRead = 0L;
1606            byte[] buffer = new byte[8 * 1024];
1607            int read;
1608            if (limit < 1 || limit > (Integer.MAX_VALUE - 8)) {
1609                limit = 10_000_000;
1610            }
1611            try (ByteArrayInputStream bis = new ByteArrayInputStream(compressedBytes); GZIPInputStream gzipInputStream = new GZIPInputStream(new BufferedInputStream(bis))) {
1612                while ((read = gzipInputStream.read(buffer)) != -1) {
1613                    totalRead += read;
1614                    if (totalRead > limit) {
1615                        throw new Exception();
1616                    }
1617                }
1618            }
1619            isSafe = true;
1620        } catch (Exception e) {
1621            isSafe = false;
1622        }
1623
1624        return isSafe;
1625    }
1626
1627    /**
1628     * Process a string, intended to be written in a log, to remove as much as possible information that can lead to an exposure to a log injection vulnerability.<br><br>
1629     * <b>Log injection</b> is also called <b>log forging</b>.<br><br>
1630     * The following information are removed:
1631     * <ul>
1632     *     <li>Characters: Carriage Return (CR), Linefeed (LF) and Tabulation (TAB).</li>
1633     *     <li>Characters: Unicode LINE SEPARATOR and Unicode PARAGRAPH SEPARATOR.</li>
1634     *     <li>Characters: CSI sequences and bare ESC.</li>
1635     *     <li>Leading and trailing spaces.</li>
1636     *     <li>Any HTML tags.</li>
1637     * </ul><br>
1638     * A parameter is also used to limit the maximum length of the sanitized message.
1639     * To remove any HTML tags, the OWASP project <a href="https://owasp.org/www-project-java-html-sanitizer/">Java HTML Sanitizer</a> is leveraged.<br>
1640     * I delegated such removal to a dedicated library to prevent missing of edge cases as well as potential bypasses.
1641     *
1642     * @param message          The original string message intended to be written in a log.
1643     * @param maxMessageLength The maximum number of characters after which the sanitized message must be truncated. If inferior to 1 then default to the value of 500.
1644     * @return The string message cleaned.
1645     * @see "https://www.wallarm.com/what/log-forging-attack"
1646     * @see "https://www.invicti.com/learn/crlf-injection"
1647     * @see "https://knowledge-base.secureflag.com/vulnerabilities/inadequate_input_validation/log_injection_vulnerability.html"
1648     * @see "https://capec.mitre.org/data/definitions/93.html"
1649     * @see "https://codeql.github.com/codeql-query-help/javascript/js-log-injection/"
1650     * @see "https://owasp.org/www-project-java-html-sanitizer/"
1651     * @see "https://github.com/OWASP/java-html-sanitizer"
1652     */
1653    public static String sanitizeLogMessage(String message, int maxMessageLength) {
1654        String sanitized = message;
1655        int maxSanitizedMessageLength = maxMessageLength;
1656
1657        if (sanitized != null && !sanitized.isBlank()) {
1658            if (maxSanitizedMessageLength < 1) {
1659                maxSanitizedMessageLength = 500;
1660            }
1661            //Step 1: Remove any CR/LR/TAB characters as well as leading and trailing spaces
1662            sanitized = sanitized.replaceAll("[\\n\\r\\t]", "").trim();
1663            //Step 2: Remove any Unicode LINE SEPARATOR or Unicode PARAGRAPH SEPARATOR as well as leading and trailing spaces
1664            sanitized = sanitized.replace("\u2028", "").replace("\u2029", "").trim();
1665            //Step 3: Remove ANSI escape sequences as well as leading and trailing spaces
1666            sanitized = sanitized.replaceAll("\u001B\\[[\\d;]*[a-zA-Z]", "").replace("\u001B", "").trim();
1667            //Step 4: Remove any HTML tags
1668            PolicyFactory htmlSanitizerPolicy = new HtmlPolicyBuilder().toFactory();
1669            sanitized = htmlSanitizerPolicy.sanitize(sanitized);
1670            //Step 5: Truncate the string in case of need
1671            if (sanitized.length() > maxSanitizedMessageLength) {
1672                sanitized = sanitized.substring(0, maxSanitizedMessageLength);
1673            }
1674        }
1675
1676        return sanitized;
1677    }
1678
1679    /**
1680     * Identify if an XML is an SVG image.<br>
1681     * The goal of this method is to prevent to leverage SVG, as an vector, to achieve a XSS when XML format is accepted.<br>
1682     * Leverage <a href="https://xmlgraphics.apache.org/batik/">Apache Batik</a> to delegate the parsing and support for the SVG format.<br><br>
1683     * <b>Due to the intended usage of the method, the following choice were made:</b>
1684     * <ul>
1685     * <li>Raise an exception when a non SVG related external references is identified.</li>
1686     * <li>Throw any exception that can occur if the provided content is invalid like for example an invalid XML file or a non existing file.</li>
1687     * <li>Explicitly check the XML prior to pass it to Batik even if Batik seems not prone to XXE/SSRF classes of vulnerability.</li>
1688     * </ul>
1689     *
1690     * @param xmlFilePath Filename of the XML file to check.
1691     * @return True only if XML is an valid SVG image.
1692     * @throws SecurityException If a non SVG external references is detected into the XML content.
1693     * @throws Exception         If a error occur due to an invalid content provided.
1694     * @see "https://developer.mozilla.org/en-US/docs/Web/SVG"
1695     * @see "https://www.fortinet.com/blog/threat-research/scalable-vector-graphics-attack-surface-anatomy"
1696     * @see "https://portswigger.net/web-security/cross-site-scripting"
1697     * @see "https://xmlgraphics.apache.org/batik/"
1698     * @see "https://github.com/apache/xmlgraphics-batik/blob/main/batik-dom/src/main/java/org/apache/batik/dom/util/SAXDocumentFactory.java#L420"
1699     * @see "https://mvnrepository.com/artifact/org.apache.xmlgraphics/batik-dom"
1700     * @see "https://mvnrepository.com/artifact/org.apache.xmlgraphics/batik-anim"
1701     * @see "https://portswigger.net/web-security/xxe"
1702     * @see "https://portswigger.net/web-security/ssrf"
1703     */
1704    public static boolean isXMLSVGImage(String xmlFilePath) throws Exception {
1705        boolean isSvg = true;
1706        List<String> svgValidSystemIDs = List.of("http://www.w3.org/Graphics/SVG/1.1/DTD/svg11.dtd", "http://www.w3.org/Graphics/SVG/1.1/DTD/svg11-basic.dtd", "http://www.w3.org/Graphics/SVG/1.1/DTD/svg11-tiny.dtd", "http://www.w3.org/TR/2001/REC-SVG-20010904/DTD/svg10.dtd");
1707
1708        //Load the XML content into a reader
1709        String xmlContent = Files.readString(Paths.get(xmlFilePath));
1710        //Then ensure that the XML document does not contains any non SVG external references
1711        try (Reader reader = StringReader.of(xmlContent)) {
1712            DocumentBuilderFactory xmlFactory = DocumentBuilderFactory.newInstance();
1713            DocumentBuilder docBuilder = xmlFactory.newDocumentBuilder();
1714            docBuilder.setEntityResolver((publicId, systemId) -> {
1715                if (systemId != null && !svgValidSystemIDs.contains(systemId)) {
1716                    throw new SecurityException("External references detected: " + systemId);
1717                }
1718                return new InputSource(new ByteArrayInputStream("".getBytes()));
1719            });
1720            docBuilder.parse(new InputSource(reader));
1721        }
1722        //Then parse the XML with Apache Batik
1723        try (Reader reader = StringReader.of(xmlContent)) {
1724            //Method SAXDocumentFactory.createDocument() do not load external DTD or entities.
1725            String parserClassName = XMLResourceDescriptor.getXMLParserClassName();
1726            SAXSVGDocumentFactory svgFactory = new SAXSVGDocumentFactory(parserClassName);
1727            //Method svgFactory.createSVGDocument() raise an IO exception if the XML is not a valid SVG image
1728            try {
1729                SVGDocument doc = svgFactory.createSVGDocument(null, reader);
1730                isSvg = (doc != null && doc.getRootElement() != null);
1731            } catch (IOException e) {
1732                isSvg = false;
1733            }
1734        }
1735
1736        return isSvg;
1737    }
1738}