001package eu.righettod; 002 003 004import com.auth0.jwt.interfaces.DecodedJWT; 005import org.apache.batik.anim.dom.SAXSVGDocumentFactory; 006import org.apache.batik.util.XMLResourceDescriptor; 007import org.apache.commons.csv.CSVFormat; 008import org.apache.commons.csv.CSVRecord; 009import org.apache.commons.imaging.ImageInfo; 010import org.apache.commons.imaging.Imaging; 011import org.apache.commons.imaging.common.ImageMetadata; 012import org.apache.commons.validator.routines.CreditCardValidator; 013import org.apache.commons.validator.routines.EmailValidator; 014import org.apache.commons.validator.routines.InetAddressValidator; 015import org.apache.pdfbox.Loader; 016import org.apache.pdfbox.pdmodel.PDDocument; 017import org.apache.pdfbox.pdmodel.PDDocumentCatalog; 018import org.apache.pdfbox.pdmodel.PDDocumentInformation; 019import org.apache.pdfbox.pdmodel.PDDocumentNameDictionary; 020import org.apache.pdfbox.pdmodel.common.PDMetadata; 021import org.apache.pdfbox.pdmodel.interactive.action.*; 022import org.apache.pdfbox.pdmodel.interactive.annotation.AnnotationFilter; 023import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotation; 024import org.apache.pdfbox.pdmodel.interactive.annotation.PDAnnotationLink; 025import org.apache.pdfbox.pdmodel.interactive.form.PDAcroForm; 026import org.apache.poi.poifs.filesystem.DirectoryEntry; 027import org.apache.poi.poifs.filesystem.POIFSFileSystem; 028import org.apache.poi.poifs.macros.VBAMacroReader; 029import org.apache.tika.detect.DefaultDetector; 030import org.apache.tika.detect.Detector; 031import org.apache.tika.io.TemporaryResources; 032import org.apache.tika.io.TikaInputStream; 033import org.apache.tika.metadata.Metadata; 034import org.apache.tika.mime.MediaType; 035import org.apache.tika.mime.MimeTypes; 036import org.apache.tika.parser.ParseContext; 037import org.iban4j.IbanUtil; 038import org.owasp.html.HtmlPolicyBuilder; 039import org.owasp.html.PolicyFactory; 040import org.w3c.dom.Document; 041import org.w3c.dom.svg.SVGDocument; 042import org.xml.sax.EntityResolver; 043import org.xml.sax.InputSource; 044import org.xml.sax.SAXException; 045 046import javax.crypto.Mac; 047import javax.crypto.spec.SecretKeySpec; 048import javax.imageio.ImageIO; 049import javax.json.Json; 050import javax.json.JsonReader; 051import javax.xml.XMLConstants; 052import javax.xml.parsers.DocumentBuilder; 053import javax.xml.parsers.DocumentBuilderFactory; 054import javax.xml.parsers.ParserConfigurationException; 055import javax.xml.stream.XMLInputFactory; 056import javax.xml.stream.XMLStreamReader; 057import javax.xml.stream.events.XMLEvent; 058import javax.xml.validation.Schema; 059import javax.xml.validation.SchemaFactory; 060import java.awt.*; 061import java.awt.image.BufferedImage; 062import java.io.*; 063import java.net.*; 064import java.net.http.HttpClient; 065import java.net.http.HttpRequest; 066import java.net.http.HttpResponse; 067import java.nio.ByteBuffer; 068import java.nio.charset.Charset; 069import java.nio.charset.StandardCharsets; 070import java.nio.file.Files; 071import java.nio.file.Paths; 072import java.security.MessageDigest; 073import java.security.SecureRandom; 074import java.time.Duration; 075import java.time.LocalDate; 076import java.time.YearMonth; 077import java.time.ZoneId; 078import java.util.*; 079import java.util.List; 080import java.util.concurrent.*; 081import java.util.concurrent.atomic.AtomicInteger; 082import java.util.regex.Matcher; 083import java.util.regex.Pattern; 084import java.util.zip.GZIPInputStream; 085import java.util.zip.ZipEntry; 086import java.util.zip.ZipFile; 087 088/** 089 * Provides different utilities methods to apply processing from a security perspective.<br> 090 * These code snippet: 091 * <ul> 092 * <li>Can be used, as "foundation", to customize the validation to the app context.</li> 093 * <li>Were implemented in a way to facilitate adding or removal of validations depending on usage context.</li> 094 * <li>Were centralized on one class to be able to enhance them across time as well as <a href="https://github.com/righettod/code-snippets-security-utils/issues">missing case/bug identification</a>.</li> 095 * </ul> 096 * <br> 097 * <a href="https://github.com/righettod/code-snippets-security-utils">GitHub repository</a>.<br><br> 098 * <a href="https://github.com/righettod/code-snippets-security-utils/blob/main/src/main/java/eu/righettod/SecurityUtils.java">Source code of the class</a>. 099 */ 100public class SecurityUtils { 101 /** 102 * Default constructor: Not needed as the class only provides static methods. 103 */ 104 private SecurityUtils() { 105 } 106 107 /** 108 * Apply a collection of validation to verify if a provided PIN code is considered weak (easy to guess) or none.<br> 109 * This method consider that format of the PIN code is [0-9]{6,}<br> 110 * Rule to consider a PIN code as weak: 111 * <ul> 112 * <li>Length is inferior to 6 positions.</li> 113 * <li>Contain only the same number or only a sequence of zero.</li> 114 * <li>Contain sequence of following incremental or decremental numbers.</li> 115 * </ul> 116 * 117 * @param pinCode PIN code to verify. 118 * @return True only if the PIN is considered as weak. 119 */ 120 public static boolean isWeakPINCode(String pinCode) { 121 boolean isWeak = true; 122 //Length is inferior to 6 positions 123 //Use "Long.parseLong(pinCode)" to cause a NumberFormatException if the PIN is not a numeric one 124 //and to ensure that the PIN is not only a sequence of zero 125 if (pinCode != null && Long.parseLong(pinCode) > 0 && pinCode.trim().length() > 5) { 126 //Contain only the same number 127 String regex = String.format("^[%s]{%s}$", pinCode.charAt(0), pinCode.length()); 128 if (!Pattern.matches(regex, pinCode)) { 129 //Contain sequence of following incremental or decremental numbers 130 char previousChar = 'X'; 131 boolean containSequence = false; 132 for (char c : pinCode.toCharArray()) { 133 if (previousChar != 'X') { 134 int previousNbr = Integer.parseInt(String.valueOf(previousChar)); 135 int currentNbr = Integer.parseInt(String.valueOf(c)); 136 if (currentNbr == (previousNbr - 1) || currentNbr == (previousNbr + 1)) { 137 containSequence = true; 138 break; 139 } 140 } 141 previousChar = c; 142 } 143 if (!containSequence) { 144 isWeak = false; 145 } 146 } 147 } 148 return isWeak; 149 } 150 151 /** 152 * Apply a collection of validations on a Word 97-2003 (binary format) document file provided: 153 * <ul> 154 * <li>Real Microsoft Word 97-2003 document file.</li> 155 * <li>No VBA Macro.<br></li> 156 * <li>No embedded objects.</li> 157 * </ul> 158 * 159 * @param wordFilePath Filename of the Word document file to check. 160 * @return True only if the file pass all validations. 161 * @see "https://poi.apache.org/components/" 162 * @see "https://poi.apache.org/components/document/" 163 * @see "https://poi.apache.org/components/poifs/how-to.html" 164 * @see "https://poi.apache.org/components/poifs/embeded.html" 165 * @see "https://poi.apache.org/" 166 * @see "https://mvnrepository.com/artifact/org.apache.poi/poi" 167 */ 168 public static boolean isWord972003DocumentSafe(String wordFilePath) { 169 boolean isSafe = false; 170 try { 171 File wordFile = new File(wordFilePath); 172 if (wordFile.exists() && wordFile.canRead() && wordFile.isFile()) { 173 //Step 1: Try to load the file, if its fail then it imply that is not a valid Word 97-2003 format file 174 try (POIFSFileSystem fs = new POIFSFileSystem(wordFile)) { 175 //Step 2: Check if the document contains VBA macros, in our case is not allowed 176 VBAMacroReader macroReader = new VBAMacroReader(fs); 177 Map<String, String> macros = macroReader.readMacros(); 178 if (macros == null || macros.isEmpty()) { 179 //Step 3: Check if the document contains any embedded objects, in our case is not allowed 180 //From POI documentation: 181 //Word normally stores embedded files in subdirectories of the ObjectPool directory, itself a subdirectory of the filesystem root. 182 //Typically, these subdirectories and named starting with an underscore, followed by 10 numbers. 183 final List<String> embeddedObjectFound = new ArrayList<>(); 184 DirectoryEntry root = fs.getRoot(); 185 if (root.getEntryCount() > 0) { 186 root.iterator().forEachRemaining(entry -> { 187 if ("ObjectPool".equalsIgnoreCase(entry.getName()) && entry instanceof DirectoryEntry) { 188 DirectoryEntry objPoolDirectory = (DirectoryEntry) entry; 189 if (objPoolDirectory.getEntryCount() > 0) { 190 objPoolDirectory.iterator().forEachRemaining(objPoolDirectoryEntry -> { 191 if (objPoolDirectoryEntry instanceof DirectoryEntry) { 192 DirectoryEntry objPoolDirectoryEntrySubDirectoryEntry = (DirectoryEntry) objPoolDirectoryEntry; 193 if (objPoolDirectoryEntrySubDirectoryEntry.getEntryCount() > 0) { 194 objPoolDirectoryEntrySubDirectoryEntry.forEach(objPoolDirectoryEntrySubDirectoryEntryEntry -> { 195 if (objPoolDirectoryEntrySubDirectoryEntryEntry.isDocumentEntry()) { 196 embeddedObjectFound.add(objPoolDirectoryEntrySubDirectoryEntryEntry.getName()); 197 } 198 }); 199 } 200 } 201 }); 202 } 203 } 204 }); 205 } 206 isSafe = embeddedObjectFound.isEmpty(); 207 } 208 } 209 } 210 } catch (Exception e) { 211 isSafe = false; 212 } 213 return isSafe; 214 } 215 216 /** 217 * Ensure that an XML file does not contain any External Entity, DTD or XInclude instructions. 218 * 219 * @param xmlFilePath Filename of the XML file to check. 220 * @return True only if the file pass all validations. 221 * @see "https://portswigger.net/web-security/xxe" 222 * @see "https://cheatsheetseries.owasp.org/cheatsheets/XML_External_Entity_Prevention_Cheat_Sheet.html#java" 223 * @see "https://docs.oracle.com/en/java/javase/13/security/java-api-xml-processing-jaxp-security-guide.html#GUID-82F8C206-F2DF-4204-9544-F96155B1D258" 224 * @see "https://www.w3.org/TR/xinclude-11/" 225 * @see "https://en.wikipedia.org/wiki/XInclude" 226 */ 227 public static boolean isXMLSafe(String xmlFilePath) { 228 boolean isSafe = false; 229 try { 230 File xmlFile = new File(xmlFilePath); 231 if (xmlFile.exists() && xmlFile.canRead() && xmlFile.isFile()) { 232 //Step 1a: Verify that the XML file content does not contain any XInclude instructions 233 boolean containXInclude = Files.readAllLines(xmlFile.toPath()).stream().anyMatch(line -> line.toLowerCase(Locale.ROOT).contains(":include ")); 234 if (!containXInclude) { 235 //Step 1b: Parse the XML file, if an exception occur than it's imply that the XML specified is not a valid ones 236 //Create an XML document builder throwing Exception if a DOCTYPE instruction is present 237 DocumentBuilderFactory dbfInstance = DocumentBuilderFactory.newInstance(); 238 dbfInstance.setFeature("http://apache.org/xml/features/disallow-doctype-decl", true); 239 //Xerces 2 only 240 //dbfInstance.setFeature("http://xerces.apache.org/xerces2-j/features.html#disallow-doctype-decl",true); 241 dbfInstance.setXIncludeAware(false); 242 DocumentBuilder builder = dbfInstance.newDocumentBuilder(); 243 //Parse the document 244 Document doc = builder.parse(xmlFile); 245 isSafe = (doc != null && doc.getDocumentElement() != null); 246 } 247 } 248 } catch (Exception e) { 249 isSafe = false; 250 } 251 return isSafe; 252 } 253 254 255 /** 256 * Extract all URL links from a PDF file provided.<br> 257 * This can be used to apply validation on a PDF against contained links. 258 * 259 * @param pdfFilePath pdfFilePath Filename of the PDF file to process. 260 * @return A List of URL objects that is empty if no links is found. 261 * @throws Exception If any error occurs during the processing of the PDF file. 262 * @see "https://www.gushiciku.cn/pl/21KQ" 263 * @see "https://pdfbox.apache.org/" 264 * @see "https://mvnrepository.com/artifact/org.apache.pdfbox/pdfbox" 265 */ 266 public static List<URL> extractAllPDFLinks(String pdfFilePath) throws Exception { 267 final List<URL> links = new ArrayList<>(); 268 File pdfFile = new File(pdfFilePath); 269 try (PDDocument document = Loader.loadPDF(pdfFile)) { 270 PDDocumentCatalog documentCatalog = document.getDocumentCatalog(); 271 AnnotationFilter actionURIAnnotationFilter = new AnnotationFilter() { 272 @Override 273 public boolean accept(PDAnnotation annotation) { 274 boolean keep = false; 275 if (annotation instanceof PDAnnotationLink) { 276 keep = (((PDAnnotationLink) annotation).getAction() instanceof PDActionURI); 277 } 278 return keep; 279 } 280 }; 281 documentCatalog.getPages().forEach(page -> { 282 try { 283 page.getAnnotations(actionURIAnnotationFilter).forEach(annotation -> { 284 PDActionURI linkAnnotation = (PDActionURI) ((PDAnnotationLink) annotation).getAction(); 285 try { 286 URL urlObj = new URL(linkAnnotation.getURI()); 287 if (!links.contains(urlObj)) { 288 links.add(urlObj); 289 } 290 } catch (MalformedURLException e) { 291 throw new RuntimeException(e); 292 } 293 }); 294 } catch (Exception e) { 295 throw new RuntimeException(e); 296 } 297 }); 298 } 299 return links; 300 } 301 302 /** 303 * Apply a collection of validations on a PDF file provided: 304 * <ul> 305 * <li>Real PDF file.</li> 306 * <li>No attachments.</li> 307 * <li>No Javascript code.</li> 308 * <li>No links using action of type URI/Launch/RemoteGoTo/ImportData.</li> 309 * <li>No XFA forms in order to prevent exposure to XXE/SSRF like CVE-2025-54988.</li> 310 * </ul> 311 * 312 * @param pdfFilePath Filename of the PDF file to check. 313 * @return True only if the file pass all validations. 314 * @see "https://stackoverflow.com/a/36161267" 315 * @see "https://www.gushiciku.cn/pl/21KQ" 316 * @see "https://github.com/jonaslejon/malicious-pdf" 317 * @see "https://pdfbox.apache.org/" 318 * @see "https://mvnrepository.com/artifact/org.apache.pdfbox/pdfbox" 319 * @see "https://nvd.nist.gov/vuln/detail/CVE-2025-54988" 320 * @see "https://github.com/mgthuramoemyint/POC-CVE-2025-54988" 321 * @see "https://en.wikipedia.org/wiki/XFA" 322 */ 323 public static boolean isPDFSafe(String pdfFilePath) { 324 boolean isSafe = false; 325 try { 326 File pdfFile = new File(pdfFilePath); 327 if (pdfFile.exists() && pdfFile.canRead() && pdfFile.isFile()) { 328 //Step 1: Try to load the file, if its fail then it imply that is not a valid PDF file 329 try (PDDocument document = Loader.loadPDF(pdfFile)) { 330 //Step 2: Check if the file contains attached files, in our case is not allowed 331 PDDocumentCatalog documentCatalog = document.getDocumentCatalog(); 332 PDDocumentNameDictionary namesDictionary = new PDDocumentNameDictionary(documentCatalog); 333 if (namesDictionary.getEmbeddedFiles() == null) { 334 //Step 3: Check if the file contains any XFA forms 335 PDAcroForm acroForm = documentCatalog.getAcroForm(); 336 boolean hasForm = (acroForm != null && acroForm.getXFA() != null); 337 if (!hasForm) { 338 //Step 4: Check if the file contains Javascript code, in our case is not allowed 339 if (namesDictionary.getJavaScript() == null) { 340 //Step 5: Check if the file contains links using action of type URI/Launch/RemoteGoTo/ImportData, in our case is not allowed 341 final List<Integer> notAllowedAnnotationCounterList = new ArrayList<>(); 342 AnnotationFilter notAllowedAnnotationFilter = new AnnotationFilter() { 343 @Override 344 public boolean accept(PDAnnotation annotation) { 345 boolean keep = false; 346 if (annotation instanceof PDAnnotationLink) { 347 PDAnnotationLink link = (PDAnnotationLink) annotation; 348 PDAction action = link.getAction(); 349 if ((action instanceof PDActionURI) || (action instanceof PDActionLaunch) || (action instanceof PDActionRemoteGoTo) || (action instanceof PDActionImportData)) { 350 keep = true; 351 } 352 } 353 return keep; 354 } 355 }; 356 documentCatalog.getPages().forEach(page -> { 357 try { 358 notAllowedAnnotationCounterList.add(page.getAnnotations(notAllowedAnnotationFilter).size()); 359 } catch (IOException e) { 360 throw new RuntimeException(e); 361 } 362 }); 363 if (notAllowedAnnotationCounterList.stream().reduce(0, Integer::sum) == 0) { 364 isSafe = true; 365 } 366 } 367 } 368 } 369 } 370 } 371 } catch (Exception e) { 372 isSafe = false; 373 } 374 return isSafe; 375 } 376 377 /** 378 * Remove as much as possible metadata from the provided PDF document object. 379 * 380 * @param document PDFBox PDF document object on which metadata must be removed. 381 * @see "https://gist.github.com/righettod/d7e07443c43d393a39de741a0d920069" 382 * @see "https://pdfbox.apache.org/" 383 * @see "https://mvnrepository.com/artifact/org.apache.pdfbox/pdfbox" 384 */ 385 public static void clearPDFMetadata(PDDocument document) { 386 if (document != null) { 387 PDDocumentInformation infoEmpty = new PDDocumentInformation(); 388 document.setDocumentInformation(infoEmpty); 389 PDMetadata newMetadataEmpty = new PDMetadata(document); 390 document.getDocumentCatalog().setMetadata(newMetadataEmpty); 391 } 392 } 393 394 395 /** 396 * Validate that the URL provided is really a relative URL. 397 * 398 * @param targetUrl URL to validate. 399 * @return True only if the file pass all validations. 400 * @see "https://portswigger.net/web-security/ssrf" 401 * @see "https://stackoverflow.com/q/6785442" 402 */ 403 public static boolean isRelativeURL(String targetUrl) { 404 boolean isValid = false; 405 String work = targetUrl; 406 Pattern startingPrefix = Pattern.compile("^[/a-zA-Z0-9\\-_].*"); 407 //Reject any URL no starting with a slash, letter, number, dash, or underscore 408 if (startingPrefix.matcher(work).find()) { 409 //Reject any URL encoded content and URL starting with a double slash 410 if (!work.startsWith("//") && !work.contains("%")) { 411 //Try to create en URI object 412 try { 413 URI u = new URI(work); 414 //Scheme must be null 415 if (u.getScheme() == null) { 416 isValid = (!u.isAbsolute()); 417 } 418 } catch (URISyntaxException mf) { 419 isValid = false; 420 } 421 } 422 } 423 424 return isValid; 425 } 426 427 /** 428 * Apply a collection of validations on a ZIP file provided: 429 * <ul> 430 * <li>Real ZIP file.</li> 431 * <li>Contain less than a specified level of deepness.</li> 432 * <li>Do not contain Zip-Slip entry path.</li> 433 * </ul> 434 * 435 * @param zipFilePath Filename of the ZIP file to check. 436 * @param maxLevelDeepness Threshold of deepness above which a ZIP archive will be rejected. 437 * @param rejectArchiveFile Flag to specify if presence of any archive entry will cause the rejection of the ZIP file. 438 * @return True only if the file pass all validations. 439 * @see "https://rules.sonarsource.com/java/type/Security%20Hotspot/RSPEC-5042" 440 * @see "https://security.snyk.io/research/zip-slip-vulnerability" 441 * @see "https://en.wikipedia.org/wiki/Zip_bomb" 442 * @see "https://github.com/ptoomey3/evilarc" 443 * @see "https://github.com/abdulfatir/ZipBomb" 444 * @see "https://www.baeldung.com/cs/zip-bomb" 445 * @see "https://thesecurityvault.com/attacks-with-zip-files-and-mitigations/" 446 * @see "https://wiki.sei.cmu.edu/confluence/display/java/IDS04-J.+Safely+extract+files+from+ZipInputStream" 447 */ 448 public static boolean isZIPSafe(String zipFilePath, int maxLevelDeepness, boolean rejectArchiveFile) { 449 List<String> archiveExtensions = Arrays.asList("zip", "tar", "7z", "gz", "jar", "phar", "bz2", "tgz"); 450 boolean isSafe = false; 451 try { 452 File zipFile = new File(zipFilePath); 453 if (zipFile.exists() && zipFile.canRead() && zipFile.isFile() && maxLevelDeepness > 0) { 454 //Step 1: Try to load the file, if its fail then it imply that is not a valid ZIP file 455 try (ZipFile zipArch = new ZipFile(zipFile)) { 456 //Step 2: Parse entries 457 long deepness = 0; 458 ZipEntry zipEntry; 459 String entryExtension; 460 String zipEntryName; 461 boolean validationsFailed = false; 462 Enumeration<? extends ZipEntry> entries = zipArch.entries(); 463 while (entries.hasMoreElements()) { 464 zipEntry = entries.nextElement(); 465 zipEntryName = zipEntry.getName(); 466 entryExtension = zipEntryName.substring(zipEntryName.lastIndexOf(".") + 1).toLowerCase(Locale.ROOT).trim(); 467 //Step 2a: Check if the current entry is an archive file 468 if (rejectArchiveFile && archiveExtensions.contains(entryExtension)) { 469 validationsFailed = true; 470 break; 471 } 472 //Step 2b: Check that level of deepness is inferior to the threshold specified 473 if (zipEntryName.contains("/")) { 474 //Determine deepness by inspecting the entry name. 475 //Indeed, folder will be represented like this: folder/folder/folder/ 476 //So we can count the number of "/" to identify the deepness of the entry 477 deepness = zipEntryName.chars().filter(ch -> ch == '/').count(); 478 if (deepness > maxLevelDeepness) { 479 validationsFailed = true; 480 break; 481 } 482 } 483 //Step 2c: Check if any entries match pattern of zip slip payload 484 if (zipEntryName.contains("..\\") || zipEntryName.contains("../")) { 485 validationsFailed = true; 486 break; 487 } 488 } 489 if (!validationsFailed) { 490 isSafe = true; 491 } 492 } 493 } 494 } catch (Exception e) { 495 isSafe = false; 496 } 497 return isSafe; 498 } 499 500 /** 501 * Identify the mime type of the content specified (array of bytes).<br> 502 * Note that it cannot be fully trusted (see the tweet '1595824709186519041' referenced), so, additional validations are required. 503 * 504 * @param content The content as an array of bytes. 505 * @return The mime type in lower case or null if it cannot be identified. 506 * @see "https://twitter.com/righettod/status/1595824709186519041" 507 * @see "https://tika.apache.org/" 508 * @see "https://mvnrepository.com/artifact/org.apache.tika/tika-core" 509 * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types" 510 * @see "https://www.iana.org/assignments/media-types/media-types.xhtml" 511 */ 512 public static String identifyMimeType(byte[] content) { 513 String mimeType = null; 514 if (content != null && content.length > 0) { 515 Detector detector = new DefaultDetector(MimeTypes.getDefaultMimeTypes()); 516 Metadata metadata = new Metadata(); 517 try { 518 try (TemporaryResources temporaryResources = new TemporaryResources(); TikaInputStream tikaInputStream = TikaInputStream.get(new ByteArrayInputStream(content), temporaryResources, metadata)) { 519 MediaType mt = detector.detect(tikaInputStream, metadata, new ParseContext()); 520 if (mt != null) { 521 mimeType = mt.toString().toLowerCase(Locale.ROOT); 522 } 523 } 524 } catch (IOException ioe) { 525 mimeType = null; 526 } 527 } 528 return mimeType; 529 } 530 531 /** 532 * Apply a collection of validations on a string expected to be an public IP address: 533 * <ul> 534 * <li>Is a valid IP v4 or v6 address.</li> 535 * <li>Is public from an Internet perspective.</li> 536 * </ul> 537 * <br> 538 * <b>Note:</b> I often see missing such validation in the value read from HTTP request headers like "X-Forwarded-For" or "Forwarded". 539 * <br><br> 540 * <b>Note for IPv6:</b> I used documentation found so it is really experimental! 541 * 542 * @param ip String expected to be a valid IP address. 543 * @return True only if the string pass all validations. 544 * @see "https://commons.apache.org/proper/commons-validator/" 545 * @see "https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/InetAddressValidator.html" 546 * @see "https://cheatsheetseries.owasp.org/cheatsheets/Server_Side_Request_Forgery_Prevention_Cheat_Sheet.html" 547 * @see "https://cheatsheetseries.owasp.org/assets/Server_Side_Request_Forgery_Prevention_Cheat_Sheet_Orange_Tsai_Talk.pdf" 548 * @see "https://cheatsheetseries.owasp.org/assets/Server_Side_Request_Forgery_Prevention_Cheat_Sheet_SSRF_Bible.pdf" 549 * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/X-Forwarded-For" 550 * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/Forwarded" 551 * @see "https://ipcisco.com/lesson/ipv6-address/" 552 * @see "https://www.juniper.net/documentation/us/en/software/junos/interfaces-security-devices/topics/topic-map/security-interface-ipv4-ipv6-protocol.html" 553 * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.base/java/net/InetAddress.html#getByName(java.lang.String)" 554 * @see "https://www.arin.net/reference/research/statistics/address_filters/" 555 * @see "https://en.wikipedia.org/wiki/Multicast_address" 556 * @see "https://stackoverflow.com/a/5619409" 557 * @see "https://www.ripe.net/media/documents/ipv6-address-types.pdf" 558 * @see "https://www.iana.org/assignments/ipv6-unicast-address-assignments/ipv6-unicast-address-assignments.xhtml" 559 * @see "https://developer.android.com/reference/java/net/Inet6Address" 560 * @see "https://en.wikipedia.org/wiki/Unique_local_address" 561 */ 562 public static boolean isPublicIPAddress(String ip) { 563 boolean isValid = false; 564 try { 565 //Quick validation on the string itself based on characters used to compose an IP v4/v6 address 566 if (Pattern.matches("[0-9a-fA-F:.]+", ip)) { 567 //If OK then use the dedicated InetAddressValidator from Apache Commons Validator 568 if (InetAddressValidator.getInstance().isValid(ip)) { 569 //If OK then validate that is an public IP address 570 //From Javadoc for "InetAddress.getByName": If a literal IP address is supplied, only the validity of the address format is checked. 571 InetAddress addr = InetAddress.getByName(ip); 572 isValid = (!addr.isAnyLocalAddress() && !addr.isLinkLocalAddress() && !addr.isLoopbackAddress() && !addr.isMulticastAddress() && !addr.isSiteLocalAddress()); 573 //If OK and the IP is an V6 one then make additional validation because the built-in Java API will let pass some V6 IP 574 //For the prefix map, the start of the key indicates if the value is a regex or a string 575 if (isValid && (addr instanceof Inet6Address)) { 576 Map<String, String> prefixes = new HashMap<>(); 577 prefixes.put("REGEX_LOOPBACK", "^(0|:)+1$"); 578 prefixes.put("REGEX_UNIQUE-LOCAL-ADDRESSES", "^f(c|d)[a-f0-9]{2}:.*$"); 579 prefixes.put("STRING_LINK-LOCAL-ADDRESSES", "fe80:"); 580 prefixes.put("REGEX_TEREDO", "^2001:[0]*:.*$"); 581 prefixes.put("REGEX_BENCHMARKING", "^2001:[0]*2:.*$"); 582 prefixes.put("REGEX_ORCHID", "^2001:[0]*10:.*$"); 583 prefixes.put("STRING_DOCUMENTATION", "2001:db8:"); 584 prefixes.put("STRING_GLOBAL-UNICAST", "2000:"); 585 prefixes.put("REGEX_MULTICAST", "^ff[0-9]{2}:.*$"); 586 final List<Boolean> results = new ArrayList<>(); 587 final String ipLower = ip.trim().toLowerCase(Locale.ROOT); 588 prefixes.forEach((addressType, expr) -> { 589 String exprLower = expr.trim().toLowerCase(); 590 if (addressType.startsWith("STRING_")) { 591 results.add(ipLower.startsWith(exprLower)); 592 } else { 593 results.add(Pattern.matches(exprLower, ipLower)); 594 } 595 }); 596 isValid = ((results.size() == prefixes.size()) && !results.contains(Boolean.TRUE)); 597 } 598 } 599 } 600 } catch (Exception e) { 601 isValid = false; 602 } 603 return isValid; 604 } 605 606 /** 607 * Compute a SHA256 hash from an input composed of a collection of strings.<br><br> 608 * This method take care to build the source string in a way to prevent this source string to be prone to abuse targeting the different parts composing it.<br><br> 609 * <p> 610 * Example of possible abuse without precautions applied during the hash calculation logic:<br> 611 * Hash of <code>SHA256("Hello", "My", "World!!!")</code> will be equals to the hash of <code>SHA256("Hell", "oMyW", "orld!!!")</code>.<br> 612 * </p> 613 * This method ensure that both hash above will be different.<br><br> 614 * 615 * <b>Note:</b> The character <code>|</code> is used, as separator, of every parts so a part is not allowed to contains this character. 616 * 617 * @param parts Ordered list of strings to use to build the input string for which the hash must be computed on. No null value is accepted on object composing the collection. 618 * @return The hash, as an array of bytes, to allow caller to convert it to the final representation wanted (HEX, Base64, etc.). If the collection passed is null or empty then the method return null. 619 * @throws Exception If any exception occurs 620 * @see "https://github.com/righettod/code-snippets-security-utils/issues/16" 621 * @see "https://pentesterlab.com/badges/codereview" 622 * @see "https://blog.trailofbits.com/2024/08/21/yolo-is-not-a-valid-hash-construction/" 623 * @see "https://www.nist.gov/publications/sha-3-derived-functions-cshake-kmac-tuplehash-and-parallelhash" 624 */ 625 public static byte[] computeHashNoProneToAbuseOnParts(List<String> parts) throws Exception { 626 byte[] hash = null; 627 String separator = "|"; 628 if (parts != null && !parts.isEmpty()) { 629 //Ensure that not part is null 630 if (parts.stream().anyMatch(Objects::isNull)) { 631 throw new IllegalArgumentException("No part must be null!"); 632 } 633 //Ensure that the separator is absent from every part 634 if (parts.stream().anyMatch(part -> part.contains(separator))) { 635 throw new IllegalArgumentException(String.format("The character '%s', used as parts separator, must be absent from every parts!", separator)); 636 } 637 MessageDigest digest = MessageDigest.getInstance("SHA-256"); 638 final StringBuilder buffer = new StringBuilder(separator); 639 parts.forEach(p -> { 640 buffer.append(p).append(separator); 641 }); 642 hash = digest.digest(buffer.toString().getBytes(StandardCharsets.UTF_8)); 643 } 644 return hash; 645 } 646 647 /** 648 * Ensure that an XML file only uses DTD/XSD references (called System Identifier) present in the allowed list provided.<br><br> 649 * The code is based on the validation implemented into the OpenJDK 21, by the class <b><a href="https://github.com/openjdk/jdk/blob/jdk-21%2B35/src/java.prefs/share/classes/java/util/prefs/XmlSupport.java">java.util.prefs.XmlSupport</a></b>, in the method <b><a href="https://github.com/openjdk/jdk/blob/jdk-21%2B35/src/java.prefs/share/classes/java/util/prefs/XmlSupport.java#L240">loadPrefsDoc()</a></b>.<br><br> 650 * The method also ensure that no Public Identifier is used to prevent potential bypasses of the validations. 651 * 652 * @param xmlFilePath Filename of the XML file to check. 653 * @param allowedSystemIdentifiers List of URL allowed for System Identifier specified for any XSD/DTD references. 654 * @return True only if the file pass all validations. 655 * @see "https://www.w3schools.com/xml/prop_documenttype_systemid.asp" 656 * @see "https://www.ibm.com/docs/en/integration-bus/9.0.0?topic=doctypedecl-xml-systemid" 657 * @see "https://www.liquid-technologies.com/Reference/Glossary/XML_DocType.html" 658 * @see "https://www.xml.com/pub/98/08/xmlqna0.html" 659 * @see "https://github.com/openjdk/jdk/blob/jdk-21%2B35/src/java.prefs/share/classes/java/util/prefs/XmlSupport.java#L397" 660 * @see "https://en.wikipedia.org/wiki/Formal_Public_Identifier" 661 */ 662 public static boolean isXMLOnlyUseAllowedXSDorDTD(String xmlFilePath, final List<String> allowedSystemIdentifiers) { 663 boolean isSafe = false; 664 final String errorTemplate = "Non allowed %s ID detected!"; 665 final String emptyFakeDTD = "<?xml version=\"1.0\" encoding=\"UTF-8\"?><!ELEMENT dummy EMPTY>"; 666 final String emptyFakeXSD = "<xs:schema xmlns:xs=\"http://www.w3.org/2001/XMLSchema\"> <xs:element name=\"dummy\"/></xs:schema>"; 667 668 if (allowedSystemIdentifiers == null || allowedSystemIdentifiers.isEmpty()) { 669 throw new IllegalArgumentException("At least one SID must be specified!"); 670 } 671 File xmlFile = new File(xmlFilePath); 672 if (xmlFile.exists() && xmlFile.canRead() && xmlFile.isFile()) { 673 try { 674 EntityResolver resolverValidator = (publicId, systemId) -> { 675 if (publicId != null) { 676 throw new SAXException(String.format(errorTemplate, "PUBLIC")); 677 } 678 if (!allowedSystemIdentifiers.contains(systemId)) { 679 throw new SAXException(String.format(errorTemplate, "SYSTEM")); 680 } 681 //If it is OK then return a empty DTD/XSD 682 return new InputSource(new StringReader(systemId.toLowerCase().endsWith(".dtd") ? emptyFakeDTD : emptyFakeXSD)); 683 }; 684 DocumentBuilderFactory dbfInstance = DocumentBuilderFactory.newInstance(); 685 dbfInstance.setIgnoringElementContentWhitespace(true); 686 dbfInstance.setXIncludeAware(false); 687 dbfInstance.setValidating(false); 688 dbfInstance.setCoalescing(true); 689 dbfInstance.setIgnoringComments(false); 690 DocumentBuilder builder = dbfInstance.newDocumentBuilder(); 691 builder.setEntityResolver(resolverValidator); 692 Document doc = builder.parse(xmlFile); 693 isSafe = (doc != null); 694 } catch (SAXException | IOException | ParserConfigurationException e) { 695 isSafe = false; 696 } 697 } 698 699 return isSafe; 700 } 701 702 /** 703 * Apply a collection of validations on a EXCEL CSV file provided (file was expected to be opened in Microsoft EXCEL): 704 * <ul> 705 * <li>Real CSV file.</li> 706 * <li>Do not contains any payload related to a CSV injections.</li> 707 * </ul> 708 * Ensure that, if Apache Commons CSV does not find any record then, the file will be considered as NOT safe (prevent potential bypasses).<br><br> 709 * <b>Note:</b> Record delimiter used is the <code>,</code> (comma) character. See the Apache Commons CSV reference provided for EXCEL.<br> 710 * 711 * @param csvFilePath Filename of the CSV file to check. 712 * @return True only if the file pass all validations. 713 * @see "https://commons.apache.org/proper/commons-csv/" 714 * @see "https://commons.apache.org/proper/commons-csv/apidocs/org/apache/commons/csv/CSVFormat.html#EXCEL" 715 * @see "https://www.we45.com/post/your-excel-sheets-are-not-safe-heres-how-to-beat-csv-injection" 716 * @see "https://www.whiteoaksecurity.com/blog/2020-4-23-csv-injection-whats-the-risk/" 717 * @see "https://book.hacktricks.xyz/pentesting-web/formula-csv-doc-latex-ghostscript-injection" 718 * @see "https://owasp.org/www-community/attacks/CSV_Injection" 719 * @see "https://payatu.com/blog/csv-injection-basic-to-exploit/" 720 * @see "https://cwe.mitre.org/data/definitions/1236.html" 721 */ 722 public static boolean isExcelCSVSafe(String csvFilePath) { 723 boolean isSafe; 724 final AtomicInteger recordCount = new AtomicInteger(); 725 final List<Character> payloadDetectionCharacters = List.of('=', '+', '@', '-', '\r', '\t'); 726 727 try { 728 final List<String> payloadsIdentified = new ArrayList<>(); 729 try (Reader in = new FileReader(csvFilePath)) { 730 Iterable<CSVRecord> records = CSVFormat.EXCEL.parse(in); 731 records.forEach(record -> { 732 record.forEach(recordValue -> { 733 if (recordValue != null && !recordValue.trim().isEmpty() && payloadDetectionCharacters.contains(recordValue.trim().charAt(0))) { 734 payloadsIdentified.add(recordValue); 735 } 736 recordCount.getAndIncrement(); 737 }); 738 }); 739 } 740 isSafe = (payloadsIdentified.isEmpty() && recordCount.get() > 0); 741 } catch (Exception e) { 742 isSafe = false; 743 } 744 745 return isSafe; 746 } 747 748 /** 749 * Provide a way to add an integrity marker (<a href="https://en.wikipedia.org/wiki/HMAC">HMAC</a>) to a serialized object serialized using the <a href="https://www.baeldung.com/java-serialization">java native system</a> (binary).<br> 750 * The goal is to provide <b>a temporary workaround</b> to try to prevent deserialization attacks and give time to move to a text-based serialization approach. 751 * 752 * @param processingModeType Define the mode of processing i.e. protect or validate. ({@link ProcessingModeType}) 753 * @param input When the processing mode is "protect" than the expected input (string) is a java serialized object encoded in Base64 otherwise (processing mode is "validate") expected input is the output of this method when the "protect" mode was used. 754 * @param secret Secret to use to compute the SHA256 HMAC. 755 * @return A map with the following keys: <ul><li><b>PROCESSING_MODE</b>: Processing mode used to compute the result.</li><li><b>STATUS</b>: A boolean indicating if the processing was successful or not.</li><li><b>RESULT</b>: Always contains a string representing the protected serialized object in the format <code>[SERIALIZED_OBJECT_BASE64_ENCODED]:[SERIALIZED_OBJECT_HMAC_BASE64_ENCODED]</code>.</li></ul> 756 * @throws Exception If any exception occurs. 757 * @see "https://cheatsheetseries.owasp.org/cheatsheets/Deserialization_Cheat_Sheet.html" 758 * @see "https://owasp.org/www-project-top-ten/2017/A8_2017-Insecure_Deserialization" 759 * @see "https://portswigger.net/web-security/deserialization" 760 * @see "https://www.baeldung.com/java-serialization-approaches" 761 * @see "https://www.baeldung.com/java-serialization" 762 * @see "https://cryptobook.nakov.com/mac-and-key-derivation/hmac-and-key-derivation" 763 * @see "https://en.wikipedia.org/wiki/HMAC" 764 * @see "https://smattme.com/posts/how-to-generate-hmac-signature-in-java/" 765 */ 766 public static Map<String, Object> ensureSerializedObjectIntegrity(ProcessingModeType processingModeType, String input, byte[] secret) throws Exception { 767 Map<String, Object> results; 768 String resultFormatTemplate = "%s:%s"; 769 //Verify input provided to be consistent 770 if (processingModeType == null) { 771 throw new IllegalArgumentException("The processing mode is mandatory!"); 772 } 773 if (input == null || input.trim().isEmpty()) { 774 throw new IllegalArgumentException("Input data is mandatory!"); 775 } 776 if (secret == null || secret.length == 0) { 777 throw new IllegalArgumentException("The HMAC secret is mandatory!"); 778 } 779 if (processingModeType.equals(ProcessingModeType.VALIDATE) && input.split(":").length != 2) { 780 throw new IllegalArgumentException("Input data provided is invalid for the processing mode specified!"); 781 } 782 //Processing 783 Base64.Decoder b64Decoder = Base64.getDecoder(); 784 Base64.Encoder b64Encoder = Base64.getEncoder(); 785 String hmacAlgorithm = "HmacSHA256"; 786 Mac mac = Mac.getInstance(hmacAlgorithm); 787 SecretKeySpec key = new SecretKeySpec(secret, hmacAlgorithm); 788 mac.init(key); 789 results = new HashMap<>(); 790 results.put("PROCESSING_MODE", processingModeType.toString()); 791 switch (processingModeType) { 792 case PROTECT -> { 793 byte[] objectBytes = b64Decoder.decode(input); 794 byte[] hmac = mac.doFinal(objectBytes); 795 String encodedHmac = b64Encoder.encodeToString(hmac); 796 results.put("STATUS", Boolean.TRUE); 797 results.put("RESULT", String.format(resultFormatTemplate, input, encodedHmac)); 798 } 799 case VALIDATE -> { 800 String[] parts = input.split(":"); 801 byte[] objectBytes = b64Decoder.decode(parts[0].trim()); 802 byte[] hmacProvided = b64Decoder.decode(parts[1].trim()); 803 byte[] hmacComputed = mac.doFinal(objectBytes); 804 String encodedHmacComputed = b64Encoder.encodeToString(hmacComputed); 805 Boolean hmacIsValid = Arrays.equals(hmacProvided, hmacComputed); 806 results.put("STATUS", hmacIsValid); 807 results.put("RESULT", String.format(resultFormatTemplate, parts[0].trim(), encodedHmacComputed)); 808 } 809 default -> throw new IllegalArgumentException("Not supported processing mode!"); 810 } 811 return results; 812 } 813 814 /** 815 * Apply a collection of validations on a JSON string provided: 816 * <ul> 817 * <li>Real JSON structure.</li> 818 * <li>Contain less than a specified number of deepness for nested objects or arrays.</li> 819 * <li>Contain less than a specified number of items in any arrays.</li> 820 * </ul> 821 * <br> 822 * <b>Note:</b> I decided to use a parsing approach using only string processing to prevent any StackOverFlow or OutOfMemory error that can be abused.<br><br> 823 * I used the following assumption: 824 * <ul> 825 * <li>The character <code>{</code> identify the beginning of an object.</li> 826 * <li>The character <code>}</code> identify the end of an object.</li> 827 * <li>The character <code>[</code> identify the beginning of an array.</li> 828 * <li>The character <code>]</code> identify the end of an array.</li> 829 * <li>The character <code>"</code> identify the delimiter of a string.</li> 830 * <li>The character sequence <code>\"</code> identify the escaping of an double quote.</li> 831 * </ul> 832 * 833 * @param json String containing the JSON data to validate. 834 * @param maxItemsByArraysCount Maximum number of items allowed in an array. 835 * @param maxDeepnessAllowed Maximum number nested objects or arrays allowed. 836 * @return True only if the string pass all validations. 837 * @see "https://javaee.github.io/jsonp/" 838 * @see "https://community.f5.com/discussions/technicalforum/disable-buffer-overflow-in-json-parameters/124306" 839 * @see "https://github.com/InductiveComputerScience/pbJson/issues/2" 840 */ 841 public static boolean isJSONSafe(String json, int maxItemsByArraysCount, int maxDeepnessAllowed) { 842 boolean isSafe = false; 843 844 try { 845 //Step 1: Analyse the JSON string 846 int currentDeepness = 0; 847 int currentArrayItemsCount = 0; 848 int maxDeepnessReached = 0; 849 int maxArrayItemsCountReached = 0; 850 boolean currentlyInArray = false; 851 boolean currentlyInString = false; 852 int currentNestedArrayLevel = 0; 853 String jsonEscapedDoubleQuote = "\\\"";//Escaped double quote must not be considered as a string delimiter 854 String work = json.replace(jsonEscapedDoubleQuote, "'"); 855 for (char c : work.toCharArray()) { 856 switch (c) { 857 case '{': { 858 if (!currentlyInString) { 859 currentDeepness++; 860 } 861 break; 862 } 863 case '}': { 864 if (!currentlyInString) { 865 currentDeepness--; 866 } 867 break; 868 } 869 case '[': { 870 if (!currentlyInString) { 871 currentDeepness++; 872 if (currentlyInArray) { 873 currentNestedArrayLevel++; 874 } 875 currentlyInArray = true; 876 } 877 break; 878 } 879 case ']': { 880 if (!currentlyInString) { 881 currentDeepness--; 882 currentArrayItemsCount = 0; 883 if (currentNestedArrayLevel > 0) { 884 currentNestedArrayLevel--; 885 } 886 if (currentNestedArrayLevel == 0) { 887 currentlyInArray = false; 888 } 889 } 890 break; 891 } 892 case '"': { 893 currentlyInString = !currentlyInString; 894 break; 895 } 896 case ',': { 897 if (!currentlyInString && currentlyInArray) { 898 currentArrayItemsCount++; 899 } 900 break; 901 } 902 } 903 if (currentDeepness > maxDeepnessReached) { 904 maxDeepnessReached = currentDeepness; 905 } 906 if (currentArrayItemsCount > maxArrayItemsCountReached) { 907 maxArrayItemsCountReached = currentArrayItemsCount; 908 } 909 } 910 //Step 2: Apply validation against the value specified as limits 911 isSafe = ((maxItemsByArraysCount > maxArrayItemsCountReached) && (maxDeepnessAllowed > maxDeepnessReached)); 912 913 //Step 3: If the content is safe then ensure that it is valid JSON structure using the "Java API for JSON Processing" (JSR 374) parser reference implementation. 914 if (isSafe) { 915 JsonReader reader = Json.createReader(new StringReader(json)); 916 isSafe = (reader.read() != null); 917 } 918 919 } catch (Exception e) { 920 isSafe = false; 921 } 922 return isSafe; 923 } 924 925 /** 926 * Apply a collection of validations on a image file provided: 927 * <ul> 928 * <li>Real image file.</li> 929 * <li>Its mime type is into the list of allowed mime types.</li> 930 * <li>Its metadata fields do not contains any characters related to a malicious payloads.</li> 931 * </ul> 932 * <br> 933 * <b>Important note:</b> This implementation is prone to bypass using the "<b>raw insertion</b>" method documented in the <a href="https://www.synacktiv.com/en/publications/persistent-php-payloads-in-pngs-how-to-inject-php-code-in-an-image-and-keep-it-there">blog post</a> from the Synacktiv team. 934 * To handle such case, it is recommended to resize the image to remove any non image-related content, see <a href="https://github.com/righettod/document-upload-protection/blob/master/src/main/java/eu/righettod/poc/sanitizer/ImageDocumentSanitizerImpl.java#L54">here</a> for an example.<br> 935 * 936 * @param imageFilePath Filename of the image file to check. 937 * @param imageAllowedMimeTypes List of image mime types allowed. 938 * @return True only if the file pass all validations. 939 * @see "https://commons.apache.org/proper/commons-imaging/" 940 * @see "https://commons.apache.org/proper/commons-imaging/formatsupport.html" 941 * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Basics_of_HTTP/MIME_types/Common_types" 942 * @see "https://www.iana.org/assignments/media-types/media-types.xhtml#image" 943 * @see "https://www.synacktiv.com/en/publications/persistent-php-payloads-in-pngs-how-to-inject-php-code-in-an-image-and-keep-it-there" 944 * @see "https://cheatsheetseries.owasp.org/cheatsheets/File_Upload_Cheat_Sheet.html" 945 * @see "https://github.com/righettod/document-upload-protection/blob/master/src/main/java/eu/righettod/poc/sanitizer/ImageDocumentSanitizerImpl.java" 946 * @see "https://exiftool.org/examples.html" 947 * @see "https://en.wikipedia.org/wiki/List_of_file_signatures" 948 * @see "https://hexed.it/" 949 * @see "https://github.com/sighook/pixload" 950 */ 951 public static boolean isImageSafe(String imageFilePath, List<String> imageAllowedMimeTypes) { 952 boolean isSafe = false; 953 Pattern payloadDetectionRegex = Pattern.compile("[<>${}`]+", Pattern.CASE_INSENSITIVE); 954 try { 955 File imgFile = new File(imageFilePath); 956 if (imgFile.exists() && imgFile.canRead() && imgFile.isFile() && !imageAllowedMimeTypes.isEmpty()) { 957 final byte[] imgBytes = Files.readAllBytes(imgFile.toPath()); 958 //Step 1: Check the mime type of the file against the allowed ones 959 ImageInfo imgInfo = Imaging.getImageInfo(imgBytes); 960 if (imageAllowedMimeTypes.contains(imgInfo.getMimeType())) { 961 //Step 2: Load the image into an object using the Image API 962 BufferedImage imgObject = Imaging.getBufferedImage(imgBytes); 963 if (imgObject != null && imgObject.getWidth() > 0 && imgObject.getHeight() > 0) { 964 //Step 3: Check the metadata if the image format support it - Highly experimental 965 List<String> metadataWithPayloads = new ArrayList<>(); 966 final ImageMetadata imgMetadata = Imaging.getMetadata(imgBytes); 967 if (imgMetadata != null) { 968 imgMetadata.getItems().forEach(item -> { 969 String metadata = item.toString(); 970 if (payloadDetectionRegex.matcher(metadata).find()) { 971 metadataWithPayloads.add(metadata); 972 } 973 }); 974 } 975 isSafe = metadataWithPayloads.isEmpty(); 976 } 977 } 978 } 979 } catch (Exception e) { 980 isSafe = false; 981 } 982 return isSafe; 983 } 984 985 /** 986 * Rewrite the input file to remove any embedded files that is not embedded using a methods supported by the official format of the file.<br> 987 * Example: a file can be embedded by adding it to the end of the source file, see the reference provided for details. 988 * 989 * @param inputFilePath Filename of the file to clean up. 990 * @param inputFileType Type of the file provided. 991 * @return A array of bytes with the cleaned file. 992 * @throws IllegalArgumentException If an invalid parameter is passed 993 * @throws Exception If any technical error during the cleaning processing 994 * @see "https://www.synacktiv.com/en/publications/persistent-php-payloads-in-pngs-how-to-inject-php-code-in-an-image-and-keep-it-there" 995 * @see "https://github.com/righettod/toolbox-pentest-web/tree/master/misc" 996 * @see "https://github.com/righettod/toolbox-pentest-web?tab=readme-ov-file#misc" 997 * @see "https://stackoverflow.com/a/13605411" 998 */ 999 public static byte[] sanitizeFile(String inputFilePath, InputFileType inputFileType) throws Exception { 1000 ByteArrayOutputStream sanitizedContent = new ByteArrayOutputStream(); 1001 File inputFile = new File(inputFilePath); 1002 if (!inputFile.exists() || !inputFile.canRead() || !inputFile.isFile()) { 1003 throw new IllegalArgumentException("Cannot read the content of the input file!"); 1004 } 1005 switch (inputFileType) { 1006 case PDF -> { 1007 try (PDDocument document = Loader.loadPDF(inputFile)) { 1008 document.save(sanitizedContent); 1009 } 1010 } 1011 case IMAGE -> { 1012 // Load the original image 1013 BufferedImage originalImage = ImageIO.read(inputFile); 1014 String originalFormat = identifyMimeType(Files.readAllBytes(inputFile.toPath())).split("/")[1].trim(); 1015 // Check that image has been successfully loaded 1016 if (originalImage == null) { 1017 throw new IOException("Cannot load the original image !"); 1018 } 1019 // Get current Width and Height of the image 1020 int originalWidth = originalImage.getWidth(null); 1021 int originalHeight = originalImage.getHeight(null); 1022 // Resize the image by removing 1px on Width and Height 1023 Image resizedImage = originalImage.getScaledInstance(originalWidth - 1, originalHeight - 1, Image.SCALE_SMOOTH); 1024 // Resize the resized image by adding 1px on Width and Height - In fact set image to is initial size 1025 Image initialSizedImage = resizedImage.getScaledInstance(originalWidth, originalHeight, Image.SCALE_SMOOTH); 1026 // Save image to a bytes buffer 1027 int bufferedImageType = BufferedImage.TYPE_INT_ARGB;//By default use a format supporting transparency 1028 //Sometimes for BMP, the format detected is "bmp; format=compressed" 1029 if ("jpeg".equalsIgnoreCase(originalFormat) || "bmp".equalsIgnoreCase(originalFormat) || originalFormat.startsWith("bmp;")) { 1030 bufferedImageType = BufferedImage.TYPE_INT_RGB; 1031 } 1032 BufferedImage sanitizedImage = new BufferedImage(initialSizedImage.getWidth(null), initialSizedImage.getHeight(null), bufferedImageType); 1033 Graphics2D drawer = sanitizedImage.createGraphics(); 1034 drawer.drawImage(initialSizedImage, 0, 0, null); 1035 drawer.dispose(); 1036 //Handle "bmp; format=compressed" case 1037 String formatToUse = originalFormat; 1038 if (formatToUse.startsWith("bmp;")) { 1039 formatToUse = formatToUse.split(";")[0].trim(); 1040 } 1041 ImageIO.write(sanitizedImage, formatToUse, sanitizedContent); 1042 } 1043 default -> throw new IllegalArgumentException("Type of file not supported !"); 1044 } 1045 if (sanitizedContent.size() == 0) { 1046 throw new IOException("An error occur during the rewrite operation!"); 1047 } 1048 return sanitizedContent.toByteArray(); 1049 } 1050 1051 /** 1052 * Apply a collection of validations on a string expected to be an email address: 1053 * <ul> 1054 * <li>Is a valid email address, from a parser perspective, following RFCs on email addresses.</li> 1055 * <li>Is not using "Encoded-word" format.</li> 1056 * <li>Is not using comment format.</li> 1057 * <li>Is not using "Punycode" format.</li> 1058 * <li>Is not using UUCP style addresses.</li> 1059 * <li>Is not using address literals.</li> 1060 * <li>Is not using source routes.</li> 1061 * <li>Is not using the "percent hack".</li> 1062 * <li>Does not contain newline or carriage-return characters (CRLF injection prevention).</li> 1063 * <li>The domain part contains at least one dot (reject single-label domains such as localhost or internal hostnames).</li> 1064 * <li>The local part is not a quoted string (i.e. not wrapped in double quotes).</li> 1065 * <li>Respect the RFC 5321 length limits: local part ≤ 64 characters, domain ≤ 255 characters, total address ≤ 320 characters.</li> 1066 * </ul><br> 1067 * This is based on the research work from <a href="https://portswigger.net/research/gareth-heyes">Gareth Heyes</a> added in references (Portswigger).<br><br> 1068 * 1069 * <b>Note:</b> The notion of valid, here, is to take from a secure usage of the data perspective. 1070 * 1071 * @param addr String expected to be a valid email address. 1072 * @return True only if the string pass all validations. 1073 * @see "https://commons.apache.org/proper/commons-validator/" 1074 * @see "https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/EmailValidator.html" 1075 * @see "https://datatracker.ietf.org/doc/html/rfc2047#section-2" 1076 * @see "https://portswigger.net/research/splitting-the-email-atom" 1077 * @see "https://www.jochentopf.com/email/address.html" 1078 * @see "https://en.wikipedia.org/wiki/Email_address" 1079 */ 1080 public static boolean isEmailAddress(String addr) { 1081 boolean isValid = false; 1082 String work = addr.toLowerCase(Locale.ROOT); 1083 Pattern encodedWordRegex = Pattern.compile("[=?]+", Pattern.CASE_INSENSITIVE); 1084 Pattern forbiddenCharacterRegex = Pattern.compile("[():!%\\[\\],;\"\n\r]+", Pattern.CASE_INSENSITIVE); 1085 try { 1086 //Start with the use of the dedicated EmailValidator from Apache Commons Validator 1087 if (EmailValidator.getInstance(true, true).isValid(work)) { 1088 //If OK then validate it does not contains "Encoded-word" patterns using an aggressive approach 1089 if (!encodedWordRegex.matcher(work).find()) { 1090 //If OK then validate it does not contains punycode 1091 if (!work.contains("xn--")) { 1092 //If OK then validate it does not use: 1093 // UUCP style addresses, 1094 // Comment format, 1095 // Address literals, 1096 // Source routes, 1097 // The percent hack. 1098 if (!forbiddenCharacterRegex.matcher(work).find()) { 1099 //If OK ensure that the domain part contains at least one dot 1100 long arobaseCount = addr.chars().filter(c -> c == '@').count(); 1101 if (arobaseCount == 1) { 1102 String[] parts = addr.split("@"); 1103 String localPart = parts[0]; 1104 String domainPart = parts[1]; 1105 if (domainPart.contains(".")) { 1106 //If OK the check the respect to the RFC 5321 length limits: 1107 // local part ≤ 64 characters, domain ≤ 255 characters, total address ≤ 320 characters. 1108 if (localPart.length() <= 64 && domainPart.length() <= 255 && addr.length() <= 320) { 1109 isValid = true; 1110 } 1111 } 1112 } 1113 } 1114 } 1115 } 1116 1117 } 1118 } catch (Exception e) { 1119 isValid = false; 1120 } 1121 return isValid; 1122 } 1123 1124 /** 1125 * The <a href="https://www.stet.eu/en/psd2/">PSD2 STET</a> specification require to use <a href="https://datatracker.ietf.org/doc/draft-cavage-http-signatures/">HTTP Signature</a>. 1126 * <br> 1127 * Section <b>3.5.1.2</b> of the document <a href="https://www.stet.eu/assets/files/PSD2/1-6-3/api-dsp2-stet-v1.6.3.1-part-1-framework.pdf">Documentation Framework</a> version <b>1.6.3</b>. 1128 * <br> 1129 * The problem is that, by design, the HTTP Signature specification is prone to blind SSRF. 1130 * <br> 1131 * URL example taken from the STET specification: <code>https://path.to/myQsealCertificate_714f8154ec259ac40b8a9786c9908488b2582b68b17e865fede4636d726b709f</code>. 1132 * <br> 1133 * The objective of this code is to try to decrease the "exploitability/interest" of this SSRF for an attacker. 1134 * 1135 * @param certificateUrl Url pointing to a Qualified Certificate (QSealC) encoded in PEM format and respecting the ETSI/TS119495 technical Specification . 1136 * @return TRUE only if the url point to a Qualified Certificate in PEM format. 1137 * @see "https://www.stet.eu/en/psd2/" 1138 * @see "https://www.stet.eu/assets/files/PSD2/1-6-3/api-dsp2-stet-v1.6.3.1-part-1-framework.pdf" 1139 * @see "https://datatracker.ietf.org/doc/draft-cavage-http-signatures/" 1140 * @see "https://datatracker.ietf.org/doc/rfc9421/" 1141 * @see "https://openjdk.org/groups/net/httpclient/intro.html" 1142 * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.net.http/java/net/http/package-summary.html" 1143 * @see "https://portswigger.net/web-security/ssrf" 1144 * @see "https://developer.mozilla.org/en-US/docs/Web/HTTP/Headers/Cache-Control" 1145 */ 1146 public static boolean isPSD2StetSafeCertificateURL(String certificateUrl) { 1147 boolean isValid = false; 1148 long connectionTimeoutInSeconds = 10; 1149 String userAgent = "PSD2-STET-HTTPSignature-CertificateRequest"; 1150 try { 1151 //1. Ensure that the URL end with the SHA-256 fingerprint encoded in HEX of the certificate like requested by STET 1152 if (certificateUrl != null && certificateUrl.lastIndexOf("_") != -1) { 1153 String digestPart = certificateUrl.substring(certificateUrl.lastIndexOf("_") + 1); 1154 if (Pattern.matches("^[0-9a-f]{64}$", digestPart)) { 1155 //2. Ensure that the URL is a valid url by creating a instance of the class URI 1156 URI uri = URI.create(certificateUrl); 1157 //3. Require usage of HTTPS and reject any url containing query parameters 1158 if ("https".equalsIgnoreCase(uri.getScheme()) && uri.getQuery() == null) { 1159 //4. Perform a HTTP HEAD request in order to get the content type of the remote resource 1160 //and limit the interest to use the SSRF because to pass the check the url need to: 1161 //- Do not having any query parameters. 1162 //- Use HTTPS protocol. 1163 //- End with a string having the format "_[0-9a-f]{64}". 1164 //- Trigger the malicious action that the attacker want but with a HTTP HEAD without any redirect and parameters. 1165 HttpResponse<String> response; 1166 try (HttpClient client = HttpClient.newBuilder().followRedirects(HttpClient.Redirect.NEVER).build()) { 1167 HttpRequest request = HttpRequest.newBuilder().uri(uri).timeout(Duration.ofSeconds(connectionTimeoutInSeconds)).method("HEAD", HttpRequest.BodyPublishers.noBody()).header("User-Agent", userAgent)//To provide an hint to the target about the initiator of the request 1168 .header("Cache-Control", "no-store, max-age=0")//To prevent caching issues or abuses 1169 .build(); 1170 response = client.send(request, HttpResponse.BodyHandlers.ofString()); 1171 if (response.statusCode() == 200) { 1172 //5. Ensure that the response content type is "text/plain" 1173 Optional<String> contentType = response.headers().firstValue("Content-Type"); 1174 isValid = (contentType.isPresent() && contentType.get().trim().toLowerCase(Locale.ENGLISH).startsWith("text/plain")); 1175 } 1176 } 1177 } 1178 } 1179 } 1180 } catch (Exception e) { 1181 isValid = false; 1182 } 1183 return isValid; 1184 } 1185 1186 /** 1187 * Perform sequential URL decoding operations against a URL encoded data until the data is not URL encoded anymore or if the specified threshold is reached. 1188 * 1189 * @param encodedData URL encoded data. 1190 * @param decodingRoundThreshold Threshold above which decoding will fail. 1191 * @return The decoded data. 1192 * @throws SecurityException If the threshold is reached. 1193 * @see "https://en.wikipedia.org/wiki/Percent-encoding" 1194 * @see "https://owasp.org/www-community/Double_Encoding" 1195 * @see "https://portswigger.net/web-security/essential-skills/obfuscating-attacks-using-encodings" 1196 * @see "https://capec.mitre.org/data/definitions/120.html" 1197 */ 1198 public static String applyURLDecoding(String encodedData, int decodingRoundThreshold) throws SecurityException { 1199 if (decodingRoundThreshold < 1) { 1200 throw new IllegalArgumentException("Threshold must be a positive number !"); 1201 } 1202 if (encodedData == null) { 1203 throw new IllegalArgumentException("Data provided must not be null !"); 1204 } 1205 Charset charset = StandardCharsets.UTF_8; 1206 int currentDecodingRound = 0; 1207 boolean isFinished = false; 1208 String currentRoundData = encodedData; 1209 String previousRoundData = encodedData; 1210 while (!isFinished) { 1211 if (currentDecodingRound > decodingRoundThreshold) { 1212 throw new SecurityException(String.format("Decoding round threshold of %s reached!", decodingRoundThreshold)); 1213 } 1214 currentRoundData = URLDecoder.decode(currentRoundData, charset); 1215 isFinished = currentRoundData.equals(previousRoundData); 1216 previousRoundData = currentRoundData; 1217 currentDecodingRound++; 1218 } 1219 return currentRoundData; 1220 } 1221 1222 /** 1223 * Apply a collection of validations on a string expected to be an system file/folder path: 1224 * <ul> 1225 * <li>Does not contains path traversal payload.</li> 1226 * <li>The canonical path is equals to the absolute path.</li> 1227 * </ul><br> 1228 * 1229 * @param path String expected to be a valid system file/folder path. 1230 * @return True only if the string pass all validations. 1231 * @see "https://portswigger.net/web-security/file-path-traversal" 1232 * @see "https://learn.snyk.io/lesson/directory-traversal/" 1233 * @see "https://capec.mitre.org/data/definitions/126.html" 1234 * @see "https://owasp.org/www-community/attacks/Path_Traversal" 1235 */ 1236 public static boolean isPathSafe(String path) { 1237 boolean isSafe = false; 1238 int decodingRoundThreshold = 3; 1239 try { 1240 if (path != null && !path.isEmpty()) { 1241 //URL decode the path if case of data coming from a web context 1242 String decodedPath = applyURLDecoding(path, decodingRoundThreshold); 1243 //Ensure that no path traversal expression is present 1244 if (!decodedPath.contains("..")) { 1245 File f = new File(decodedPath); 1246 String canonicalPath = f.getCanonicalPath(); 1247 String absolutePath = f.getAbsolutePath(); 1248 isSafe = canonicalPath.equals(absolutePath); 1249 } 1250 } 1251 } catch (Exception e) { 1252 isSafe = false; 1253 } 1254 return isSafe; 1255 } 1256 1257 /** 1258 * Identify if an XML contains any XML comments or have any XSL processing instructions.<br> 1259 * Stream reader based parsing is used to support large XML tree. 1260 * 1261 * @param xmlFilePath Filename of the XML file to check. 1262 * @return True only if XML comments or XSL processing instructions are identified. 1263 * @see "https://www.tutorialspoint.com/xml/xml_processing.htm" 1264 * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.xml/javax/xml/stream/XMLInputFactory.html" 1265 * @see "https://portswigger.net/kb/issues/00400700_xml-entity-expansion" 1266 * @see "https://www.w3.org/Style/styling-XML.en.html" 1267 */ 1268 public static boolean isXMLHaveCommentsOrXSLProcessingInstructions(String xmlFilePath) { 1269 boolean itemsDetected = false; 1270 try { 1271 //Ensure that the parser will not be prone XML external entity (XXE) injection or XML entity expansion (XEE) attacks 1272 XMLInputFactory xmlInputFactory = XMLInputFactory.newFactory(); 1273 xmlInputFactory.setProperty(XMLInputFactory.SUPPORT_DTD, false); 1274 xmlInputFactory.setProperty(XMLConstants.ACCESS_EXTERNAL_DTD, ""); 1275 xmlInputFactory.setProperty(XMLInputFactory.IS_REPLACING_ENTITY_REFERENCES, false); 1276 xmlInputFactory.setProperty(XMLInputFactory.IS_SUPPORTING_EXTERNAL_ENTITIES, false); 1277 1278 //Parse file 1279 try (FileInputStream fis = new FileInputStream(xmlFilePath)) { 1280 XMLStreamReader reader = xmlInputFactory.createXMLStreamReader(fis); 1281 int eventType; 1282 while (reader.hasNext() && !itemsDetected) { 1283 eventType = reader.next(); 1284 if (eventType == XMLEvent.COMMENT) { 1285 itemsDetected = true; 1286 } else if (eventType == XMLEvent.PROCESSING_INSTRUCTION && "xml-stylesheet".equalsIgnoreCase(reader.getPITarget())) { 1287 itemsDetected = true; 1288 } 1289 } 1290 } 1291 } catch (Exception e) { 1292 //In case of error then assume that the check failed 1293 itemsDetected = true; 1294 } 1295 return itemsDetected; 1296 } 1297 1298 1299 /** 1300 * Perform a set of additional validations against a JWT token: 1301 * <ul> 1302 * <li>Do not use the <b>NONE</b> signature algorithm.</li> 1303 * <li>Have a <a href="https://www.iana.org/assignments/jwt/jwt.xhtml">EXP claim</a> defined.</li> 1304 * <li>The token identifier (<a href="https://www.iana.org/assignments/jwt/jwt.xhtml">JTI claim</a>) is NOT part of the list of revoked token.</li> 1305 * <li>Match the expected type of token: ACCESS or ID or REFRESH.</li> 1306 * </ul> 1307 * 1308 * @param token JWT token for which <b>signature was already validated</b> and on which a set of additional validations will be applied. 1309 * @param expectedTokenType The type of expected token using the enumeration provided. 1310 * @param revokedTokenJTIList A list of token identifier (<b>JTI</b> claim) referring to tokens that were revoked and to which the JTI claim of the token will be compared to. 1311 * @return True only the token pass all the validations. 1312 * @see "https://www.iana.org/assignments/jwt/jwt.xhtml" 1313 * @see "https://auth0.com/docs/secure/tokens/access-tokens" 1314 * @see "https://auth0.com/docs/secure/tokens/id-tokens" 1315 * @see "https://auth0.com/docs/secure/tokens/refresh-tokens" 1316 * @see "https://auth0.com/blog/id-token-access-token-what-is-the-difference/" 1317 * @see "https://jwt.io/libraries?language=Java" 1318 * @see "https://pentesterlab.com/blog/secure-jwt-library-design" 1319 * @see "https://github.com/auth0/java-jwt" 1320 */ 1321 public static boolean applyJWTExtraValidation(DecodedJWT token, TokenType expectedTokenType, List<String> revokedTokenJTIList) { 1322 boolean isValid = false; 1323 TokenType tokenType; 1324 try { 1325 if (!"none".equalsIgnoreCase(token.getAlgorithm().trim())) { 1326 if (!token.getClaim("exp").isMissing() && token.getExpiresAt() != null) { 1327 String jti = token.getId(); 1328 if (jti != null && !jti.trim().isEmpty()) { 1329 boolean jtiIsRevoked = revokedTokenJTIList.stream().anyMatch(jti::equalsIgnoreCase); 1330 if (!jtiIsRevoked) { 1331 //Determine the token type based on the presence of specifics claims 1332 if (!token.getClaim("scope").isMissing()) { 1333 tokenType = TokenType.ACCESS; 1334 } else if (!token.getClaim("name").isMissing() || !token.getClaim("email").isMissing()) { 1335 tokenType = TokenType.ID; 1336 } else { 1337 tokenType = TokenType.REFRESH; 1338 } 1339 isValid = (tokenType.equals(expectedTokenType)); 1340 } 1341 } 1342 } 1343 } 1344 1345 } catch (Exception e) { 1346 //In case of error then assume that the check failed 1347 isValid = false; 1348 } 1349 return isValid; 1350 } 1351 1352 /** 1353 * Apply a validations on a regular expression to ensure that is not prone to the ReDOS attack. 1354 * <br>If your technology is supported by <a href="https://github.com/doyensec/regexploit">regexploit</a> then <b>use it instead of this method!</b> 1355 * <br>Indeed, the <a href="https://www.doyensec.com/">Doyensec</a> team has made an intensive and amazing work on this topic and created this effective tool. 1356 * 1357 * @param regex String expected to be a valid regular expression (regex). 1358 * @param data Test data on which the regular expression is executed for the test. 1359 * @param maximumRunningTimeInSeconds Optional parameter to specify a number of seconds above which a regex execution time is considered as not safe (default to 4 seconds when not specified). 1360 * @return True only if the string pass all validations. 1361 * @see "https://github.blog/security/how-to-fix-a-redos/" 1362 * @see "https://learn.snyk.io/lesson/redos" 1363 * @see "https://rules.sonarsource.com/java/RSPEC-2631/" 1364 * @see "https://github.com/doyensec/regexploit" 1365 * @see "https://github.com/makenowjust-labs/recheck" 1366 * @see "https://github.com/tjenkinson/redos-detector" 1367 * @see "https://wiki.owasp.org/images/2/23/OWASP_IL_2009_ReDoS.pdf" 1368 * @see "https://owasp.org/www-community/attacks/Regular_expression_Denial_of_Service_-_ReDoS" 1369 */ 1370 public static boolean isRegexSafe(String regex, String data, Optional<Integer> maximumRunningTimeInSeconds) { 1371 Objects.requireNonNull(maximumRunningTimeInSeconds, "Use 'Optional.empty()' to leverage the default value."); 1372 Objects.requireNonNull(data, "A sample data is needed to perform the test."); 1373 Objects.requireNonNull(regex, "A regular expression is needed to perform the test."); 1374 boolean isSafe = false; 1375 int executionTimeout = maximumRunningTimeInSeconds.orElse(4); 1376 ExecutorService executor = Executors.newSingleThreadExecutor(); 1377 try { 1378 Callable<Boolean> task = () -> { 1379 Pattern pattern = Pattern.compile(regex); 1380 return pattern.matcher(data).matches(); 1381 }; 1382 List<Future<Boolean>> tasks = executor.invokeAll(List.of(task), executionTimeout, TimeUnit.SECONDS); 1383 if (!tasks.getFirst().isCancelled()) { 1384 isSafe = true; 1385 } 1386 } catch (Exception e) { 1387 isSafe = false; 1388 } finally { 1389 executor.shutdownNow(); 1390 } 1391 return isSafe; 1392 } 1393 1394 /** 1395 * Compute a UUID version 7 without using any external dependency.<br><br> 1396 * <b>Below are my personal point of view and perhaps I'm totally wrong!</b> 1397 * <br><br> 1398 * Why such method? 1399 * <ul> 1400 * <li>Java inferior or equals to 21 does not supports natively the generation of an UUID version 7.</li> 1401 * <li>Import a library just to generate such value is overkill for me.</li> 1402 * <li>Library that I have found, generating such version of an UUID, are not provided by entities commonly used in the java world, such as the SPRING framework provider.</li> 1403 * </ul> 1404 * <br> 1405 * <b>Full credits for this implementation goes to the authors and contributors of the <a href="https://github.com/nalgeon/uuidv7">UUIDv7</a> project.</b> 1406 * <br><br> 1407 * Below are the java libraries that I have found but, for which, I do not trust enough the provider to use them directly: 1408 * <ul> 1409 * <li><a href="https://github.com/cowtowncoder/java-uuid-generator">java-uuid-generator</a></li> 1410 * <li><a href="https://github.com/f4b6a3/uuid-creator">uuid-creator</a></li> 1411 * </ul> 1412 * 1413 * @return A UUID object representing the UUID v7. 1414 * @see "https://uuid7.com/" 1415 * @see "https://antonz.org/uuidv7/" 1416 * @see "https://mccue.dev/pages/3-11-25-life-altering-postgresql-patterns" 1417 * @see "https://www.ietf.org/archive/id/draft-peabody-dispatch-new-uuid-format-04.html#name-uuid-version-7" 1418 * @see "https://www.baeldung.com/java-generating-time-based-uuids" 1419 * @see "https://en.wikipedia.org/wiki/Universally_unique_identifier" 1420 * @see "https://buildkite.com/resources/blog/goodbye-integers-hello-uuids/" 1421 */ 1422 public static UUID computeUUIDv7() { 1423 SecureRandom secureRandom = new SecureRandom(); 1424 // Generate truly random bytes 1425 byte[] value = new byte[16]; 1426 secureRandom.nextBytes(value); 1427 // Get current timestamp in milliseconds 1428 ByteBuffer timestamp = ByteBuffer.allocate(Long.BYTES); 1429 timestamp.putLong(System.currentTimeMillis()); 1430 // Create the TIMESTAMP part of the UUID 1431 System.arraycopy(timestamp.array(), 2, value, 0, 6); 1432 // Create the VERSION and the VARIANT parts of the UUID 1433 value[6] = (byte) ((value[6] & 0x0F) | 0x70); 1434 value[8] = (byte) ((value[8] & 0x3F) | 0x80); 1435 //Create the HIGH and LOW parts of the UUID 1436 ByteBuffer buf = ByteBuffer.wrap(value); 1437 long high = buf.getLong(); 1438 long low = buf.getLong(); 1439 //Create and return the UUID object 1440 UUID uuidv7 = new UUID(high, low); 1441 return uuidv7; 1442 } 1443 1444 /** 1445 * Ensure that an XSD file does not contain any include/import/redefine instruction (prevent exposure to SSRF). 1446 * 1447 * @param xsdFilePath Filename of the XSD file to check. 1448 * @return True only if the file pass all validations. 1449 * @see "https://portswigger.net/web-security/ssrf" 1450 * @see "https://www.w3schools.com/Xml/el_import.asp" 1451 * @see "https://www.w3schools.com/xml/el_include.asp" 1452 * @see "https://www.linkedin.com/posts/righettod_appsec-appsecurity-java-activity-7344048434326188053-6Ru9" 1453 * @see "https://docs.oracle.com/en/java/javase/21/docs/api/java.xml/javax/xml/validation/SchemaFactory.html#setProperty(java.lang.String,java.lang.Object)" 1454 */ 1455 public static boolean isXSDSafe(String xsdFilePath) { 1456 boolean isSafe = false; 1457 try { 1458 File xsdFile = new File(xsdFilePath); 1459 if (xsdFile.exists() && xsdFile.canRead() && xsdFile.isFile()) { 1460 //Parse the XSD file, if an exception occur then it's imply that the XSD specified is not a valid ones 1461 //Create an schema factory throwing Exception if a external schema is specified 1462 SchemaFactory schemaFactory = SchemaFactory.newDefaultInstance(); 1463 schemaFactory.setProperty(XMLConstants.ACCESS_EXTERNAL_DTD, ""); 1464 schemaFactory.setProperty(XMLConstants.ACCESS_EXTERNAL_SCHEMA, ""); 1465 //Parse the schema 1466 Schema schema = schemaFactory.newSchema(xsdFile); 1467 isSafe = (schema != null); 1468 } 1469 } catch (Exception e) { 1470 isSafe = false; 1471 } 1472 return isSafe; 1473 } 1474 1475 1476 /** 1477 * Extract all sensitive information from a string provided.<br> 1478 * This can be used to identify any sensitive information into a <a href="https://cwe.mitre.org/data/definitions/532.html">message expected to be written in a log</a> and then replace every sensitive values by an obfuscated ones.<br><br> 1479 * For the luxembourg national identification number, this method focus on detecting identifiers for a physical entity (people) and not a moral one (company).<br><br> 1480 * I delegated the validation of the IBAN to a dedicated library (<a href="https://github.com/arturmkrtchyan/iban4j">iban4j</a>) to not "reinvent the wheel" and then introduce buggy validation myself. I used <b>iban4j</b> over the <b><a href="https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/IBANValidator.html">IBANValidator</a></b> class from the <a href="https://commons.apache.org/proper/commons-validator/"><b>Apache Commons Validator</b></a> library because <b>iban4j</b> perform a full official IBAN specification validation so its reduce risks of false-positives by ensuring that an IBAN detected is a real IBAN.<br><br> 1481 * Same thing and reason regarding the validation of the bank card PAN using the class <a href="https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/CreditCardValidator.html">CreditCardValidator</a> from the <b>Apache Commons Validator</b> library. 1482 * 1483 * @param content String in which sensitive information must be searched. 1484 * @return A map with the collection of identified sensitive information gathered by sensitive information type. If nothing is found then the map is empty. A type of sensitive information is only present if there is at least one item found. A set is used to not store duplicates occurrence of the same sensitive information. 1485 * @throws Exception If any error occurs during the processing. 1486 * @see "https://guichet.public.lu/en/citoyens/citoyennete/registre-national/identification/demande-numero-rnpp.html" 1487 * @see "https://cnpd.public.lu/fr/decisions-avis/2009/identifiant-unique.html" 1488 * @see "https://cnpd.public.lu/content/dam/cnpd/fr/decisions-avis/2009/identifiant-unique/48_2009.pdf" 1489 * @see "https://en.wikipedia.org/wiki/International_Bank_Account_Number" 1490 * @see "https://www.iban.com/structure" 1491 * @see "https://github.com/arturmkrtchyan/iban4j" 1492 * @see "https://cwe.mitre.org/data/definitions/532.html" 1493 * @see "https://www.baeldung.com/logback-mask-sensitive-data" 1494 * @see "https://en.wikipedia.org/wiki/Payment_card_number" 1495 * @see "https://commons.apache.org/proper/commons-validator/apidocs/org/apache/commons/validator/routines/CreditCardValidator.html" 1496 * @see "https://commons.apache.org/proper/commons-validator/" 1497 */ 1498 public static Map<SensitiveInformationType, Set<String>> extractAllSensitiveInformation(String content) throws Exception { 1499 CreditCardValidator creditCardValidator = CreditCardValidator.genericCreditCardValidator(); 1500 Pattern nationalIdentifierRegex = Pattern.compile("([0-9]{13})"); 1501 Pattern ibanNonHumanFormattedRegex = Pattern.compile("([A-Z]{2}[0-9]{2}[A-Z0-9]{11,30})", Pattern.CASE_INSENSITIVE); 1502 Pattern ibanHumanFormattedRegex = Pattern.compile("([A-Z]{2}[0-9]{2}(?:\\s[A-Z0-9]{4}){2,7}\\s[A-Z0-9]{1,4})", Pattern.CASE_INSENSITIVE); 1503 Pattern panRegex = Pattern.compile("((?:\\d[ -]*?){13,19})"); 1504 Map<SensitiveInformationType, Set<String>> data = new HashMap<>(); 1505 data.put(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER, new HashSet<>()); 1506 data.put(SensitiveInformationType.IBAN, new HashSet<>()); 1507 data.put(SensitiveInformationType.BANK_CARD_PAN, new HashSet<>()); 1508 1509 if (content != null && !content.isBlank()) { 1510 /* Step 1: Search for LU national identifier */ 1511 //A national identifier have the following structure: [BIRTHDATE_YEAR_YYYY][BIRTHDATE_MONTH_MM][BIRTHDATE_DAY_DD][FIVE_INTEGER] 1512 //Define minimal and maximal birth year base on current year 1513 //Assume people live less than 120 years 1514 int maxBirthYear = LocalDate.now(ZoneId.of("Europe/Luxembourg")).getYear(); 1515 int minBirthYear = maxBirthYear - 120; 1516 Matcher matcher = nationalIdentifierRegex.matcher(content); 1517 String nationalIdentierFull; 1518 int nationalIdentierYear, nationalIdentierMonth, nationalIdentierDay; 1519 while (matcher.find()) { 1520 nationalIdentierFull = matcher.group(1); 1521 //Check that the string is a valid national identifier and if yes then add it 1522 nationalIdentierYear = Integer.parseInt(nationalIdentierFull.substring(0, 4)); 1523 nationalIdentierMonth = Integer.parseInt(nationalIdentierFull.substring(4, 6)); 1524 nationalIdentierDay = Integer.parseInt(nationalIdentierFull.substring(6, 8)); 1525 if (nationalIdentierYear >= minBirthYear && nationalIdentierYear <= maxBirthYear) { 1526 if (nationalIdentierMonth >= 1 && nationalIdentierMonth <= 12) { 1527 if (YearMonth.of(nationalIdentierYear, nationalIdentierMonth).isValidDay(nationalIdentierDay)) { 1528 data.get(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER).add(nationalIdentierFull); 1529 } 1530 } 1531 } 1532 } 1533 1534 /* Step 2a: Search for IBAN that are non human formatted */ 1535 matcher = ibanNonHumanFormattedRegex.matcher(content); 1536 String iban, ibanUpperCased; 1537 while (matcher.find()) { 1538 iban = matcher.group(1); 1539 ibanUpperCased = iban.toUpperCase(Locale.ROOT); 1540 //Check that the string is a valid IBAN and if yes then add it 1541 if (IbanUtil.isValid(ibanUpperCased)) { 1542 data.get(SensitiveInformationType.IBAN).add(iban); 1543 } 1544 } 1545 1546 /* Step 2b: Search for IBAN that are human formatted */ 1547 matcher = ibanHumanFormattedRegex.matcher(content); 1548 String ibanUpperCasedNoSpace; 1549 while (matcher.find()) { 1550 iban = matcher.group(1); 1551 ibanUpperCasedNoSpace = iban.toUpperCase(Locale.ROOT).replace(" ", ""); 1552 //Check that the string is a valid IBAN and if yes then add it 1553 if (IbanUtil.isValid(ibanUpperCasedNoSpace)) { 1554 data.get(SensitiveInformationType.IBAN).add(iban); 1555 } 1556 } 1557 1558 /* Step 3: Search for bank card PAN */ 1559 matcher = panRegex.matcher(content); 1560 String pan, panNoSeparator; 1561 while (matcher.find()) { 1562 pan = matcher.group(1); 1563 panNoSeparator = pan.toUpperCase(Locale.ROOT).replace(" ", "").replace("-", ""); 1564 //Check that the string is a valid PAN and if yes then add it 1565 if (creditCardValidator.isValid(panNoSeparator)) { 1566 data.get(SensitiveInformationType.BANK_CARD_PAN).add(pan); 1567 } 1568 } 1569 1570 } 1571 1572 //Cleanup if a set is empty 1573 if (data.get(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER).isEmpty()) { 1574 data.remove(SensitiveInformationType.LUXEMBOURG_NATIONAL_IDENTIFICATION_NUMBER); 1575 } 1576 if (data.get(SensitiveInformationType.IBAN).isEmpty()) { 1577 data.remove(SensitiveInformationType.IBAN); 1578 } 1579 if (data.get(SensitiveInformationType.BANK_CARD_PAN).isEmpty()) { 1580 data.remove(SensitiveInformationType.BANK_CARD_PAN); 1581 } 1582 1583 return data; 1584 } 1585 1586 /** 1587 * Apply a collection of validations on a bytes array provided representing GZIP compressed data: 1588 * <ul> 1589 * <li>Are valid GZIP compressed data.</li> 1590 * <li>The number of bytes once decompressed is under the specified limit.</li> 1591 * </ul> 1592 * <br><b>Note:</b> The value <code>Integer.MAX_VALUE - 8</code> was chosen because during my tests on Java 25 (JDK 64 bits on Windows 11 Pro), it was possible to decompress such amount of data with the default JVM settings without causing an <a href="https://docs.oracle.com/en/java/javase/25/docs/api//java.base/java/lang/OutOfMemoryError.html">Out Of Memory error</a>. 1593 * 1594 * @param compressedBytes Array of bytes containing the GZIP compressed data to check. 1595 * @param maxCountOfDecompressedBytesAllowed Maximum number of decompressed bytes allowed. Default to 10 MB if the specified value is inferior to 1 or superior to Integer.MAX_VALUE - 8. 1596 * @return True only if the file pass all validations. 1597 * @see "https://en.wikipedia.org/wiki/Gzip" 1598 * @see "https://www.rapid7.com/db/modules/auxiliary/dos/http/gzip_bomb_dos/" 1599 */ 1600 public static boolean isGZIPCompressedDataSafe(byte[] compressedBytes, long maxCountOfDecompressedBytesAllowed) { 1601 boolean isSafe = false; 1602 1603 try { 1604 long limit = maxCountOfDecompressedBytesAllowed; 1605 long totalRead = 0L; 1606 byte[] buffer = new byte[8 * 1024]; 1607 int read; 1608 if (limit < 1 || limit > (Integer.MAX_VALUE - 8)) { 1609 limit = 10_000_000; 1610 } 1611 try (ByteArrayInputStream bis = new ByteArrayInputStream(compressedBytes); GZIPInputStream gzipInputStream = new GZIPInputStream(new BufferedInputStream(bis))) { 1612 while ((read = gzipInputStream.read(buffer)) != -1) { 1613 totalRead += read; 1614 if (totalRead > limit) { 1615 throw new Exception(); 1616 } 1617 } 1618 } 1619 isSafe = true; 1620 } catch (Exception e) { 1621 isSafe = false; 1622 } 1623 1624 return isSafe; 1625 } 1626 1627 /** 1628 * Process a string, intended to be written in a log, to remove as much as possible information that can lead to an exposure to a log injection vulnerability.<br><br> 1629 * <b>Log injection</b> is also called <b>log forging</b>.<br><br> 1630 * The following information are removed: 1631 * <ul> 1632 * <li>Characters: Carriage Return (CR), Linefeed (LF) and Tabulation (TAB).</li> 1633 * <li>Characters: Unicode LINE SEPARATOR and Unicode PARAGRAPH SEPARATOR.</li> 1634 * <li>Characters: CSI sequences and bare ESC.</li> 1635 * <li>Leading and trailing spaces.</li> 1636 * <li>Any HTML tags.</li> 1637 * </ul><br> 1638 * A parameter is also used to limit the maximum length of the sanitized message. 1639 * To remove any HTML tags, the OWASP project <a href="https://owasp.org/www-project-java-html-sanitizer/">Java HTML Sanitizer</a> is leveraged.<br> 1640 * I delegated such removal to a dedicated library to prevent missing of edge cases as well as potential bypasses. 1641 * 1642 * @param message The original string message intended to be written in a log. 1643 * @param maxMessageLength The maximum number of characters after which the sanitized message must be truncated. If inferior to 1 then default to the value of 500. 1644 * @return The string message cleaned. 1645 * @see "https://www.wallarm.com/what/log-forging-attack" 1646 * @see "https://www.invicti.com/learn/crlf-injection" 1647 * @see "https://knowledge-base.secureflag.com/vulnerabilities/inadequate_input_validation/log_injection_vulnerability.html" 1648 * @see "https://capec.mitre.org/data/definitions/93.html" 1649 * @see "https://codeql.github.com/codeql-query-help/javascript/js-log-injection/" 1650 * @see "https://owasp.org/www-project-java-html-sanitizer/" 1651 * @see "https://github.com/OWASP/java-html-sanitizer" 1652 */ 1653 public static String sanitizeLogMessage(String message, int maxMessageLength) { 1654 String sanitized = message; 1655 int maxSanitizedMessageLength = maxMessageLength; 1656 1657 if (sanitized != null && !sanitized.isBlank()) { 1658 if (maxSanitizedMessageLength < 1) { 1659 maxSanitizedMessageLength = 500; 1660 } 1661 //Step 1: Remove any CR/LR/TAB characters as well as leading and trailing spaces 1662 sanitized = sanitized.replaceAll("[\\n\\r\\t]", "").trim(); 1663 //Step 2: Remove any Unicode LINE SEPARATOR or Unicode PARAGRAPH SEPARATOR as well as leading and trailing spaces 1664 sanitized = sanitized.replace("\u2028", "").replace("\u2029", "").trim(); 1665 //Step 3: Remove ANSI escape sequences as well as leading and trailing spaces 1666 sanitized = sanitized.replaceAll("\u001B\\[[\\d;]*[a-zA-Z]", "").replace("\u001B", "").trim(); 1667 //Step 4: Remove any HTML tags 1668 PolicyFactory htmlSanitizerPolicy = new HtmlPolicyBuilder().toFactory(); 1669 sanitized = htmlSanitizerPolicy.sanitize(sanitized); 1670 //Step 5: Truncate the string in case of need 1671 if (sanitized.length() > maxSanitizedMessageLength) { 1672 sanitized = sanitized.substring(0, maxSanitizedMessageLength); 1673 } 1674 } 1675 1676 return sanitized; 1677 } 1678 1679 /** 1680 * Identify if an XML is an SVG image.<br> 1681 * The goal of this method is to prevent to leverage SVG, as an vector, to achieve a XSS when XML format is accepted.<br> 1682 * Leverage <a href="https://xmlgraphics.apache.org/batik/">Apache Batik</a> to delegate the parsing and support for the SVG format.<br><br> 1683 * <b>Due to the intended usage of the method, the following choice were made:</b> 1684 * <ul> 1685 * <li>Raise an exception when a non SVG related external references is identified.</li> 1686 * <li>Throw any exception that can occur if the provided content is invalid like for example an invalid XML file or a non existing file.</li> 1687 * <li>Explicitly check the XML prior to pass it to Batik even if Batik seems not prone to XXE/SSRF classes of vulnerability.</li> 1688 * </ul> 1689 * 1690 * @param xmlFilePath Filename of the XML file to check. 1691 * @return True only if XML is an valid SVG image. 1692 * @throws SecurityException If a non SVG external references is detected into the XML content. 1693 * @throws Exception If a error occur due to an invalid content provided. 1694 * @see "https://developer.mozilla.org/en-US/docs/Web/SVG" 1695 * @see "https://www.fortinet.com/blog/threat-research/scalable-vector-graphics-attack-surface-anatomy" 1696 * @see "https://portswigger.net/web-security/cross-site-scripting" 1697 * @see "https://xmlgraphics.apache.org/batik/" 1698 * @see "https://github.com/apache/xmlgraphics-batik/blob/main/batik-dom/src/main/java/org/apache/batik/dom/util/SAXDocumentFactory.java#L420" 1699 * @see "https://mvnrepository.com/artifact/org.apache.xmlgraphics/batik-dom" 1700 * @see "https://mvnrepository.com/artifact/org.apache.xmlgraphics/batik-anim" 1701 * @see "https://portswigger.net/web-security/xxe" 1702 * @see "https://portswigger.net/web-security/ssrf" 1703 */ 1704 public static boolean isXMLSVGImage(String xmlFilePath) throws Exception { 1705 boolean isSvg = true; 1706 List<String> svgValidSystemIDs = List.of("http://www.w3.org/Graphics/SVG/1.1/DTD/svg11.dtd", "http://www.w3.org/Graphics/SVG/1.1/DTD/svg11-basic.dtd", "http://www.w3.org/Graphics/SVG/1.1/DTD/svg11-tiny.dtd", "http://www.w3.org/TR/2001/REC-SVG-20010904/DTD/svg10.dtd"); 1707 1708 //Load the XML content into a reader 1709 String xmlContent = Files.readString(Paths.get(xmlFilePath)); 1710 //Then ensure that the XML document does not contains any non SVG external references 1711 try (Reader reader = StringReader.of(xmlContent)) { 1712 DocumentBuilderFactory xmlFactory = DocumentBuilderFactory.newInstance(); 1713 DocumentBuilder docBuilder = xmlFactory.newDocumentBuilder(); 1714 docBuilder.setEntityResolver((publicId, systemId) -> { 1715 if (systemId != null && !svgValidSystemIDs.contains(systemId)) { 1716 throw new SecurityException("External references detected: " + systemId); 1717 } 1718 return new InputSource(new ByteArrayInputStream("".getBytes())); 1719 }); 1720 docBuilder.parse(new InputSource(reader)); 1721 } 1722 //Then parse the XML with Apache Batik 1723 try (Reader reader = StringReader.of(xmlContent)) { 1724 //Method SAXDocumentFactory.createDocument() do not load external DTD or entities. 1725 String parserClassName = XMLResourceDescriptor.getXMLParserClassName(); 1726 SAXSVGDocumentFactory svgFactory = new SAXSVGDocumentFactory(parserClassName); 1727 //Method svgFactory.createSVGDocument() raise an IO exception if the XML is not a valid SVG image 1728 try { 1729 SVGDocument doc = svgFactory.createSVGDocument(null, reader); 1730 isSvg = (doc != null && doc.getRootElement() != null); 1731 } catch (IOException e) { 1732 isSvg = false; 1733 } 1734 } 1735 1736 return isSvg; 1737 } 1738}