001/* 002 * Licensed to the Apache Software Foundation (ASF) under one or more 003 * contributor license agreements. See the NOTICE file distributed with 004 * this work for additional information regarding copyright ownership. 005 * The ASF licenses this file to You under the Apache License, Version 2.0 006 * (the "License"); you may not use this file except in compliance with 007 * the License. You may obtain a copy of the License at 008 * 009 * https://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 014 * See the License for the specific language governing permissions and 015 * limitations under the License. 016 */ 017 018package org.apache.commons.codec.binary; 019 020import java.math.BigInteger; 021import java.util.Arrays; 022import java.util.Objects; 023 024import org.apache.commons.codec.CodecPolicy; 025 026/** 027 * Provides Base64 encoding and decoding as defined by <a href="https://www.ietf.org/rfc/rfc2045">RFC 2045 Multipurpose Internet Mail Extensions (MIME) Part 028 * One: Format of Internet Message Bodies</a> and portions of <a href="https://datatracker.ietf.org/doc/html/rfc4648">RFC 4648 The Base16, Base32, and Base64 029 * Data Encodings</a> 030 * 031 * <p> 032 * This class implements <a href="https://www.ietf.org/rfc/rfc2045#section-6.8">RFC 2045 6.8. Base64 Content-Transfer-Encoding</a>. 033 * </p> 034 * <p> 035 * The class can be parameterized in the following manner with its {@link Builder}: 036 * </p> 037 * <ul> 038 * <li>URL-safe mode: Default off.</li> 039 * <li>Line length: Default 76. Line length that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data. 040 * <li>Line separator: Default is CRLF ({@code "\r\n"})</li> 041 * <li>Strict or lenient decoding policy; default is {@link CodecPolicy#LENIENT}.</li> 042 * <li>Custom decoding table.</li> 043 * <li>Custom encoding table.</li> 044 * <li>Padding; defaults is {@code '='}.</li> 045 * </ul> 046 * <p> 047 * The URL-safe parameter selects the encoding alphabet. Lenient decoding seamlessly handles both modes; strict decoding requires the encoding alphabet. See also 048 * {@code Builder#setDecodeTableFormat(DecodeTableFormat)}. 049 * </p> 050 * <p> 051 * Since this class operates directly on byte streams, and not character streams, it is hard-coded to only encode/decode character encodings which are 052 * compatible with the lower 127 ASCII chart (ISO-8859-1, Windows-1252, UTF-8, etc). 053 * </p> 054 * <p> 055 * This class is thread-safe. 056 * </p> 057 * <p> 058 * To configure a new instance, use a {@link Builder}. For example: 059 * </p> 060 * 061 * <pre> 062 * Base64 base64 = Base64.builder() 063 * .setDecodingPolicy(CodecPolicy.LENIENT) // default is lenient, null resets to default 064 * .setEncodeTable(customEncodeTable) // default is built in, null resets to default 065 * .setLineLength(0) // default is none 066 * .setLineSeparator('\r', '\n') // default is CR LF, null resets to default 067 * .setPadding('=') // default is '=' 068 * .setUrlSafe(false) // default is false 069 * .get() 070 * </pre> 071 * 072 * <p> 073 * The static decoding convenience methods use {@link CodecPolicy#LENIENT}. They accept noncanonical input, so different encoded strings can decode to the 074 * same bytes. Selecting a standard or URL-safe decode table does not enable strict validation. To require canonical input, configure a strict instance: 075 * </p> 076 * 077 * <pre> 078 * Base64 standard = Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get(); 079 * Base64 urlSafe = Base64.builder().setUrlSafe(true).setDecodingPolicy(CodecPolicy.STRICT).get(); 080 * </pre> 081 * 082 * <p> 083 * These instances accept unchunked input using their respective encoding alphabets. The standard instance requires padding for partial blocks; the URL-safe 084 * instance requires unpadded input. See {@link BaseNCodec} for the full canonical decoding contract and guidance on comparing encoded values. 085 * </p> 086 * 087 * @see Base64InputStream 088 * @see Base64OutputStream 089 * @see <a href="https://www.ietf.org/rfc/rfc2045">RFC 2045 Multipurpose Internet Mail Extensions (MIME) Part One: Format of Internet Message Bodies</a> 090 * @see <a href="https://datatracker.ietf.org/doc/html/rfc4648">RFC 4648 The Base16, Base32, and Base64 Data Encodings</a> 091 * @since 1.0 092 */ 093public class Base64 extends BaseNCodec { 094 095 /** 096 * Builds {@link Base64} instances. 097 * 098 * <p> 099 * To configure a new instance, use a {@link Builder}. For example: 100 * </p> 101 * 102 * <pre> 103 * Base64 base64 = Base64.builder() 104 * .setCodecPolicy(CodecPolicy.LENIENT) // default is lenient, null resets to default 105 * .setEncodeTable(customEncodeTable) // default is built in, null resets to default 106 * .setLineLength(0) // default is none 107 * .setLineSeparator('\r', '\n') // default is CR LF, null resets to default 108 * .setPadding('=') // default is '=' 109 * .setUrlSafe(false) // default is false 110 * .get() 111 * </pre> 112 * 113 * @since 1.17.0 114 */ 115 public static class Builder extends AbstractBuilder<Base64, Builder> { 116 117 /** 118 * Constructs a new instance. 119 */ 120 public Builder() { 121 super(STANDARD_ENCODE_TABLE); 122 setDecodeTableRaw(DECODE_TABLE); 123 setEncodeTableRaw(STANDARD_ENCODE_TABLE); 124 setEncodedBlockSize(BYTES_PER_ENCODED_BLOCK); 125 setUnencodedBlockSize(BYTES_PER_UNENCODED_BLOCK); 126 } 127 128 @Override 129 public Base64 get() { 130 return new Base64(this); 131 } 132 133 /** 134 * Sets the format of the decoding table. This method allows callers to explicitly state whether a standard or URL-safe Base64 decoding is expected. This method 135 * does not modify behavior on encoding operations. For configuration of the encoding behavior, please use {@link #setUrlSafe(boolean)} method. 136 * <p> 137 * By default, the implementation uses the {@link DecodeTableFormat#MIXED} approach, allowing a seamless handling of both 138 * {@link DecodeTableFormat#URL_SAFE} and {@link DecodeTableFormat#STANDARD} base64 in lenient mode. Strict decoding additionally requires each character 139 * to match the configured encoding table. 140 * </p> 141 * 142 * @param format table format to be used on Base64 decoding. Use {@link DecodeTableFormat#MIXED} or null to reset to the default behavior. 143 * @return {@code this} instance. 144 * @since 1.21 145 */ 146 public Builder setDecodeTableFormat(final DecodeTableFormat format) { 147 if (format == null) { 148 return setDecodeTableRaw(DECODE_TABLE); 149 } 150 switch (format) { 151 case STANDARD: 152 return setDecodeTableRaw(STANDARD_DECODE_TABLE); 153 case URL_SAFE: 154 return setDecodeTableRaw(URL_SAFE_DECODE_TABLE); 155 case MIXED: 156 default: 157 return setDecodeTableRaw(DECODE_TABLE); 158 } 159 } 160 161 /** 162 * Sets the encode table. 163 * 164 * @param encodeTable The encode table with exactly 64 unique entries, null resets to the default. 165 * @return {@code this} instance. 166 * @throws IllegalArgumentException Thrown if {@code encodeTable} does not contain exactly 64 unique entries. 167 */ 168 @Override 169 public Builder setEncodeTable(final byte... encodeTable) { 170 setDecodeTableRaw(toDecodeTable(encodeTable)); 171 return super.setEncodeTable(encodeTable); 172 } 173 174 /** 175 * Sets the URL-safe encoding policy. 176 * <p> 177 * Strict decoding requires this alphabet and its padding convention. Lenient decoding accepts both alphabets by default; use 178 * {@code Builder.setDecodeTableFormat(DecodeTableFormat)} to select a decoding table. 179 * </p> 180 * 181 * @param urlSafe URL-safe encoding policy. 182 * @return {@code this} instance. 183 */ 184 public Builder setUrlSafe(final boolean urlSafe) { 185 // Javadoc 8 can't find {@link #setDecodeTableFormat(DecodeTableFormat)} 186 return setEncodeTable(toUrlSafeEncodeTable(urlSafe)); 187 } 188 189 } 190 191 /** 192 * Enumerates the Base64 table format to be used on decoding. 193 * <p> 194 * By default, the method uses {@link DecodeTableFormat#MIXED} approach, allowing a seamless handling of both {@link DecodeTableFormat#URL_SAFE} and 195 * {@link DecodeTableFormat#STANDARD} base64 options. 196 * </p> 197 * 198 * @since 1.21 199 */ 200 public enum DecodeTableFormat { 201 202 /** 203 * Corresponds to the standard Base64 coding table, as specified in 204 * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>. 205 */ 206 STANDARD, 207 208 /** 209 * Corresponds to the URL-safe Base64 coding table, as specified in 210 * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 211 * 4648 Table 2: The "URL and Filename safe" Base 64 Alphabet</a>. 212 */ 213 URL_SAFE, 214 215 /** 216 * Represents a joint approach, allowing a seamless decoding of both character sets, corresponding to either 217 * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a> or 218 * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 219 * 4648 Table 2: The "URL and Filename safe" Base 64 Alphabet</a>. This decoding table is used by default. 220 */ 221 MIXED 222 } 223 224 /** 225 * BASE64 characters are 6 bits in length. 226 * They are formed by taking a block of 3 octets to form a 24-bit string, 227 * which is converted into 4 BASE64 characters. 228 */ 229 private static final int BITS_PER_ENCODED_BYTE = 6; 230 private static final int BYTES_PER_UNENCODED_BLOCK = 3; 231 private static final int BYTES_PER_ENCODED_BLOCK = 4; 232 private static final int DECODING_TABLE_LENGTH = 256; 233 234 /** 235 * This array is a lookup table that translates 6-bit positive integer index values into their "Base64 Alphabet" equivalents as specified in 236 * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>. 237 * <p> 238 * Thanks to "commons" project in ws.apache.org for this code. https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/ 239 * </p> 240 */ 241 // @formatter:off 242 private static final byte[] STANDARD_ENCODE_TABLE = { 243 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M', 244 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', 245 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 246 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 247 '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '+', '/' 248 }; 249 250 /** 251 * This is a copy of the STANDARD_ENCODE_TABLE above, but with + and / changed to - and _ to make the encoded Base64 results more URL-SAFE. This table is 252 * only used when the Base64's mode is set to URL-SAFE. 253 */ 254 // @formatter:off 255 private static final byte[] URL_SAFE_ENCODE_TABLE = { 256 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M', 257 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', 258 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 259 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', 260 '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '-', '_' 261 }; 262 // @formatter:on 263 264 /** 265 * This array is a lookup table that translates Unicode characters drawn from the "Base64 Alphabet" (as specified in 266 * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>) into their 6-bit 267 * positive integer equivalents. Characters that are not in the Base64 or Base64 URL-safe alphabets but fall within the bounds of the array are translated 268 * to -1. 269 * <p> 270 * The characters '+' and '-' both decode to 62. '/' and '_' both decode to 63. This means decoder seamlessly handles both URL_SAFE and STANDARD base64. 271 * (The encoder, on the other hand, needs to know ahead of time what to emit). 272 * </p> 273 * <p> 274 * Thanks to "commons" project in ws.apache.org for this code. https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/ 275 * </p> 276 */ 277 private static final byte[] DECODE_TABLE = { 278 // 0 1 2 3 4 5 6 7 8 9 A B C D E F 279 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f 280 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f 281 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, 62, -1, 62, -1, 63, // 20-2f + - / 282 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, -1, -1, -1, -1, -1, -1, // 30-3f 0-9 283 -1, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, // 40-4f A-O 284 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, -1, -1, -1, -1, 63, // 50-5f P-Z _ 285 -1, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, // 60-6f a-o 286 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51 // 70-7a p-z 287 }; 288 289 /** 290 * This array is a lookup table that translates Unicode characters drawn from the "Base64 Alphabet" (as specified in 291 * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>) into their 6-bit 292 * positive integer equivalents. Characters that are not in the Base64 alphabet but fall within the bounds of the array are translated to -1. This decoding 293 * table handles only the standard base64 characters, such as '+' and '/'. The "url-safe" characters such as '-' and '_' are not supported by the table. 294 */ 295 private static final byte[] STANDARD_DECODE_TABLE = { 296 // 0 1 2 3 4 5 6 7 8 9 A B C D E F 297 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f 298 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f 299 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, 62, -1, -1, -1, 63, // 20-2f + / 300 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, -1, -1, -1, -1, -1, -1, // 30-3f 0-9 301 -1, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, // 40-4f A-O 302 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, -1, -1, -1, -1, -1, // 50-5f P-Z 303 -1, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, // 60-6f a-o 304 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51 // 70-7a p-z 305 }; 306 307 /** 308 * This array is a lookup table that translates Unicode characters drawn from the "Base64 URL-safe Alphabet" (as specified in 309 * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648 310 * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>) into their 6-bit positive integer equivalents. Characters that are not in the Base64 URL-safe 311 * alphabet but fall within the bounds of the array are translated to -1. This decoding table handles only the URL-safe base64 characters, such as '-' and 312 * '_'. The standard characters such as '+' and '/' are not supported by the table. 313 */ 314 private static final byte[] URL_SAFE_DECODE_TABLE = { 315 // 0 1 2 3 4 5 6 7 8 9 A B C D E F 316 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f 317 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f 318 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, 62, -1, -1, // 20-2f - 319 52, 53, 54, 55, 56, 57, 58, 59, 60, 61, -1, -1, -1, -1, -1, -1, // 30-3f 0-9 320 -1, 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, // 40-4f A-O 321 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, -1, -1, -1, -1, 63, // 50-5f P-Z _ 322 -1, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, // 60-6f a-o 323 41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51 // 70-7a p-z 324 }; 325 326 /** 327 * Base64 uses 6-bit fields. 328 */ 329 330 /** Mask used to extract 6 bits, used when encoding */ 331 private static final int MASK_6_BITS = 0x3f; 332 333 // The static final fields above are used for the original static byte[] methods on Base64. 334 // The private member fields below are used with the new streaming approach, which requires 335 // some state be preserved between calls of encode() and decode(). 336 337 /** Mask used to extract 4 bits, used when decoding final trailing character. */ 338 private static final int MASK_4_BITS = 0xf; 339 340 /** Mask used to extract 2 bits, used when decoding final trailing character. */ 341 private static final int MASK_2_BITS = 0x3; 342 343 /** 344 * Creates a new Builder. 345 * 346 * <p> 347 * To configure a new instance, use a {@link Builder}. For example: 348 * </p> 349 * 350 * <pre> 351 * Base64 base64 = Base64.builder() 352 * .setDecodingPolicy(CodecPolicy.LENIENT) // default is lenient, null resets to default 353 * .setEncodeTable(customEncodeTable) // default is built in, null resets to default 354 * .setLineLength(0) // default is none 355 * .setLineSeparator('\r', '\n') // default is CR LF, null resets to default 356 * .setPadding('=') // default is '=' 357 * .setUrlSafe(false) // default is false 358 * .get() 359 * </pre> 360 * 361 * @return A new Builder. 362 * @since 1.17.0 363 */ 364 public static Builder builder() { 365 return new Builder(); 366 } 367 368 /** 369 * Calculates a decode table for a given encode table. 370 * 371 * @param encodeTable that is used to determine decode lookup table. 372 * @return A new decode table. 373 */ 374 private static byte[] calculateDecodeTable(final byte[] encodeTable) { 375 if (encodeTable.length != STANDARD_ENCODE_TABLE.length) { 376 throw new IllegalArgumentException("encodeTable must have exactly 64 entries."); 377 } 378 final byte[] decodeTable = new byte[DECODING_TABLE_LENGTH]; 379 Arrays.fill(decodeTable, (byte) -1); 380 for (int i = 0; i < encodeTable.length; i++) { 381 final int encodedByte = encodeTable[i] & 0xff; 382 if (decodeTable[encodedByte] != -1) { 383 throw new IllegalArgumentException("encodeTable must not contain duplicate entries."); 384 } 385 decodeTable[encodedByte] = (byte) i; 386 } 387 return decodeTable; 388 } 389 390 private static boolean contains(final byte[] bytes, final byte value) { 391 for (final byte element : bytes) { 392 if (element == value) { 393 return true; 394 } 395 } 396 return false; 397 } 398 399 /** 400 * Decodes Base64 data into octets using lenient decoding. 401 * 402 * <p> 403 * This method uses the standard and URL-safe alphabets. It skips unsupported input, discards data after the first padding character, and accepts 404 * noncanonical padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical 405 * input. 406 * </p> 407 * 408 * <p> 409 * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64Data)}. 410 * See {@link BaseNCodec} for guidance on comparing encoded values. 411 * </p> 412 * 413 * @param base64Data Byte array containing Base64 data. 414 * @return New array containing decoded data. 415 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 416 */ 417 public static byte[] decodeBase64(final byte[] base64Data) { 418 return new Base64().decode(base64Data); 419 } 420 421 /** 422 * Decodes a Base64 string into octets using lenient decoding. 423 * 424 * <p> 425 * This method uses the standard and URL-safe alphabets. It skips unsupported input, discards data after the first padding character, and accepts 426 * noncanonical padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical 427 * input. 428 * </p> 429 * 430 * <p> 431 * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64String)}. 432 * See {@link BaseNCodec} for guidance on comparing encoded values. 433 * </p> 434 * 435 * @param base64String String containing Base64 data. 436 * @return New array containing decoded data. 437 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 438 * @since 1.4 439 */ 440 public static byte[] decodeBase64(final String base64String) { 441 return new Base64().decode(base64String); 442 } 443 444 /** 445 * Decodes standard Base64 data into octets using lenient decoding. 446 * 447 * <p> 448 * This method uses the standard alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical 449 * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input. 450 * </p> 451 * 452 * <p> 453 * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64Data)}. 454 * See {@link BaseNCodec} for guidance on comparing encoded values. 455 * </p> 456 * 457 * @param base64Data Byte array containing Base64 data. 458 * @return New array containing decoded data. 459 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 460 * @since 1.21 461 */ 462 public static byte[] decodeBase64Standard(final byte[] base64Data) { 463 return builder().setDecodeTableFormat(DecodeTableFormat.STANDARD).get().decode(base64Data); 464 } 465 466 /** 467 * Decodes a standard Base64 string into octets using lenient decoding. 468 * 469 * <p> 470 * This method uses the standard alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical 471 * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input. 472 * </p> 473 * 474 * <p> 475 * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64String)}. 476 * See {@link BaseNCodec} for guidance on comparing encoded values. 477 * </p> 478 * 479 * @param base64String String containing Base64 data. 480 * @return New array containing decoded data. 481 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 482 * @since 1.21 483 */ 484 public static byte[] decodeBase64Standard(final String base64String) { 485 return builder().setDecodeTableFormat(DecodeTableFormat.STANDARD).get().decode(base64String); 486 } 487 488 /** 489 * Decodes URL-safe Base64 data into octets using lenient decoding. 490 * 491 * <p> 492 * This method uses the URL-safe alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical 493 * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input. 494 * </p> 495 * 496 * <p> 497 * For canonical decoding, use {@code Base64.builder().setUrlSafe(true).setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64Data)}. 498 * See {@link BaseNCodec} for guidance on comparing encoded values. 499 * </p> 500 * 501 * @param base64Data Byte array containing Base64 data. 502 * @return New array containing decoded data. 503 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 504 * @since 1.21 505 */ 506 public static byte[] decodeBase64UrlSafe(final byte[] base64Data) { 507 return builder().setDecodeTableFormat(DecodeTableFormat.URL_SAFE).get().decode(base64Data); 508 } 509 510 /** 511 * Decodes a URL-safe Base64 string into octets using lenient decoding. 512 * 513 * <p> 514 * This method uses the URL-safe alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical 515 * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input. 516 * </p> 517 * 518 * <p> 519 * For canonical decoding, use {@code Base64.builder().setUrlSafe(true).setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64String)}. 520 * See {@link BaseNCodec} for guidance on comparing encoded values. 521 * </p> 522 * 523 * @param base64String String containing Base64 data. 524 * @return New array containing decoded data. 525 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 526 * @since 1.21 527 */ 528 public static byte[] decodeBase64UrlSafe(final String base64String) { 529 return builder().setDecodeTableFormat(DecodeTableFormat.URL_SAFE).get().decode(base64String); 530 } 531 532 /** 533 * Decodes a byte64-encoded integer according to crypto standards such as W3C's XML-Signature. 534 * 535 * @param array A byte array containing base64 character data. 536 * @return A BigInteger. 537 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 538 * @since 1.4 539 */ 540 public static BigInteger decodeInteger(final byte[] array) { 541 return new BigInteger(1, decodeBase64(array)); 542 } 543 544 /** 545 * Encodes binary data using the base64 algorithm but does not chunk the output. 546 * 547 * @param binaryData binary data to encode. 548 * @return byte[] containing Base64 characters in their UTF-8 representation. 549 */ 550 public static byte[] encodeBase64(final byte[] binaryData) { 551 return encodeBase64(binaryData, false); 552 } 553 554 /** 555 * Encodes binary data using the base64 algorithm, optionally chunking the output into 76 character blocks. 556 * 557 * @param binaryData Array containing binary data to encode. 558 * @param isChunked if {@code true} this encoder will chunk the base64 output into 76 character blocks. 559 * @return Base64-encoded data. 560 * @throws IllegalArgumentException Thrown when the input array needs an output array bigger than {@link Integer#MAX_VALUE}. 561 */ 562 public static byte[] encodeBase64(final byte[] binaryData, final boolean isChunked) { 563 return encodeBase64(binaryData, isChunked, false); 564 } 565 566 /** 567 * Encodes binary data using the base64 algorithm, optionally chunking the output into 76 character blocks. 568 * 569 * @param binaryData Array containing binary data to encode. 570 * @param isChunked if {@code true} this encoder will chunk the base64 output into 76 character blocks. 571 * @param urlSafe if {@code true} this encoder will emit - and _ instead of the usual + and / characters. <strong>No padding is added when encoding using 572 * the URL-safe alphabet.</strong> 573 * @return Base64-encoded data. 574 * @throws IllegalArgumentException Thrown when the input array needs an output array bigger than {@link Integer#MAX_VALUE}. 575 * @since 1.4 576 */ 577 public static byte[] encodeBase64(final byte[] binaryData, final boolean isChunked, final boolean urlSafe) { 578 return encodeBase64(binaryData, isChunked, urlSafe, Integer.MAX_VALUE); 579 } 580 581 /** 582 * Encodes binary data using the base64 algorithm, optionally chunking the output into 76 character blocks. 583 * 584 * @param binaryData Array containing binary data to encode. 585 * @param isChunked if {@code true} this encoder will chunk the base64 output into 76 character blocks. 586 * @param urlSafe if {@code true} this encoder will emit - and _ instead of the usual + and / characters. <strong>No padding is added when encoding 587 * using the URL-safe alphabet.</strong> 588 * @param maxResultSize The maximum result size to accept. 589 * @return Base64-encoded data. 590 * @throws IllegalArgumentException Thrown when the input array needs an output array bigger than maxResultSize. 591 * @since 1.4 592 */ 593 public static byte[] encodeBase64(final byte[] binaryData, final boolean isChunked, final boolean urlSafe, final int maxResultSize) { 594 if (BinaryCodec.isEmpty(binaryData)) { 595 return binaryData; 596 } 597 // Create this so can use the super-class method 598 // Also ensures that the same roundings are performed by the ctor and the code 599 final Base64 b64 = isChunked ? new Base64(urlSafe) : new Base64(0, CHUNK_SEPARATOR, urlSafe); 600 final long len = b64.getEncodedLength(binaryData); 601 if (len > maxResultSize) { 602 throw new IllegalArgumentException( 603 "Input array too big, the output array would be bigger (" + len + ") than the specified maximum size of " + maxResultSize); 604 } 605 return b64.encode(binaryData); 606 } 607 608 /** 609 * Encodes binary data using the base64 algorithm and chunks the encoded output into 76 character blocks 610 * 611 * @param binaryData binary data to encode. 612 * @return Base64 characters chunked in 76 character blocks. 613 */ 614 public static byte[] encodeBase64Chunked(final byte[] binaryData) { 615 return encodeBase64(binaryData, true); 616 } 617 618 /** 619 * Encodes binary data using the base64 algorithm but does not chunk the output. 620 * <p> 621 * <strong> We changed the behavior of this method from multi-line chunking (1.4) to single-line non-chunking (1.5).</strong> 622 * </p> 623 * 624 * @param binaryData binary data to encode. 625 * @return String containing Base64 characters. 626 * @since 1.4 (NOTE: 1.4 chunked the output, whereas 1.5 does not). 627 */ 628 public static String encodeBase64String(final byte[] binaryData) { 629 return StringUtils.newStringUsAscii(encodeBase64(binaryData, false)); 630 } 631 632 /** 633 * Encodes binary data using a URL-safe variation of the base64 algorithm but does not chunk the output. The url-safe variation emits - and _ instead of + 634 * and / characters. <strong>No padding is added.</strong> 635 * 636 * @param binaryData binary data to encode. 637 * @return byte[] containing Base64 characters in their UTF-8 representation. 638 * @since 1.4 639 */ 640 public static byte[] encodeBase64URLSafe(final byte[] binaryData) { 641 return encodeBase64(binaryData, false, true); 642 } 643 644 /** 645 * Encodes binary data using a URL-safe variation of the base64 algorithm but does not chunk the output. The url-safe variation emits - and _ instead of + 646 * and / characters. <strong>No padding is added.</strong> 647 * 648 * @param binaryData binary data to encode. 649 * @return String containing Base64 characters. 650 * @since 1.4 651 */ 652 public static String encodeBase64URLSafeString(final byte[] binaryData) { 653 return StringUtils.newStringUsAscii(encodeBase64(binaryData, false, true)); 654 } 655 656 /** 657 * Encodes to a byte64-encoded integer according to crypto standards such as W3C's XML-Signature. 658 * 659 * @param bigInteger A BigInteger. 660 * @return A byte array containing base64 character data. 661 * @throws NullPointerException Thrown if null is passed in. 662 * @since 1.4 663 */ 664 public static byte[] encodeInteger(final BigInteger bigInteger) { 665 Objects.requireNonNull(bigInteger, "bigInteger"); 666 return encodeBase64(toUnsignedBytes(bigInteger), false); 667 } 668 669 /** 670 * Tests a given byte array to see if it contains only valid characters within the Base64 alphabet. Currently the method treats whitespace as valid. 671 * 672 * @param arrayOctet byte array to test. 673 * @return {@code true} if all bytes are valid characters in the Base64 alphabet or if the byte array is empty; {@code false}, otherwise. 674 * @deprecated 1.5 Use {@link #isBase64(byte[])}, will be removed in 2.0. 675 */ 676 @Deprecated 677 public static boolean isArrayByteBase64(final byte[] arrayOctet) { 678 return isBase64(arrayOctet); 679 } 680 681 /** 682 * Tests whether or not the {@code octet} is in the Base64 alphabet. 683 * <p> 684 * This method treats all characters included within standard base64 and base64url encodings as valid base64 characters. This includes the '+' and '/' 685 * (standard base64), as well as '-' and '_' (URL-safe base64) characters. To test membership in only the standard Base64 or Base64 URL-safe 686 * alphabet, use {@link #isBase64Standard(byte)} or {@link #isBase64Url(byte)} methods respectively. 687 * </p> 688 * 689 * @param octet The value to test. 690 * @return {@code true} if the value is defined in the Base64 alphabet, {@code false} otherwise. 691 * @since 1.4 692 */ 693 public static boolean isBase64(final byte octet) { 694 return octet == PAD_DEFAULT || octet >= 0 && octet < DECODE_TABLE.length && DECODE_TABLE[octet] != -1; 695 } 696 697 /** 698 * Tests a given byte array to see if it contains only valid characters within the Base64 alphabet. Currently the method treats whitespace as valid. 699 * <p> 700 * This method treats all characters included within standard base64 and base64url encodings as valid base64 characters. This includes the '+' and '/' 701 * (standard base64), as well as '-' and '_' (URL-safe base64) characters. To test membership in only the standard Base64 or Base64 URL-safe 702 * alphabet, use {@link #isBase64Standard(byte[])} or {@link #isBase64Url(byte[])} methods respectively. 703 * </p> 704 * 705 * <p> 706 * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits. 707 * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input. 708 * </p> 709 * 710 * @param arrayOctet byte array to test. 711 * @return {@code true} if all bytes are valid characters in the Base64 alphabet or if the byte array is empty; {@code false}, otherwise. 712 * @since 1.5 713 */ 714 public static boolean isBase64(final byte[] arrayOctet) { 715 for (final byte element : arrayOctet) { 716 if (!isBase64(element) && !Character.isWhitespace(element)) { 717 return false; 718 } 719 } 720 return true; 721 } 722 723 /** 724 * Tests a given String to see if it contains only valid characters within the Base64 alphabet. Currently the method treats whitespace as valid. 725 * <p> 726 * This method treats all characters included within standard base64 and base64url encodings as valid base64 characters. This includes the '+' and '/' 727 * (standard base64), as well as '-' and '_' (URL-safe base64) characters. To test membership in only the standard Base64 or Base64 URL-safe 728 * alphabet, use {@link #isBase64Standard(String)} or {@link #isBase64Url(String)} methods respectively. 729 * </p> 730 * 731 * <p> 732 * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits. 733 * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input. 734 * </p> 735 * 736 * @param base64 String to test. 737 * @return {@code true} if all characters in the String are valid characters in the Base64 alphabet or if the String is empty; {@code false}, otherwise. 738 * @since 1.5 739 */ 740 public static boolean isBase64(final String base64) { 741 return isBase64(StringUtils.getBytesUtf8(base64)); 742 } 743 744 /** 745 * Tests whether or not the {@code octet} is in the standard Base64 alphabet. 746 * <p> 747 * This implementation is aligned with <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The 748 * Base64 Alphabet</a>. 749 * </p> 750 * 751 * @param octet The value to test. 752 * @return {@code true} if the value is defined in the standard Base64 alphabet, {@code false} otherwise. 753 * @since 1.21 754 */ 755 public static boolean isBase64Standard(final byte octet) { 756 return octet == PAD_DEFAULT || octet >= 0 && octet < STANDARD_DECODE_TABLE.length && STANDARD_DECODE_TABLE[octet] != -1; 757 } 758 759 /** 760 * Tests a given byte array to see if it contains only valid characters within the standard Base64 alphabet. The method treats whitespace as valid. 761 * <p> 762 * This implementation is aligned with <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The 763 * Base64 Alphabet</a>. 764 * </p> 765 * 766 * <p> 767 * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits. 768 * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input. 769 * </p> 770 * 771 * @param arrayOctet byte array to test. 772 * @return {@code true} if all bytes are valid characters in the standard Base64 alphabet. {@code false}, otherwise. 773 * @since 1.21 774 */ 775 public static boolean isBase64Standard(final byte[] arrayOctet) { 776 for (final byte element : arrayOctet) { 777 if (!isBase64Standard(element) && !Character.isWhitespace(element)) { 778 return false; 779 } 780 } 781 return true; 782 } 783 784 /** 785 * Tests a given String to see if it contains only valid characters within the standard Base64 alphabet. The method treats whitespace as valid. 786 * <p> 787 * This implementation is aligned with <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The 788 * Base64 Alphabet</a>. 789 * </p> 790 * 791 * <p> 792 * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits. 793 * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input. 794 * </p> 795 * 796 * @param base64 String to test. 797 * @return {@code true} if all characters in the String are valid characters in the standard Base64 alphabet or if the String is empty; {@code false}, 798 * otherwise. 799 * @since 1.21 800 */ 801 public static boolean isBase64Standard(final String base64) { 802 return isBase64Standard(StringUtils.getBytesUtf8(base64)); 803 } 804 805 /** 806 * Tests whether or not the {@code octet} is in the URL-safe Base64 alphabet. 807 * <p> 808 * This implementation is aligned with 809 * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648 810 * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>. 811 * </p> 812 * 813 * @param octet The value to test. 814 * @return {@code true} if the value is defined in the URL-safe Base64 alphabet, {@code false} otherwise. 815 * @since 1.21 816 */ 817 public static boolean isBase64Url(final byte octet) { 818 return octet == PAD_DEFAULT || octet >= 0 && octet < URL_SAFE_DECODE_TABLE.length && URL_SAFE_DECODE_TABLE[octet] != -1; 819 } 820 821 /** 822 * Tests a given byte array to see if it contains only valid characters within the URL-safe Base64 alphabet. The method treats whitespace as valid. 823 * <p> 824 * This implementation is aligned with 825 * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648 826 * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>. 827 * </p> 828 * 829 * <p> 830 * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits. 831 * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input. 832 * </p> 833 * 834 * @param arrayOctet byte array to test. 835 * @return {@code true} if all bytes are valid characters in the URL-safe Base64 alphabet, {@code false}, otherwise. 836 * @since 1.21 837 */ 838 public static boolean isBase64Url(final byte[] arrayOctet) { 839 for (final byte element : arrayOctet) { 840 if (!isBase64Url(element) && !Character.isWhitespace(element)) { 841 return false; 842 } 843 } 844 return true; 845 } 846 847 /** 848 * Tests a given String to see if it contains only valid characters within the URL-safe Base64 alphabet. The method treats whitespace as valid. 849 * <p> 850 * This implementation is aligned with 851 * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648 852 * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>. 853 * </p> 854 * 855 * <p> 856 * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits. 857 * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input. 858 * </p> 859 * 860 * @param base64 String to test. 861 * @return {@code true} if all characters in the String are valid characters in the URL-safe Base64 alphabet or if the String is empty; {@code false}, 862 * otherwise. 863 * @since 1.21 864 */ 865 public static boolean isBase64Url(final String base64) { 866 return isBase64Url(StringUtils.getBytesUtf8(base64)); 867 } 868 869 private static byte[] toDecodeTable(final byte[] encodeTable) { 870 final byte[] table = encodeTable != null ? encodeTable : STANDARD_ENCODE_TABLE; 871 if (Arrays.equals(table, STANDARD_ENCODE_TABLE) || Arrays.equals(table, URL_SAFE_ENCODE_TABLE)) { 872 return DECODE_TABLE; 873 } 874 return calculateDecodeTable(table); 875 } 876 877 static byte[] toUrlSafeEncodeTable(final boolean urlSafe) { 878 return urlSafe ? URL_SAFE_ENCODE_TABLE : STANDARD_ENCODE_TABLE; 879 } 880 881 /** 882 * Line separator for encoding and strict decoding. Only used if lineLength > 0. 883 */ 884 private final byte[] lineSeparator; 885 886 /** 887 * Convenience variable to help us determine when our buffer is going to run out of room and needs resizing. {@code encodeSize = 4 + lineSeparator.length;} 888 */ 889 private final int encodeSize; 890 private final boolean isUrlSafe; 891 private final boolean isStandardEncodeTable; 892 893 /** 894 * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode. 895 * <p> 896 * When encoding the line length is 0 (no chunking), and the encoding table is STANDARD_ENCODE_TABLE. 897 * </p> 898 * <p> 899 * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout. 900 * </p> 901 */ 902 public Base64() { 903 this(0); 904 } 905 906 /** 907 * Constructs a Base64 codec used for decoding (all modes) and encoding in the given URL-safe mode. 908 * <p> 909 * When encoding the line length is 76, the line separator is CRLF, and the encoding table is STANDARD_ENCODE_TABLE. 910 * </p> 911 * <p> 912 * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout. 913 * </p> 914 * 915 * @param urlSafe if {@code true}, URL-safe encoding is used. In most cases this should be set to {@code false}. 916 * @since 1.4 917 * @deprecated Use {@link #builder()} and {@link Builder}. 918 */ 919 @Deprecated 920 public Base64(final boolean urlSafe) { 921 this(MIME_CHUNK_SIZE, CHUNK_SEPARATOR, urlSafe); 922 } 923 924 private Base64(final Builder builder) { 925 super(builder); 926 final byte[] encTable = builder.getEncodeTable(); 927 if (encTable.length != STANDARD_ENCODE_TABLE.length) { 928 throw new IllegalArgumentException("encodeTable must have exactly 64 entries."); 929 } 930 if (contains(encTable, pad)) { 931 throw new IllegalArgumentException("encodeTable must not contain the padding byte."); 932 } 933 this.isStandardEncodeTable = Arrays.equals(encTable, STANDARD_ENCODE_TABLE); 934 this.isUrlSafe = Arrays.equals(encTable, URL_SAFE_ENCODE_TABLE); 935 // TODO could be simplified if there is no requirement to reject invalid line sep when length <=0 936 // @see test case Base64Test.testConstructors() 937 if (builder.getLineSeparator().length > 0) { 938 final byte[] lineSeparatorB = builder.getLineSeparator(); 939 if (containsAlphabetOrPad(lineSeparatorB)) { 940 final String sep = StringUtils.newStringUtf8(lineSeparatorB); 941 throw new IllegalArgumentException("lineSeparator must not contain base64 characters: [" + sep + "]"); 942 } 943 if (builder.getLineLength() > 0) { // null line-sep forces no chunking rather than throwing IAE 944 this.encodeSize = BYTES_PER_ENCODED_BLOCK + lineSeparatorB.length; 945 this.lineSeparator = lineSeparatorB; 946 } else { 947 this.encodeSize = BYTES_PER_ENCODED_BLOCK; 948 this.lineSeparator = null; 949 } 950 } else { 951 this.encodeSize = BYTES_PER_ENCODED_BLOCK; 952 this.lineSeparator = null; 953 } 954 } 955 956 /** 957 * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode. 958 * <p> 959 * When encoding the line length is given in the constructor, the line separator is CRLF, and the encoding table is STANDARD_ENCODE_TABLE. 960 * </p> 961 * <p> 962 * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data. 963 * </p> 964 * <p> 965 * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout. 966 * </p> 967 * 968 * @param lineLength Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength <= 0, then 969 * the output will not be divided into lines (chunks). Ignored when decoding leniently. 970 * @since 1.4 971 * @deprecated Use {@link #builder()} and {@link Builder}. 972 */ 973 @Deprecated 974 public Base64(final int lineLength) { 975 this(lineLength, CHUNK_SEPARATOR); 976 } 977 978 /** 979 * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode. 980 * <p> 981 * When encoding the line length and line separator are given in the constructor, and the encoding table is STANDARD_ENCODE_TABLE. 982 * </p> 983 * <p> 984 * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data. 985 * </p> 986 * <p> 987 * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout. 988 * </p> 989 * 990 * @param lineLength Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength <= 0, 991 * then the output will not be divided into lines (chunks). Ignored when decoding leniently. 992 * @param lineSeparator Each line of encoded data will end with this sequence of bytes. 993 * @throws IllegalArgumentException Thrown when the provided lineSeparator included some base64 characters. 994 * @since 1.4 995 * @deprecated Use {@link #builder()} and {@link Builder}. 996 */ 997 @Deprecated 998 public Base64(final int lineLength, final byte[] lineSeparator) { 999 this(lineLength, lineSeparator, false); 1000 } 1001 1002 /** 1003 * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode. 1004 * <p> 1005 * When encoding the line length and line separator are given in the constructor, and the encoding table is STANDARD_ENCODE_TABLE. 1006 * </p> 1007 * <p> 1008 * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data. 1009 * </p> 1010 * <p> 1011 * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout. 1012 * </p> 1013 * 1014 * @param lineLength Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength <= 0, 1015 * then the output will not be divided into lines (chunks). Ignored when decoding leniently. 1016 * @param lineSeparator Each line of encoded data will end with this sequence of bytes. 1017 * @param urlSafe Instead of emitting '+' and '/' we emit '-' and '_' respectively. urlSafe is only applied to encode operations. Decoding seamlessly 1018 * handles both modes. <strong>No padding is added when using the URL-safe alphabet.</strong> 1019 * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base64 characters. 1020 * @since 1.4 1021 * @deprecated Use {@link #builder()} and {@link Builder}. 1022 */ 1023 @Deprecated 1024 public Base64(final int lineLength, final byte[] lineSeparator, final boolean urlSafe) { 1025 this(builder().setLineLength(lineLength).setLineSeparator(lineSeparator != null ? lineSeparator : EMPTY_BYTE_ARRAY).setPadding(PAD_DEFAULT) 1026 .setEncodeTableRaw(toUrlSafeEncodeTable(urlSafe)).setDecodingPolicy(DECODING_POLICY_DEFAULT)); 1027 } 1028 1029 /** 1030 * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode. 1031 * <p> 1032 * When encoding the line length and line separator are given in the constructor, and the encoding table is STANDARD_ENCODE_TABLE. 1033 * </p> 1034 * <p> 1035 * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data. 1036 * </p> 1037 * <p> 1038 * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout. 1039 * </p> 1040 * 1041 * @param lineLength Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength <= 0, 1042 * then the output will not be divided into lines (chunks). Ignored when decoding leniently. 1043 * @param lineSeparator Each line of encoded data will end with this sequence of bytes. 1044 * @param urlSafe Instead of emitting '+' and '/' we emit '-' and '_' respectively. Strict decoding requires this alphabet. Lenient 1045 * decoding handles both modes. <strong>No padding is added when using the URL-safe alphabet.</strong> 1046 * @param decodingPolicy The decoding policy. 1047 * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base64 characters. 1048 * @since 1.15 1049 * @deprecated Use {@link #builder()} and {@link Builder}. 1050 */ 1051 @Deprecated 1052 public Base64(final int lineLength, final byte[] lineSeparator, final boolean urlSafe, final CodecPolicy decodingPolicy) { 1053 this(builder().setLineLength(lineLength).setLineSeparator(lineSeparator).setPadding(PAD_DEFAULT).setEncodeTableRaw(toUrlSafeEncodeTable(urlSafe)) 1054 .setDecodingPolicy(decodingPolicy)); 1055 } 1056 1057 /** 1058 * <p> 1059 * Decodes all of the provided data, starting at inPos, for inAvail bytes. Should be called at least twice: once with the data to decode, and once with 1060 * inAvail set to "-1" to alert decoder that EOF has been reached. Strict decoding requires the "-1" call to validate the complete input. 1061 * </p> 1062 * <p> 1063 * Lenient decoding ignores non-alphabet characters and stops at the first padding byte. Strict decoding accepts only the canonical form produced by this 1064 * instance's encoder, including its alphabet, padding, and line separators. 1065 * </p> 1066 * <p> 1067 * Thanks to "commons" project in ws.apache.org for the bitwise operations, and general approach. 1068 * https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/ 1069 * </p> 1070 * 1071 * @param input byte[] array of ASCII data to base64 decode. 1072 * @param inPos Position to start reading data from. 1073 * @param inAvail Amount of bytes available from input for decoding. 1074 * @param context The context to be used. 1075 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 1076 */ 1077 @Override 1078 void decode(final byte[] input, int inPos, final int inAvail, final Context context) { 1079 if (context.eof) { 1080 return; 1081 } 1082 if (inAvail < 0) { 1083 context.eof = true; 1084 if (isStrictDecoding()) { 1085 validateCanonicalEnd(isStandardEncodeTable, context); 1086 } 1087 } 1088 final int decodeSize = this.encodeSize - 1; 1089 for (int i = 0; i < inAvail; i++) { 1090 final int b = input[inPos++] & 0xff; 1091 if (isStrictDecoding()) { 1092 if (!validateCanonicalByte(b, lineSeparator, isStandardEncodeTable, context)) { 1093 continue; 1094 } 1095 } else if (b == (pad & 0xff)) { 1096 // We're done. 1097 context.eof = true; 1098 break; 1099 } 1100 final byte[] buffer = ensureBufferSize(decodeSize, context); 1101 if (b < decodeTable.length) { 1102 final int result = decodeTable[b]; 1103 if (result >= 0) { 1104 context.modulus = (context.modulus + 1) % BYTES_PER_ENCODED_BLOCK; 1105 context.ibitWorkArea = (context.ibitWorkArea << BITS_PER_ENCODED_BYTE) + result; 1106 if (context.modulus == 0) { 1107 buffer[context.pos++] = (byte) (context.ibitWorkArea >> 16 & MASK_8BITS); 1108 buffer[context.pos++] = (byte) (context.ibitWorkArea >> 8 & MASK_8BITS); 1109 buffer[context.pos++] = (byte) (context.ibitWorkArea & MASK_8BITS); 1110 } 1111 } 1112 } 1113 } 1114 1115 // Strict decoding waits for physical EOF to validate the complete input. 1116 // Lenient decoding also treats the first padding byte as EOF. 1117 if (context.eof && context.modulus != 0) { 1118 final byte[] buffer = ensureBufferSize(decodeSize, context); 1119 1120 // We have some spare bits remaining 1121 // Output all whole multiples of 8 bits and ignore the rest 1122 switch (context.modulus) { 1123// case 0 : // impossible, as excluded above 1124 case 1 : // 6 bits - either ignore entirely, or raise an exception 1125 validateTrailingCharacter(); 1126 break; 1127 case 2 : // 12 bits = 8 + 4 1128 validateCharacter(MASK_4_BITS, context); 1129 context.ibitWorkArea = context.ibitWorkArea >> 4; // dump the extra 4 bits 1130 buffer[context.pos++] = (byte) (context.ibitWorkArea & MASK_8BITS); 1131 break; 1132 case 3 : // 18 bits = 8 + 8 + 2 1133 validateCharacter(MASK_2_BITS, context); 1134 context.ibitWorkArea = context.ibitWorkArea >> 2; // dump 2 bits 1135 buffer[context.pos++] = (byte) (context.ibitWorkArea >> 8 & MASK_8BITS); 1136 buffer[context.pos++] = (byte) (context.ibitWorkArea & MASK_8BITS); 1137 break; 1138 default: 1139 throw new IllegalStateException("Impossible modulus " + context.modulus); 1140 } 1141 } 1142 } 1143 1144 /** 1145 * <p> 1146 * Encodes all of the provided data, starting at inPos, for inAvail bytes. Must be called at least twice: once with the data to encode, and once with 1147 * inAvail set to "-1" to alert encoder that EOF has been reached, to flush last remaining bytes (if not multiple of 3). 1148 * </p> 1149 * <p> 1150 * <strong>No padding is added when encoding using the URL-safe alphabet.</strong> 1151 * </p> 1152 * <p> 1153 * Thanks to "commons" project in ws.apache.org for the bitwise operations, and general approach. 1154 * https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/ 1155 * </p> 1156 * 1157 * @param in byte[] array of binary data to base64 encode. 1158 * @param inPos Position to start reading data from. 1159 * @param inAvail Amount of bytes available from input for encoding. 1160 * @param context The context to be used. 1161 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 1162 */ 1163 @Override 1164 void encode(final byte[] in, int inPos, final int inAvail, final Context context) { 1165 if (context.eof) { 1166 return; 1167 } 1168 // inAvail < 0 is how we're informed of EOF in the underlying data we're 1169 // encoding. 1170 if (inAvail < 0) { 1171 context.eof = true; 1172 if (0 == context.modulus && lineLength == 0) { 1173 return; // no leftovers to process and not using chunking 1174 } 1175 final byte[] buffer = ensureBufferSize(encodeSize, context); 1176 final int savedPos = context.pos; 1177 switch (context.modulus) { // 0-2 1178 case 0 : // nothing to do here 1179 break; 1180 case 1 : // 8 bits = 6 + 2 1181 // top 6 bits: 1182 buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 2 & MASK_6_BITS]; 1183 // remaining 2: 1184 buffer[context.pos++] = encodeTable[context.ibitWorkArea << 4 & MASK_6_BITS]; 1185 // URL-SAFE skips the padding to further reduce size. 1186 if (isStandardEncodeTable) { 1187 buffer[context.pos++] = pad; 1188 buffer[context.pos++] = pad; 1189 } 1190 break; 1191 1192 case 2 : // 16 bits = 6 + 6 + 4 1193 buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 10 & MASK_6_BITS]; 1194 buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 4 & MASK_6_BITS]; 1195 buffer[context.pos++] = encodeTable[context.ibitWorkArea << 2 & MASK_6_BITS]; 1196 // URL-SAFE skips the padding to further reduce size. 1197 if (isStandardEncodeTable) { 1198 buffer[context.pos++] = pad; 1199 } 1200 break; 1201 default: 1202 throw new IllegalStateException("Impossible modulus " + context.modulus); 1203 } 1204 context.currentLinePos += context.pos - savedPos; // keep track of current line position 1205 // if currentPos == 0 we are at the start of a line, so don't add CRLF 1206 if (lineLength > 0 && context.currentLinePos > 0) { 1207 System.arraycopy(lineSeparator, 0, buffer, context.pos, lineSeparator.length); 1208 context.pos += lineSeparator.length; 1209 } 1210 } else { 1211 for (int i = 0; i < inAvail; i++) { 1212 final byte[] buffer = ensureBufferSize(encodeSize, context); 1213 context.modulus = (context.modulus + 1) % BYTES_PER_UNENCODED_BLOCK; 1214 int b = in[inPos++]; 1215 if (b < 0) { 1216 b += 256; 1217 } 1218 context.ibitWorkArea = (context.ibitWorkArea << 8) + b; // BITS_PER_BYTE 1219 if (0 == context.modulus) { // 3 bytes = 24 bits = 4 * 6 bits to extract 1220 buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 18 & MASK_6_BITS]; 1221 buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 12 & MASK_6_BITS]; 1222 buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 6 & MASK_6_BITS]; 1223 buffer[context.pos++] = encodeTable[context.ibitWorkArea & MASK_6_BITS]; 1224 context.currentLinePos += BYTES_PER_ENCODED_BLOCK; 1225 if (lineLength > 0 && lineLength <= context.currentLinePos) { 1226 System.arraycopy(lineSeparator, 0, buffer, context.pos, lineSeparator.length); 1227 context.pos += lineSeparator.length; 1228 context.currentLinePos = 0; 1229 } 1230 } 1231 } 1232 } 1233 } 1234 1235 /** 1236 * Gets the line separator (for testing only). 1237 * 1238 * @return The line separator. 1239 */ 1240 byte[] getLineSeparator() { 1241 return lineSeparator; 1242 } 1243 1244 /** 1245 * Tests whether the {@code octet} is in the Base64 alphabet. 1246 * 1247 * @param octet The value to test. 1248 * @return {@code true} if the value is defined in the Base64 alphabet {@code false} otherwise. 1249 */ 1250 @Override 1251 protected boolean isInAlphabet(final byte octet) { 1252 final int value = octet & 0xff; 1253 return value < decodeTable.length && decodeTable[value] != -1; 1254 } 1255 1256 /** 1257 * Tests whether the current encoding mode is URL-safe. 1258 * 1259 * @return true if we're in URL-safe mode, false otherwise. 1260 * @since 1.4 1261 */ 1262 public boolean isUrlSafe() { 1263 return isUrlSafe; 1264 } 1265 1266 /** 1267 * Validates whether decoding the final trailing character is possible in the context of the set of possible Base64 values. 1268 * <p> 1269 * The character is valid if the lower bits within the provided mask are zero. This is used to test the final trailing base-64 digit is zero in the bits 1270 * that will be discarded. 1271 * </p> 1272 * 1273 * @param emptyBitsMask The mask of the lower bits that should be empty. 1274 * @param context The context to be used. 1275 * @throws IllegalArgumentException Thrown if the bits being checked contain any non-zero value. 1276 */ 1277 private void validateCharacter(final int emptyBitsMask, final Context context) { 1278 if (isStrictDecoding() && (context.ibitWorkArea & emptyBitsMask) != 0) { 1279 throw new IllegalArgumentException("Strict decoding: Last encoded character (before the paddings if any) is a valid " + 1280 "Base64 alphabet but not a possible encoding. Expected the discarded bits from the character to be zero."); 1281 } 1282 } 1283 1284 /** 1285 * Validates whether decoding allows an entire final trailing character that cannot be used for a complete byte. 1286 * 1287 * @throws IllegalArgumentException Thrown if strict decoding is enabled. 1288 */ 1289 private void validateTrailingCharacter() { 1290 if (isStrictDecoding()) { 1291 throw new IllegalArgumentException("Strict decoding: Last encoded character (before the paddings if any) is a valid " + 1292 "Base64 alphabet but not a possible encoding. Decoding requires at least two trailing 6-bit characters to create bytes."); 1293 } 1294 } 1295 1296}