001/* 002 * Licensed to the Apache Software Foundation (ASF) under one or more 003 * contributor license agreements. See the NOTICE file distributed with 004 * this work for additional information regarding copyright ownership. 005 * The ASF licenses this file to You under the Apache License, Version 2.0 006 * (the "License"); you may not use this file except in compliance with 007 * the License. You may obtain a copy of the License at 008 * 009 * https://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 014 * See the License for the specific language governing permissions and 015 * limitations under the License. 016 */ 017 018package org.apache.commons.codec.binary; 019 020import java.io.IOException; 021import java.math.BigInteger; 022import java.util.ArrayList; 023import java.util.Arrays; 024import java.util.Objects; 025import java.util.function.Supplier; 026 027import org.apache.commons.codec.BinaryDecoder; 028import org.apache.commons.codec.BinaryEncoder; 029import org.apache.commons.codec.CodecPolicy; 030import org.apache.commons.codec.DecoderException; 031import org.apache.commons.codec.EncoderException; 032 033/** 034 * Abstract superclass for Base-N encoders and decoders. 035 * 036 * <p> 037 * This class is thread-safe. 038 * </p> 039 * <p> 040 * The default decoding policy is lenient. Strict decoding rejects trailing bits that cannot be produced by an encoding, including nonzero unused bits and 041 * impossible counts of final characters. 042 * </p> 043 * 044 * <p> 045 * For {@link Base32} and {@link Base64}, strict decoding additionally requires the exact canonical form produced by this instance's encoder. Re-encoding 046 * successfully decoded input reproduces the input byte for byte. This includes the configured alphabet, padding, line length, and line separator, including 047 * the final line separator when chunking is enabled. Whitespace and alphabet aliases are rejected unless the encoder produces them in that position. 048 * </p> 049 * 050 * <p> 051 * Lenient decoding can map different encoded values to the same bytes. If an application uses encoded values as identifiers for blocklists, replay caches, 052 * or deduplication, validate canonical input before comparing those identifiers, or compare a consistently normalized representation throughout the 053 * application. Decoding alone does not authenticate input; signature verification must use the representation required by the signing protocol. 054 * </p> 055 * 056 * <p> 057 * For example, select canonical Base32 decoding with: 058 * </p> 059 * 060 * <pre> 061 * Base32 base32 = Base32.builder().setDecodingPolicy(CodecPolicy.STRICT).get(); 062 * </pre> 063 * 064 * <p> 065 * This instance requires the uppercase Base32 alphabet, padding for partial blocks, and no line separators. See {@link Base64} for standard and URL-safe 066 * Base64 examples. 067 * </p> 068 * 069 * <p> 070 * Strict validation completes only at the end of the input. When decoding streams, consume the input stream to EOF or finish the output stream with 071 * {@link BaseNCodecOutputStream#eof()} or {@link BaseNCodecOutputStream#close()}. A stream can emit decoded bytes before a later validation error. 072 * </p> 073 */ 074public abstract class BaseNCodec implements BinaryEncoder, BinaryDecoder { 075 076 /** 077 * Builds {@link Base64} instances. 078 * 079 * @param <T> The codec type to build. 080 * @param <B> The codec builder subtype. 081 * @since 1.17.0 082 */ 083 public abstract static class AbstractBuilder<T, B extends AbstractBuilder<T, B>> implements Supplier<T> { 084 085 /** 086 * Clones the given array or returns a default array if the array is null. 087 * 088 * @param array The array to test and clone if not null. 089 * @param defaultArray The default array to return if the array is null. 090 * @return A clone of the array or the default array if the array is null. 091 */ 092 static byte[] clone(final byte[] array, final byte[] defaultArray) { 093 return array != null ? array.clone() : defaultArray; 094 } 095 096 private int unencodedBlockSize; 097 private int encodedBlockSize; 098 private CodecPolicy decodingPolicy = DECODING_POLICY_DEFAULT; 099 private int lineLength; 100 private byte[] lineSeparator = CHUNK_SEPARATOR; 101 private final byte[] defaultEncodeTable; 102 private byte[] encodeTable; 103 private byte[] decodeTable; 104 105 /** Padding byte. */ 106 private byte padding = PAD_DEFAULT; 107 108 AbstractBuilder(final byte[] defaultEncodeTable) { 109 this.defaultEncodeTable = defaultEncodeTable; 110 this.encodeTable = defaultEncodeTable; 111 } 112 113 /** 114 * Returns this instance typed as the subclass type {@code B}. 115 * <p> 116 * This is the same as the expression: 117 * </p> 118 * 119 * <pre> 120 * (B) this 121 * </pre> 122 * 123 * @return {@code this} instance typed as the subclass type {@code B}. 124 */ 125 @SuppressWarnings("unchecked") 126 B asThis() { 127 return (B) this; 128 } 129 130 byte[] getDecodeTable() { 131 return decodeTable; 132 } 133 134 CodecPolicy getDecodingPolicy() { 135 return decodingPolicy; 136 } 137 138 int getEncodedBlockSize() { 139 return encodedBlockSize; 140 } 141 142 byte[] getEncodeTable() { 143 return encodeTable; 144 } 145 146 int getLineLength() { 147 return lineLength; 148 } 149 150 byte[] getLineSeparator() { 151 return lineSeparator; 152 } 153 154 byte getPadding() { 155 return padding; 156 } 157 158 int getUnencodedBlockSize() { 159 return unencodedBlockSize; 160 } 161 162 /** 163 * Sets the decode table. 164 * 165 * @param decodeTable The decode table. 166 * @return {@code this} instance. 167 * @since 1.20.0 168 */ 169 public B setDecodeTable(final byte[] decodeTable) { 170 this.decodeTable = clone(decodeTable, null); 171 return asThis(); 172 } 173 174 /** 175 * Sets the decode table. 176 * 177 * @param decodeTable The decode table, null resets to the default. 178 * @return {@code this} instance. 179 */ 180 B setDecodeTableRaw(final byte[] decodeTable) { 181 this.decodeTable = decodeTable; 182 return asThis(); 183 } 184 185 /** 186 * Sets the decoding policy. 187 * 188 * @param decodingPolicy The decoding policy, null resets to the default. 189 * @return {@code this} instance. 190 */ 191 public B setDecodingPolicy(final CodecPolicy decodingPolicy) { 192 this.decodingPolicy = decodingPolicy != null ? decodingPolicy : DECODING_POLICY_DEFAULT; 193 return asThis(); 194 } 195 196 /** 197 * Sets the encoded block size, subclasses normally set this on construction. 198 * 199 * @param encodedBlockSize The encoded block size, subclasses normally set this on construction. 200 * @return {@code this} instance. 201 */ 202 B setEncodedBlockSize(final int encodedBlockSize) { 203 this.encodedBlockSize = gte0(encodedBlockSize); 204 return asThis(); 205 } 206 207 /** 208 * Sets the encode table. 209 * 210 * @param encodeTable The encode table, null resets to the default. 211 * @return {@code this} instance. 212 */ 213 public B setEncodeTable(final byte... encodeTable) { 214 this.encodeTable = clone(encodeTable, defaultEncodeTable); 215 return asThis(); 216 } 217 218 /** 219 * Sets the encode table. 220 * 221 * @param encodeTable The encode table, null resets to the default. 222 * @return {@code this} instance. 223 */ 224 B setEncodeTableRaw(final byte... encodeTable) { 225 this.encodeTable = encodeTable != null ? encodeTable : defaultEncodeTable; 226 return asThis(); 227 } 228 229 /** 230 * Sets the line length. 231 * 232 * @param lineLength The line length, less than 0 resets to the default. 233 * @return {@code this} instance. 234 */ 235 public B setLineLength(final int lineLength) { 236 this.lineLength = Math.max(0, lineLength); 237 return asThis(); 238 } 239 240 /** 241 * Sets the line separator. 242 * 243 * @param lineSeparator The line separator, null resets to the default. 244 * @return {@code this} instance. 245 */ 246 public B setLineSeparator(final byte... lineSeparator) { 247 this.lineSeparator = clone(lineSeparator , CHUNK_SEPARATOR); 248 return asThis(); 249 } 250 251 /** 252 * Sets the padding byte. 253 * 254 * @param padding The padding byte. 255 * @return {@code this} instance. 256 */ 257 public B setPadding(final byte padding) { 258 this.padding = padding; 259 return asThis(); 260 } 261 262 /** 263 * Sets the unencoded block size, subclasses normally set this on construction. 264 * 265 * @param unencodedBlockSize The unencoded block size, subclasses normally set this on construction. 266 * @return {@code this} instance. 267 */ 268 B setUnencodedBlockSize(final int unencodedBlockSize) { 269 this.unencodedBlockSize = gte0(unencodedBlockSize); 270 return asThis(); 271 } 272 } 273 274 /** 275 * Holds thread context so classes can be thread-safe. 276 * 277 * This class is not itself thread-safe; each thread must allocate its own copy. 278 */ 279 static class Context { 280 281 /** 282 * Placeholder for the bytes we're dealing with for our based logic. Bitwise operations store and extract the encoding or decoding from this variable. 283 */ 284 int ibitWorkArea; 285 286 /** 287 * Placeholder for the bytes we're dealing with for our based logic. Bitwise operations store and extract the encoding or decoding from this variable. 288 */ 289 long lbitWorkArea; 290 291 /** 292 * Buffer for streaming. 293 */ 294 byte[] buffer; 295 296 /** 297 * Position where next character should be written in the buffer. 298 */ 299 int pos; 300 301 /** 302 * Position where next character should be read from the buffer. 303 */ 304 int readPos; 305 306 /** 307 * Boolean flag to indicate the EOF has been reached. Once EOF has been reached, this object becomes useless, and must be thrown away. 308 */ 309 boolean eof; 310 311 /** 312 * Variable tracks how many characters have been written to or strictly decoded from the current line. We use it to make sure each encoded line never 313 * goes beyond lineLength (if lineLength > 0). 314 */ 315 int currentLinePos; 316 317 /** 318 * Writes to the buffer only occur after every 3/5 reads when encoding, and every 4/8 reads when decoding. This variable helps track that. 319 */ 320 int modulus; 321 322 /** 323 * Number of padding bytes consumed by strict decoding. 324 */ 325 int strictPadding; 326 327 /** 328 * Position within the configured line separator during strict decoding. 329 */ 330 int strictSeparatorPos; 331 332 /** 333 * Whether strict decoding has encountered a short final line. 334 */ 335 boolean strictFinalLine; 336 337 /** 338 * Returns a String useful for debugging (especially within a debugger.) 339 * 340 * @return A String useful for debugging. 341 */ 342 @Override 343 public String toString() { 344 return String.format("%s[buffer=%s, currentLinePos=%s, eof=%s, ibitWorkArea=%s, lbitWorkArea=%s, " + "modulus=%s, pos=%s, readPos=%s]", 345 this.getClass().getSimpleName(), Arrays.toString(buffer), currentLinePos, eof, ibitWorkArea, lbitWorkArea, modulus, pos, readPos); 346 } 347 } 348 349 /** 350 * End-of-file marker. 351 * 352 * @since 1.7 353 */ 354 static final int EOF = -1; 355 356 /** 357 * MIME chunk size per RFC 2045 section 6.8. 358 * 359 * <p> 360 * The {@value} character limit does not count the trailing CRLF, but counts all other characters, including any equal signs. 361 * </p> 362 * 363 * @see <a href="https://www.ietf.org/rfc/rfc2045">RFC 2045 section 6.8</a> 364 */ 365 public static final int MIME_CHUNK_SIZE = 76; 366 367 /** 368 * PEM chunk size per RFC 1421 section 4.3.2.4. 369 * 370 * <p> 371 * The {@value} character limit does not count the trailing CRLF, but counts all other characters, including any equal signs. 372 * </p> 373 * 374 * @see <a href="https://tools.ietf.org/html/rfc1421">RFC 1421 section 4.3.2.4</a> 375 */ 376 public static final int PEM_CHUNK_SIZE = 64; 377 private static final int DEFAULT_BUFFER_RESIZE_FACTOR = 2; 378 379 /** 380 * Defines the default buffer size - currently {@value} - must be large enough for at least one encoded block+separator 381 */ 382 private static final int DEFAULT_BUFFER_SIZE = 8192; 383 384 /** 385 * The maximum size buffer to allocate. 386 * 387 * <p> 388 * This is set to the same size used in the JDK {@link ArrayList}: 389 * </p> 390 * <blockquote> Some VMs reserve some header words in an array. Attempts to allocate larger arrays may result in OutOfMemoryError: Requested array size 391 * exceeds VM limit. </blockquote> 392 */ 393 private static final int MAX_BUFFER_SIZE = Integer.MAX_VALUE - 8; 394 395 /** Mask used to extract 8 bits, used in decoding bytes */ 396 protected static final int MASK_8BITS = 0xff; 397 398 /** 399 * Byte used to pad output. 400 */ 401 protected static final byte PAD_DEFAULT = '='; // Allow static access to default 402 403 /** 404 * The default decoding policy. 405 * 406 * @since 1.15 407 */ 408 protected static final CodecPolicy DECODING_POLICY_DEFAULT = CodecPolicy.LENIENT; 409 410 /** 411 * Chunk separator per RFC 2045 section 2.1. 412 * 413 * @see <a href="https://www.ietf.org/rfc/rfc2045">RFC 2045 section 2.1</a> 414 */ 415 static final byte[] CHUNK_SEPARATOR = { '\r', '\n' }; 416 417 /** 418 * The empty byte array. 419 */ 420 static final byte[] EMPTY_BYTE_ARRAY = {}; 421 422 static void code(final boolean doEncode, final BaseNCodec baseNCodec, final byte[] buf, final int offset, final int len, final Context context) 423 throws IOException { 424 try { 425 if (doEncode) { 426 baseNCodec.encode(buf, offset, len, context); 427 } else { 428 baseNCodec.decode(buf, offset, len, context); 429 } 430 } catch (final IllegalArgumentException e) { 431 throw new IOException(e.getMessage(), e); 432 } 433 } 434 435 /** 436 * Create a positive capacity at least as large the minimum required capacity. If the minimum capacity is negative then this throws an OutOfMemoryError as 437 * no array can be allocated. 438 * 439 * @param minCapacity The minimum capacity. 440 * @return The capacity. 441 * @throws OutOfMemoryError Thrown if the {@code minCapacity} is negative. 442 */ 443 private static int createPositiveCapacity(final int minCapacity) { 444 if (minCapacity < 0) { 445 // overflow 446 throw new OutOfMemoryError("Unable to allocate array size: " + (minCapacity & 0xffffffffL)); 447 } 448 // This is called when we require buffer expansion to a very big array. 449 // Use the conservative maximum buffer size if possible, otherwise the biggest required. 450 // 451 // Note: In this situation JDK 1.8 java.util.ArrayList returns Integer.MAX_VALUE. 452 // This excludes some VMs that can exceed MAX_BUFFER_SIZE but not allocate a full 453 // Integer.MAX_VALUE length array. 454 // The result is that we may have to allocate an array of this size more than once if 455 // the capacity must be expanded again. 456 return Math.max(minCapacity, MAX_BUFFER_SIZE); 457 } 458 459 /** 460 * Gets a copy of the chunk separator per RFC 2045 section 2.1. 461 * 462 * @return The chunk separator. 463 * @see <a href="https://www.ietf.org/rfc/rfc2045">RFC 2045 section 2.1</a> 464 * @since 1.15 465 */ 466 public static byte[] getChunkSeparator() { 467 return CHUNK_SEPARATOR.clone(); 468 } 469 470 private static int gte0(final int value) { 471 if (value < 0) { 472 throw new IllegalArgumentException("value must be greater than or equal to 0."); 473 } 474 return value; 475 } 476 477 /** 478 * Tests if a byte value is whitespace or not. 479 * 480 * @param byteToCheck The byte to check. 481 * @return true if byte is whitespace, false otherwise. 482 * @see Character#isWhitespace(int) 483 * @deprecated Use {@link Character#isWhitespace(int)}. 484 */ 485 @Deprecated 486 protected static boolean isWhiteSpace(final byte byteToCheck) { 487 return Character.isWhitespace(byteToCheck); 488 } 489 490 /** 491 * Increases our buffer by the {@link #DEFAULT_BUFFER_RESIZE_FACTOR}. 492 * 493 * @param context The context to be used. 494 * @param minCapacity The minimum required capacity. 495 * @return The resized byte[] buffer. 496 * @throws OutOfMemoryError Thrown if the {@code minCapacity} is negative. 497 */ 498 private static byte[] resizeBuffer(final Context context, final int minCapacity) { 499 // Overflow-conscious code treats the min and new capacity as unsigned. 500 final int oldCapacity = context.buffer.length; 501 int newCapacity = oldCapacity * DEFAULT_BUFFER_RESIZE_FACTOR; 502 if (Integer.compareUnsigned(newCapacity, minCapacity) < 0) { 503 newCapacity = minCapacity; 504 } 505 if (Integer.compareUnsigned(newCapacity, MAX_BUFFER_SIZE) > 0) { 506 newCapacity = createPositiveCapacity(minCapacity); 507 } 508 final byte[] b = Arrays.copyOf(context.buffer, newCapacity); 509 context.buffer = b; 510 return b; 511 } 512 513 /** 514 * Returns a byte-array representation of a {@code BigInteger} without sign bit. 515 * <p> 516 * The value {@link BigInteger#ZERO} maps to an empty array. 517 * </p> 518 * 519 * @param value {@code BigInteger} to be converted. 520 * @return A byte array representation of the BigInteger parameter. 521 */ 522 static byte[] toUnsignedBytes(final BigInteger value) { 523 byte[] unsigned = value.equals(BigInteger.ZERO) ? EMPTY_BYTE_ARRAY : value.toByteArray(); 524 if (unsigned.length > 0 && unsigned[0] == 0) { 525 final byte[] tmp = new byte[unsigned.length - 1]; 526 System.arraycopy(unsigned, 1, tmp, 0, tmp.length); 527 unsigned = tmp; 528 } 529 return unsigned; 530 } 531 532 /** 533 * Deprecated: Will be removed in 2.0. 534 * <p> 535 * Instance variable just in case it needs to vary later 536 * </p> 537 * 538 * @deprecated Use {@link #pad}. Will be removed in 2.0. 539 */ 540 @Deprecated 541 protected final byte PAD = PAD_DEFAULT; 542 543 /** Pad byte. Instance variable just in case it needs to vary later. */ 544 protected final byte pad; 545 546 /** Number of bytes in each full block of unencoded data, for example 4 for Base64 and 5 for Base32 */ 547 private final int unencodedBlockSize; 548 549 /** Number of bytes in each full block of encoded data, for example 3 for Base64 and 8 for Base32 */ 550 private final int encodedBlockSize; 551 552 /** 553 * Chunk size for encoding and strict decoding. A value of zero or less implies no chunking of the encoded data. Rounded down to the nearest multiple of 554 * encodedBlockSize. 555 */ 556 protected final int lineLength; 557 558 /** 559 * Size of chunk separator. Not used unless {@link #lineLength} > 0. 560 */ 561 private final int chunkSeparatorLength; 562 563 /** 564 * Decoding policy, including canonical validation for Base32 and Base64. 565 */ 566 private final CodecPolicy decodingPolicy; 567 568 /** 569 * Decode table to use. 570 */ 571 final byte[] decodeTable; 572 573 /** 574 * Encode table. 575 */ 576 final byte[] encodeTable; 577 578 /** 579 * Constructs a new instance for a subclass. 580 * 581 * @param builder How to build this portion of the instance. 582 * @since 1.20.0 583 */ 584 protected BaseNCodec(final AbstractBuilder<?, ?> builder) { 585 this.unencodedBlockSize = gte0(builder.unencodedBlockSize); 586 this.encodedBlockSize = gte0(builder.encodedBlockSize); 587 final boolean useChunking = builder.lineLength > 0 && builder.lineSeparator.length > 0; 588 this.lineLength = useChunking ? builder.lineLength / builder.encodedBlockSize * builder.encodedBlockSize : 0; 589 this.chunkSeparatorLength = builder.lineSeparator.length; 590 this.pad = builder.padding; 591 this.decodingPolicy = Objects.requireNonNull(builder.decodingPolicy, "codecPolicy"); 592 this.encodeTable = Objects.requireNonNull(builder.getEncodeTable(), "builder.getEncodeTable()"); 593 this.decodeTable = builder.getDecodeTable(); 594 } 595 596 /** 597 * Constructs a new instance. 598 * <p> 599 * Note {@code lineLength} is rounded down to the nearest multiple of the encoded block size. If {@code chunkSeparatorLength} is zero, then chunking is 600 * disabled. 601 * </p> 602 * 603 * @param unencodedBlockSize The size of an unencoded block (for example Base64 = 3). 604 * @param encodedBlockSize The size of an encoded block (for example Base64 = 4). 605 * @param lineLength if > 0, use chunking with a length {@code lineLength}. 606 * @param chunkSeparatorLength The chunk separator length, if relevant. 607 * @deprecated Use {@link BaseNCodec#BaseNCodec(AbstractBuilder)}. 608 */ 609 @Deprecated 610 protected BaseNCodec(final int unencodedBlockSize, final int encodedBlockSize, final int lineLength, final int chunkSeparatorLength) { 611 this(unencodedBlockSize, encodedBlockSize, lineLength, chunkSeparatorLength, PAD_DEFAULT); 612 } 613 614 /** 615 * Constructs a new instance. 616 * <p> 617 * Note {@code lineLength} is rounded down to the nearest multiple of the encoded block size. If {@code chunkSeparatorLength} is zero, then chunking is 618 * disabled. 619 * </p> 620 * 621 * @param unencodedBlockSize The size of an unencoded block (for example Base64 = 3). 622 * @param encodedBlockSize The size of an encoded block (for example Base64 = 4). 623 * @param lineLength if > 0, use chunking with a length {@code lineLength}. 624 * @param chunkSeparatorLength The chunk separator length, if relevant. 625 * @param pad byte used as padding byte. 626 * @deprecated Use {@link BaseNCodec#BaseNCodec(AbstractBuilder)}. 627 */ 628 @Deprecated 629 protected BaseNCodec(final int unencodedBlockSize, final int encodedBlockSize, final int lineLength, final int chunkSeparatorLength, final byte pad) { 630 this(unencodedBlockSize, encodedBlockSize, lineLength, chunkSeparatorLength, pad, DECODING_POLICY_DEFAULT); 631 } 632 633 /** 634 * Constructs a new instance. 635 * <p> 636 * Note {@code lineLength} is rounded down to the nearest multiple of the encoded block size. If {@code chunkSeparatorLength} is zero, then chunking is 637 * disabled. 638 * </p> 639 * 640 * @param unencodedBlockSize The size of an unencoded block (for example Base64 = 3). 641 * @param encodedBlockSize The size of an encoded block (for example Base64 = 4). 642 * @param lineLength if > 0, use chunking with a length {@code lineLength}. 643 * @param chunkSeparatorLength The chunk separator length, if relevant. 644 * @param pad byte used as padding byte. 645 * @param decodingPolicy Decoding policy. 646 * @since 1.15 647 * @deprecated Use {@link BaseNCodec#BaseNCodec(AbstractBuilder)}. 648 */ 649 @Deprecated 650 protected BaseNCodec(final int unencodedBlockSize, final int encodedBlockSize, final int lineLength, final int chunkSeparatorLength, final byte pad, 651 final CodecPolicy decodingPolicy) { 652 this.unencodedBlockSize = unencodedBlockSize; 653 this.encodedBlockSize = encodedBlockSize; 654 final boolean useChunking = lineLength > 0 && chunkSeparatorLength > 0; 655 this.lineLength = useChunking ? lineLength / encodedBlockSize * encodedBlockSize : 0; 656 this.chunkSeparatorLength = chunkSeparatorLength; 657 this.pad = pad; 658 this.decodingPolicy = Objects.requireNonNull(decodingPolicy, "codecPolicy"); 659 this.encodeTable = null; 660 this.decodeTable = null; 661 } 662 663 /** 664 * Returns the amount of buffered data available for reading. 665 * 666 * @param context The context to be used. 667 * @return The amount of buffered data available for reading. 668 */ 669 int available(final Context context) { // package protected for access from I/O streams 670 return hasData(context) ? context.pos - context.readPos : 0; 671 } 672 673 /** 674 * Tests a given byte array to see if it contains any characters within the alphabet or PAD. 675 * 676 * Intended for use in checking line-ending arrays. 677 * 678 * @param arrayOctet byte array to test. 679 * @return {@code true} if any byte is a valid character in the alphabet or PAD; {@code false} otherwise. 680 */ 681 protected boolean containsAlphabetOrPad(final byte[] arrayOctet) { 682 if (arrayOctet != null) { 683 for (final byte element : arrayOctet) { 684 if (pad == element || isInAlphabet(element)) { 685 return true; 686 } 687 } 688 } 689 return false; 690 } 691 692 /** 693 * Decodes a byte[] containing characters in the Base-N alphabet. 694 * 695 * <p> 696 * Uses this instance's decoding policy. Lenient decoding can accept multiple representations of the same bytes. For canonical Base32 or Base64 input, 697 * configure {@link CodecPolicy#STRICT}; see the class documentation for examples and guidance on comparing encoded values. 698 * </p> 699 * 700 * @param array A byte array containing Base-N character data. 701 * @return A byte array containing binary data. 702 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 703 */ 704 @Override 705 public byte[] decode(final byte[] array) { 706 if (BinaryCodec.isEmpty(array)) { 707 return array; 708 } 709 final Context context = new Context(); 710 decode(array, 0, array.length, context); 711 decode(array, 0, EOF, context); // Notify decoder of EOF. 712 final byte[] result = new byte[context.pos]; 713 readResults(result, 0, result.length, context); 714 return result; 715 } 716 717 /** 718 * Decodes a byte[] containing characters in the Base-N alphabet into a temporary context buffer. 719 * <p> 720 * This method is package protected for access from I/O streams. 721 * </p> 722 * 723 * @param array A byte array containing Base-N character data. 724 * @param offset initial offset of the subarray. 725 * @param length length of the subarray. 726 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 727 */ 728 abstract void decode(byte[] array, int offset, int length, Context context); 729 730 /** 731 * Decodes an Object using the Base-N algorithm. This method is provided in order to satisfy the requirements of the Decoder interface, and will throw a 732 * DecoderException if the supplied object is not of type byte[] or String. 733 * 734 * <p> 735 * Uses this instance's decoding policy. Lenient decoding can accept multiple representations of the same bytes. For canonical Base32 or Base64 input, 736 * configure {@link CodecPolicy#STRICT}; see the class documentation for examples and guidance on comparing encoded values. 737 * </p> 738 * 739 * @param obj Object to decode. 740 * @return An object (of type byte[]) containing the binary data which corresponds to the byte[] or String supplied. 741 * @throws DecoderException Thrown if the parameter supplied is not of type byte[]. 742 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 743 */ 744 @Override 745 public Object decode(final Object obj) throws DecoderException { 746 if (obj instanceof byte[]) { 747 return decode((byte[]) obj); 748 } 749 if (obj instanceof String) { 750 return decode((String) obj); 751 } 752 throw new DecoderException("Parameter supplied to Base-N decode is not a byte[] or a String"); 753 } 754 755 /** 756 * Decodes a String containing characters in the Base-N alphabet. 757 * 758 * <p> 759 * Uses this instance's decoding policy. Lenient decoding can accept multiple representations of the same bytes. For canonical Base32 or Base64 input, 760 * configure {@link CodecPolicy#STRICT}; see the class documentation for examples and guidance on comparing encoded values. 761 * </p> 762 * 763 * @param array A String containing Base-N character data. 764 * @return A byte array containing binary data. 765 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 766 */ 767 public byte[] decode(final String array) { 768 return decode(StringUtils.getBytesUtf8(array)); 769 } 770 771 /** 772 * Encodes a byte[] containing binary data, into a byte[] containing characters in the alphabet. 773 * 774 * @param array A byte array containing binary data. 775 * @return A byte array containing only the base N alphabetic character data. 776 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 777 */ 778 @Override 779 public byte[] encode(final byte[] array) { 780 if (BinaryCodec.isEmpty(array)) { 781 return array; 782 } 783 return encode(array, 0, array.length); 784 } 785 786 /** 787 * Encodes a byte[] containing binary data, into a byte[] containing characters in the alphabet. 788 * 789 * @param array A byte array containing binary data. 790 * @param offset initial offset of the subarray. 791 * @param length length of the subarray. 792 * @return A byte array containing only the base N alphabetic character data. 793 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 794 * @since 1.11 795 */ 796 public byte[] encode(final byte[] array, final int offset, final int length) { 797 if (BinaryCodec.isEmpty(array)) { 798 return array; 799 } 800 final Context context = new Context(); 801 encode(array, offset, length, context); 802 encode(array, offset, EOF, context); // Notify encoder of EOF. 803 final byte[] buf = new byte[context.pos - context.readPos]; 804 readResults(buf, 0, buf.length, context); 805 return buf; 806 } 807 808 /** 809 * Encodes a byte[] containing characters in the Base-N alphabet into a temporary context buffer. 810 * <p> 811 * This method is package protected for access from I/O streams. 812 * </p> 813 * 814 * @param array A byte array containing Base-N character data. 815 * @param offset initial offset of the subarray. 816 * @param length length of the subarray. 817 * @throws IllegalArgumentException Thrown when a problem is detected processing data. 818 */ 819 abstract void encode(byte[] array, int offset, int length, Context context); 820 821 /** 822 * Encodes an Object using the Base-N algorithm. This method is provided in order to satisfy the requirements of the Encoder interface, and will throw an 823 * EncoderException if the supplied object is not of type byte[]. 824 * 825 * @param obj Object to encode. 826 * @return An object (of type byte[]) containing the Base-N encoded data which corresponds to the byte[] supplied. 827 * @throws EncoderException Thrown if the parameter supplied is not of type byte[]. 828 */ 829 @Override 830 public Object encode(final Object obj) throws EncoderException { 831 if (!(obj instanceof byte[])) { 832 throw new EncoderException("Parameter supplied to Base-N encode is not a byte[]"); 833 } 834 return encode((byte[]) obj); 835 } 836 837 /** 838 * Encodes a byte[] containing binary data, into a String containing characters in the appropriate alphabet. Uses UTF8 encoding. 839 * <p> 840 * This is a duplicate of {@link #encodeToString(byte[])}; it was merged during refactoring. 841 * </p> 842 * 843 * @param array A byte array containing binary data. 844 * @return String containing only character data in the appropriate alphabet. 845 * @since 1.5 846 */ 847 public String encodeAsString(final byte[] array) { 848 return StringUtils.newStringUtf8(encode(array)); 849 } 850 851 /** 852 * Encodes a byte[] containing binary data, into a String containing characters in the Base-N alphabet. Uses UTF8 encoding. 853 * 854 * @param array A byte array containing binary data. 855 * @return A String containing only Base-N character data. 856 */ 857 public String encodeToString(final byte[] array) { 858 return StringUtils.newStringUtf8(encode(array)); 859 } 860 861 /** 862 * Ensures that the buffer has room for {@code size} bytes 863 * 864 * @param size minimum spare space required. 865 * @param context The context to be used. 866 * @return The buffer. 867 */ 868 protected byte[] ensureBufferSize(final int size, final Context context) { 869 if (context.buffer == null) { 870 context.buffer = new byte[Math.max(size, getDefaultBufferSize())]; 871 context.pos = 0; 872 context.readPos = 0; 873 // Overflow-conscious: 874 // x + y > z == x + y - z > 0 875 } else if (context.pos + size - context.buffer.length > 0) { 876 return resizeBuffer(context, context.pos + size); 877 } 878 return context.buffer; 879 } 880 881 /** 882 * Gets the decoding behavior policy. 883 * 884 * <p> 885 * The default is lenient. Strict decoding rejects invalid trailing bits and, for Base32 and Base64, noncanonical input as described in this class. 886 * </p> 887 * 888 * @return The decoding policy. 889 * @since 1.15 890 */ 891 public CodecPolicy getCodecPolicy() { 892 return decodingPolicy; 893 } 894 895 /** 896 * Gets the default buffer size. Can be overridden. 897 * 898 * @return The default buffer size. 899 */ 900 protected int getDefaultBufferSize() { 901 return DEFAULT_BUFFER_SIZE; 902 } 903 904 /** 905 * Gets the amount of space needed to encode the supplied array. 906 * 907 * @param array byte[] array which will later be encoded. 908 * @return amount of space needed to encode the supplied array. Returns a long since a max-len array will require > Integer.MAX_VALUE. 909 */ 910 public long getEncodedLength(final byte[] array) { 911 // Calculate non-chunked size - rounded up to allow for padding 912 // cast to long is needed to avoid possibility of overflow 913 long len = (array.length + unencodedBlockSize - 1) / unencodedBlockSize * (long) encodedBlockSize; 914 if (lineLength > 0) { // We're using chunking 915 // Round up to nearest multiple 916 len += (len + lineLength - 1) / lineLength * chunkSeparatorLength; 917 } 918 return len; 919 } 920 921 /** 922 * Tests whether this object has buffered data for reading. 923 * 924 * @param context The context to be used. 925 * @return true if there is data still available for reading. 926 */ 927 boolean hasData(final Context context) { // package protected for access from I/O streams 928 return context.pos > context.readPos; 929 } 930 931 /** 932 * Tests whether or not the {@code octet} is in the current alphabet. Does not allow whitespace or pad. 933 * 934 * @param value The value to test. 935 * @return {@code true} if the value is defined in the current alphabet, {@code false} otherwise. 936 */ 937 protected abstract boolean isInAlphabet(byte value); 938 939 /** 940 * Tests a given byte array to see if it contains only valid characters within the alphabet. The method optionally treats whitespace and pad as valid. 941 * 942 * @param arrayOctet byte array to test. 943 * @param allowWhitespacePad if {@code true}, then whitespace and PAD are also allowed. 944 * @return {@code true} if all bytes are valid characters in the alphabet or if the byte array is empty; {@code false}, otherwise. 945 */ 946 public boolean isInAlphabet(final byte[] arrayOctet, final boolean allowWhitespacePad) { 947 for (final byte octet : arrayOctet) { 948 if (!isInAlphabet(octet) && (!allowWhitespacePad || octet != pad && !Character.isWhitespace(octet))) { 949 return false; 950 } 951 } 952 return true; 953 } 954 955 /** 956 * Tests a given String to see if it contains only valid characters within the alphabet. The method treats whitespace and PAD as valid. 957 * 958 * @param basen String to test. 959 * @return {@code true} if all characters in the String are valid characters in the alphabet or if the String is empty; {@code false}, otherwise. 960 * @see #isInAlphabet(byte[], boolean) 961 */ 962 public boolean isInAlphabet(final String basen) { 963 return isInAlphabet(StringUtils.getBytesUtf8(basen), true); 964 } 965 966 /** 967 * Tests whether decoding behavior is strict. 968 * 969 * <p> 970 * Strict decoding rejects invalid trailing bits and, for Base32 and Base64, noncanonical input as described in this class. 971 * </p> 972 * 973 * @return true if using strict decoding. 974 * @since 1.15 975 */ 976 public boolean isStrictDecoding() { 977 return decodingPolicy == CodecPolicy.STRICT; 978 } 979 980 /** 981 * Reads buffered data into the provided byte[] array, starting at position bPos, up to a maximum of bAvail bytes. Returns how many bytes were actually 982 * extracted. 983 * <p> 984 * Package private for access from I/O streams. 985 * </p> 986 * 987 * @param b byte[] array to extract the buffered data into. 988 * @param position position in byte[] array to start extraction at. 989 * @param available amount of bytes we're allowed to extract. We may extract fewer (if fewer are available). 990 * @param context The context to be used. 991 * @return The number of bytes successfully extracted into the provided byte[] array. 992 */ 993 int readResults(final byte[] b, final int position, final int available, final Context context) { 994 if (hasData(context)) { 995 final int len = Math.min(available(context), available); 996 System.arraycopy(context.buffer, context.readPos, b, position, len); 997 context.readPos += len; 998 if (!hasData(context)) { 999 // All data read. 1000 // Reset position markers but do not set buffer to null to allow its reuse. 1001 // hasData(context) will still return false, and this method will return 0 until 1002 // more data is available, or -1 if EOF. 1003 context.pos = context.readPos = 0; 1004 } 1005 return len; 1006 } 1007 return context.eof ? EOF : 0; 1008 } 1009 1010 /** 1011 * Validates a byte against the canonical Base32 or Base64 encoding, consuming padding and line separators. 1012 * 1013 * @param value The unsigned input byte. 1014 * @param lineSeparator The configured line separator. 1015 * @param padded Whether the encoder pads partial blocks. 1016 * @param context The decoding context, whose modulus counts alphabet characters only. 1017 * @return Whether the byte is an alphabet character to decode. 1018 * @throws IllegalArgumentException Thrown if the byte cannot occur in a canonical encoding. 1019 */ 1020 boolean validateCanonicalByte(final int value, final byte[] lineSeparator, final boolean padded, final Context context) { 1021 if (lineLength > 0 && (context.strictSeparatorPos > 0 || context.currentLinePos == lineLength || 1022 value == (lineSeparator[0] & MASK_8BITS))) { 1023 if (context.currentLinePos == 0 || value != (lineSeparator[context.strictSeparatorPos] & MASK_8BITS)) { 1024 throw new IllegalArgumentException("Strict decoding: Invalid line separator or line length."); 1025 } 1026 if (context.strictSeparatorPos == 0) { 1027 validateCanonicalPadding(padded, context); 1028 context.strictFinalLine = context.currentLinePos < lineLength; 1029 } 1030 if (++context.strictSeparatorPos == lineSeparator.length) { 1031 context.strictSeparatorPos = 0; 1032 context.currentLinePos = 0; 1033 } 1034 return false; 1035 } 1036 if (context.strictFinalLine) { 1037 throw new IllegalArgumentException("Strict decoding: Data follows the final line separator."); 1038 } 1039 if (value == (pad & MASK_8BITS)) { 1040 if (!padded || context.modulus == 0 || context.strictPadding >= encodedBlockSize - context.modulus) { 1041 throw new IllegalArgumentException("Strict decoding: Unexpected padding."); 1042 } 1043 context.strictPadding++; 1044 } else { 1045 final int decoded = value < decodeTable.length ? decodeTable[value] : -1; 1046 if (context.strictPadding != 0 || decoded < 0 || decoded >= encodeTable.length || (encodeTable[decoded] & MASK_8BITS) != value) { 1047 throw new IllegalArgumentException("Strict decoding: Unexpected character or data after padding."); 1048 } 1049 } 1050 if (lineLength > 0) { 1051 context.currentLinePos++; 1052 } 1053 return value != (pad & MASK_8BITS); 1054 } 1055 1056 /** 1057 * Validates the end of a canonical Base32 or Base64 encoding. 1058 * 1059 * @param padded Whether the encoder pads partial blocks. 1060 * @param context The decoding context. 1061 * @throws IllegalArgumentException Thrown if padding or the final line separator is incomplete. 1062 */ 1063 void validateCanonicalEnd(final boolean padded, final Context context) { 1064 validateCanonicalPadding(padded, context); 1065 if (context.strictSeparatorPos != 0 || context.currentLinePos != 0) { 1066 throw new IllegalArgumentException("Strict decoding: Missing or incomplete final line separator."); 1067 } 1068 } 1069 1070 /** 1071 * Validates the number of padding bytes at the end of a line or input. 1072 * 1073 * @param padded Whether the encoder pads partial blocks. 1074 * @param context The decoding context. 1075 * @throws IllegalArgumentException Thrown if required padding is missing. 1076 */ 1077 private void validateCanonicalPadding(final boolean padded, final Context context) { 1078 if (padded && context.modulus != 0 && context.strictPadding != encodedBlockSize - context.modulus) { 1079 throw new IllegalArgumentException("Strict decoding: Incorrect padding length."); 1080 } 1081 } 1082}