001/*
002 * Licensed to the Apache Software Foundation (ASF) under one or more
003 * contributor license agreements.  See the NOTICE file distributed with
004 * this work for additional information regarding copyright ownership.
005 * The ASF licenses this file to You under the Apache License, Version 2.0
006 * (the "License"); you may not use this file except in compliance with
007 * the License.  You may obtain a copy of the License at
008 *
009 *      https://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
014 * See the License for the specific language governing permissions and
015 * limitations under the License.
016 */
017
018package org.apache.commons.codec.binary;
019
020import java.math.BigInteger;
021import java.util.Arrays;
022import java.util.Objects;
023
024import org.apache.commons.codec.CodecPolicy;
025
026/**
027 * Provides Base64 encoding and decoding as defined by <a href="https://www.ietf.org/rfc/rfc2045">RFC 2045 Multipurpose Internet Mail Extensions (MIME) Part
028 * One: Format of Internet Message Bodies</a> and portions of <a href="https://datatracker.ietf.org/doc/html/rfc4648">RFC 4648 The Base16, Base32, and Base64
029 * Data Encodings</a>
030 *
031 * <p>
032 * This class implements <a href="https://www.ietf.org/rfc/rfc2045#section-6.8">RFC 2045 6.8. Base64 Content-Transfer-Encoding</a>.
033 * </p>
034 * <p>
035 * The class can be parameterized in the following manner with its {@link Builder}:
036 * </p>
037 * <ul>
038 * <li>URL-safe mode: Default off.</li>
039 * <li>Line length: Default 76. Line length that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data.
040 * <li>Line separator: Default is CRLF ({@code "\r\n"})</li>
041 * <li>Strict or lenient decoding policy; default is {@link CodecPolicy#LENIENT}.</li>
042 * <li>Custom decoding table.</li>
043 * <li>Custom encoding table.</li>
044 * <li>Padding; defaults is {@code '='}.</li>
045 * </ul>
046 * <p>
047 * The URL-safe parameter selects the encoding alphabet. Lenient decoding seamlessly handles both modes; strict decoding requires the encoding alphabet. See also
048 * {@code Builder#setDecodeTableFormat(DecodeTableFormat)}.
049 * </p>
050 * <p>
051 * Since this class operates directly on byte streams, and not character streams, it is hard-coded to only encode/decode character encodings which are
052 * compatible with the lower 127 ASCII chart (ISO-8859-1, Windows-1252, UTF-8, etc).
053 * </p>
054 * <p>
055 * This class is thread-safe.
056 * </p>
057 * <p>
058 * To configure a new instance, use a {@link Builder}. For example:
059 * </p>
060 *
061 * <pre>
062 * Base64 base64 = Base64.builder()
063 *   .setDecodingPolicy(CodecPolicy.LENIENT)    // default is lenient, null resets to default
064 *   .setEncodeTable(customEncodeTable)         // default is built in, null resets to default
065 *   .setLineLength(0)                          // default is none
066 *   .setLineSeparator('\r', '\n')              // default is CR LF, null resets to default
067 *   .setPadding('=')                           // default is '='
068 *   .setUrlSafe(false)                         // default is false
069 *   .get()
070 * </pre>
071 *
072 * <p>
073 * The static decoding convenience methods use {@link CodecPolicy#LENIENT}. They accept noncanonical input, so different encoded strings can decode to the
074 * same bytes. Selecting a standard or URL-safe decode table does not enable strict validation. To require canonical input, configure a strict instance:
075 * </p>
076 *
077 * <pre>
078 * Base64 standard = Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get();
079 * Base64 urlSafe = Base64.builder().setUrlSafe(true).setDecodingPolicy(CodecPolicy.STRICT).get();
080 * </pre>
081 *
082 * <p>
083 * These instances accept unchunked input using their respective encoding alphabets. The standard instance requires padding for partial blocks; the URL-safe
084 * instance requires unpadded input. See {@link BaseNCodec} for the full canonical decoding contract and guidance on comparing encoded values.
085 * </p>
086 *
087 * @see Base64InputStream
088 * @see Base64OutputStream
089 * @see <a href="https://www.ietf.org/rfc/rfc2045">RFC 2045 Multipurpose Internet Mail Extensions (MIME) Part One: Format of Internet Message Bodies</a>
090 * @see <a href="https://datatracker.ietf.org/doc/html/rfc4648">RFC 4648 The Base16, Base32, and Base64 Data Encodings</a>
091 * @since 1.0
092 */
093public class Base64 extends BaseNCodec {
094
095    /**
096     * Builds {@link Base64} instances.
097     *
098     * <p>
099     * To configure a new instance, use a {@link Builder}. For example:
100     * </p>
101     *
102     * <pre>
103     * Base64 base64 = Base64.builder()
104     *   .setCodecPolicy(CodecPolicy.LENIENT)       // default is lenient, null resets to default
105     *   .setEncodeTable(customEncodeTable)         // default is built in, null resets to default
106     *   .setLineLength(0)                          // default is none
107     *   .setLineSeparator('\r', '\n')              // default is CR LF, null resets to default
108     *   .setPadding('=')                           // default is '='
109     *   .setUrlSafe(false)                         // default is false
110     *   .get()
111     * </pre>
112     *
113     * @since 1.17.0
114     */
115    public static class Builder extends AbstractBuilder<Base64, Builder> {
116
117        /**
118         * Constructs a new instance.
119         */
120        public Builder() {
121            super(STANDARD_ENCODE_TABLE);
122            setDecodeTableRaw(DECODE_TABLE);
123            setEncodeTableRaw(STANDARD_ENCODE_TABLE);
124            setEncodedBlockSize(BYTES_PER_ENCODED_BLOCK);
125            setUnencodedBlockSize(BYTES_PER_UNENCODED_BLOCK);
126        }
127
128        @Override
129        public Base64 get() {
130            return new Base64(this);
131        }
132
133        /**
134         * Sets the format of the decoding table. This method allows callers to explicitly state whether a standard or URL-safe Base64 decoding is expected. This method
135         * does not modify behavior on encoding operations. For configuration of the encoding behavior, please use {@link #setUrlSafe(boolean)} method.
136         * <p>
137         * By default, the implementation uses the {@link DecodeTableFormat#MIXED} approach, allowing a seamless handling of both
138         * {@link DecodeTableFormat#URL_SAFE} and {@link DecodeTableFormat#STANDARD} base64 in lenient mode. Strict decoding additionally requires each character
139         * to match the configured encoding table.
140         * </p>
141         *
142         * @param format table format to be used on Base64 decoding. Use {@link DecodeTableFormat#MIXED} or null to reset to the default behavior.
143         * @return {@code this} instance.
144         * @since 1.21
145         */
146        public Builder setDecodeTableFormat(final DecodeTableFormat format) {
147            if (format == null) {
148                return setDecodeTableRaw(DECODE_TABLE);
149            }
150            switch (format) {
151                case STANDARD:
152                    return setDecodeTableRaw(STANDARD_DECODE_TABLE);
153                case URL_SAFE:
154                    return setDecodeTableRaw(URL_SAFE_DECODE_TABLE);
155                case MIXED:
156                default:
157                    return setDecodeTableRaw(DECODE_TABLE);
158            }
159        }
160
161        /**
162         * Sets the encode table.
163         *
164         * @param encodeTable The encode table with exactly 64 unique entries, null resets to the default.
165         * @return {@code this} instance.
166         * @throws IllegalArgumentException Thrown if {@code encodeTable} does not contain exactly 64 unique entries.
167         */
168        @Override
169        public Builder setEncodeTable(final byte... encodeTable) {
170            setDecodeTableRaw(toDecodeTable(encodeTable));
171            return super.setEncodeTable(encodeTable);
172        }
173
174        /**
175         * Sets the URL-safe encoding policy.
176         * <p>
177         * Strict decoding requires this alphabet and its padding convention. Lenient decoding accepts both alphabets by default; use
178         * {@code Builder.setDecodeTableFormat(DecodeTableFormat)} to select a decoding table.
179         * </p>
180         *
181         * @param urlSafe URL-safe encoding policy.
182         * @return {@code this} instance.
183         */
184        public Builder setUrlSafe(final boolean urlSafe) {
185            // Javadoc 8 can't find {@link #setDecodeTableFormat(DecodeTableFormat)}
186            return setEncodeTable(toUrlSafeEncodeTable(urlSafe));
187        }
188
189    }
190
191    /**
192     * Enumerates the Base64 table format to be used on decoding.
193     * <p>
194     * By default, the method uses {@link DecodeTableFormat#MIXED} approach, allowing a seamless handling of both {@link DecodeTableFormat#URL_SAFE} and
195     * {@link DecodeTableFormat#STANDARD} base64 options.
196     * </p>
197     *
198     * @since 1.21
199     */
200    public enum DecodeTableFormat {
201
202        /**
203         * Corresponds to the standard Base64 coding table, as specified in
204         * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>.
205         */
206        STANDARD,
207
208        /**
209         * Corresponds to the URL-safe Base64 coding table, as specified in
210         * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC
211         * 4648 Table 2: The "URL and Filename safe" Base 64 Alphabet</a>.
212         */
213        URL_SAFE,
214
215        /**
216         * Represents a joint approach, allowing a seamless decoding of both character sets, corresponding to either
217         * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a> or
218         * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC
219         * 4648 Table 2: The "URL and Filename safe" Base 64 Alphabet</a>. This decoding table is used by default.
220         */
221        MIXED
222    }
223
224    /**
225     * BASE64 characters are 6 bits in length.
226     * They are formed by taking a block of 3 octets to form a 24-bit string,
227     * which is converted into 4 BASE64 characters.
228     */
229    private static final int BITS_PER_ENCODED_BYTE = 6;
230    private static final int BYTES_PER_UNENCODED_BLOCK = 3;
231    private static final int BYTES_PER_ENCODED_BLOCK = 4;
232    private static final int DECODING_TABLE_LENGTH = 256;
233
234    /**
235     * This array is a lookup table that translates 6-bit positive integer index values into their "Base64 Alphabet" equivalents as specified in
236     * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>.
237     * <p>
238     * Thanks to "commons" project in ws.apache.org for this code. https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/
239     * </p>
240     */
241    // @formatter:off
242    private static final byte[] STANDARD_ENCODE_TABLE = {
243            'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
244            'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
245            'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm',
246            'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z',
247            '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '+', '/'
248    };
249
250    /**
251     * This is a copy of the STANDARD_ENCODE_TABLE above, but with + and / changed to - and _ to make the encoded Base64 results more URL-SAFE. This table is
252     * only used when the Base64's mode is set to URL-SAFE.
253     */
254    // @formatter:off
255    private static final byte[] URL_SAFE_ENCODE_TABLE = {
256            'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
257            'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
258            'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm',
259            'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z',
260            '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', '-', '_'
261    };
262    // @formatter:on
263
264    /**
265     * This array is a lookup table that translates Unicode characters drawn from the "Base64 Alphabet" (as specified in
266     * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>) into their 6-bit
267     * positive integer equivalents. Characters that are not in the Base64 or Base64 URL-safe alphabets but fall within the bounds of the array are translated
268     * to -1.
269     * <p>
270     * The characters '+' and '-' both decode to 62. '/' and '_' both decode to 63. This means decoder seamlessly handles both URL_SAFE and STANDARD base64.
271     * (The encoder, on the other hand, needs to know ahead of time what to emit).
272     * </p>
273     * <p>
274     * Thanks to "commons" project in ws.apache.org for this code. https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/
275     * </p>
276     */
277    private static final byte[] DECODE_TABLE = {
278        //   0   1   2   3   4   5   6   7   8   9   A   B   C   D   E   F
279            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f
280            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f
281            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, 62, -1, 62, -1, 63, // 20-2f + - /
282            52, 53, 54, 55, 56, 57, 58, 59, 60, 61, -1, -1, -1, -1, -1, -1, // 30-3f 0-9
283            -1,  0,  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, // 40-4f A-O
284            15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, -1, -1, -1, -1, 63, // 50-5f P-Z _
285            -1, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, // 60-6f a-o
286            41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51                      // 70-7a p-z
287    };
288
289    /**
290     * This array is a lookup table that translates Unicode characters drawn from the "Base64 Alphabet" (as specified in
291     * <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The Base64 Alphabet</a>) into their 6-bit
292     * positive integer equivalents. Characters that are not in the Base64 alphabet but fall within the bounds of the array are translated to -1. This decoding
293     * table handles only the standard base64 characters, such as '+' and '/'. The "url-safe" characters such as '-' and '_' are not supported by the table.
294     */
295    private static final byte[] STANDARD_DECODE_TABLE = {
296        //   0   1   2   3   4   5   6   7   8   9   A   B   C   D   E   F
297            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f
298            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f
299            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, 62, -1, -1, -1, 63, // 20-2f + /
300            52, 53, 54, 55, 56, 57, 58, 59, 60, 61, -1, -1, -1, -1, -1, -1, // 30-3f 0-9
301            -1,  0,  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, // 40-4f A-O
302            15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, -1, -1, -1, -1, -1, // 50-5f P-Z
303            -1, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, // 60-6f a-o
304            41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51                      // 70-7a p-z
305    };
306
307    /**
308     * This array is a lookup table that translates Unicode characters drawn from the "Base64 URL-safe Alphabet" (as specified in
309     * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648
310     * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>) into their 6-bit positive integer equivalents. Characters that are not in the Base64 URL-safe
311     * alphabet but fall within the bounds of the array are translated to -1. This decoding table handles only the URL-safe base64 characters, such as '-' and
312     * '_'. The standard characters such as '+' and '/' are not supported by the table.
313     */
314    private static final byte[] URL_SAFE_DECODE_TABLE = {
315            //   0   1   2   3   4   5   6   7   8   9   A   B   C   D   E   F
316            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f
317            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f
318            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, 62, -1, -1, // 20-2f -
319            52, 53, 54, 55, 56, 57, 58, 59, 60, 61, -1, -1, -1, -1, -1, -1, // 30-3f 0-9
320            -1,  0,  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, // 40-4f A-O
321            15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25, -1, -1, -1, -1, 63, // 50-5f P-Z _
322            -1, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36, 37, 38, 39, 40, // 60-6f a-o
323            41, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51                      // 70-7a p-z
324    };
325
326    /**
327     * Base64 uses 6-bit fields.
328     */
329
330    /** Mask used to extract 6 bits, used when encoding */
331    private static final int MASK_6_BITS = 0x3f;
332
333    // The static final fields above are used for the original static byte[] methods on Base64.
334    // The private member fields below are used with the new streaming approach, which requires
335    // some state be preserved between calls of encode() and decode().
336
337    /** Mask used to extract 4 bits, used when decoding final trailing character. */
338    private static final int MASK_4_BITS = 0xf;
339
340    /** Mask used to extract 2 bits, used when decoding final trailing character. */
341    private static final int MASK_2_BITS = 0x3;
342
343    /**
344     * Creates a new Builder.
345     *
346     * <p>
347     * To configure a new instance, use a {@link Builder}. For example:
348     * </p>
349     *
350     * <pre>
351     * Base64 base64 = Base64.builder()
352     *   .setDecodingPolicy(CodecPolicy.LENIENT) // default is lenient, null resets to default
353     *   .setEncodeTable(customEncodeTable)         // default is built in, null resets to default
354     *   .setLineLength(0)                          // default is none
355     *   .setLineSeparator('\r', '\n')              // default is CR LF, null resets to default
356     *   .setPadding('=')                           // default is '='
357     *   .setUrlSafe(false)                         // default is false
358     *   .get()
359     * </pre>
360     *
361     * @return A new Builder.
362     * @since 1.17.0
363     */
364    public static Builder builder() {
365        return new Builder();
366    }
367
368    /**
369     * Calculates a decode table for a given encode table.
370     *
371     * @param encodeTable that is used to determine decode lookup table.
372     * @return A new decode table.
373     */
374    private static byte[] calculateDecodeTable(final byte[] encodeTable) {
375        if (encodeTable.length != STANDARD_ENCODE_TABLE.length) {
376            throw new IllegalArgumentException("encodeTable must have exactly 64 entries.");
377        }
378        final byte[] decodeTable = new byte[DECODING_TABLE_LENGTH];
379        Arrays.fill(decodeTable, (byte) -1);
380        for (int i = 0; i < encodeTable.length; i++) {
381            final int encodedByte = encodeTable[i] & 0xff;
382            if (decodeTable[encodedByte] != -1) {
383                throw new IllegalArgumentException("encodeTable must not contain duplicate entries.");
384            }
385            decodeTable[encodedByte] = (byte) i;
386        }
387        return decodeTable;
388    }
389
390    private static boolean contains(final byte[] bytes, final byte value) {
391        for (final byte element : bytes) {
392            if (element == value) {
393                return true;
394            }
395        }
396        return false;
397    }
398
399    /**
400     * Decodes Base64 data into octets using lenient decoding.
401     *
402     * <p>
403     * This method uses the standard and URL-safe alphabets. It skips unsupported input, discards data after the first padding character, and accepts
404     * noncanonical padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical
405     * input.
406     * </p>
407     *
408     * <p>
409     * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64Data)}.
410     * See {@link BaseNCodec} for guidance on comparing encoded values.
411     * </p>
412     *
413     * @param base64Data Byte array containing Base64 data.
414     * @return New array containing decoded data.
415     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
416     */
417    public static byte[] decodeBase64(final byte[] base64Data) {
418        return new Base64().decode(base64Data);
419    }
420
421    /**
422     * Decodes a Base64 string into octets using lenient decoding.
423     *
424     * <p>
425     * This method uses the standard and URL-safe alphabets. It skips unsupported input, discards data after the first padding character, and accepts
426     * noncanonical padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical
427     * input.
428     * </p>
429     *
430     * <p>
431     * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64String)}.
432     * See {@link BaseNCodec} for guidance on comparing encoded values.
433     * </p>
434     *
435     * @param base64String String containing Base64 data.
436     * @return New array containing decoded data.
437     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
438     * @since 1.4
439     */
440    public static byte[] decodeBase64(final String base64String) {
441        return new Base64().decode(base64String);
442    }
443
444    /**
445     * Decodes standard Base64 data into octets using lenient decoding.
446     *
447     * <p>
448     * This method uses the standard alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical
449     * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input.
450     * </p>
451     *
452     * <p>
453     * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64Data)}.
454     * See {@link BaseNCodec} for guidance on comparing encoded values.
455     * </p>
456     *
457     * @param base64Data Byte array containing Base64 data.
458     * @return New array containing decoded data.
459     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
460     * @since 1.21
461     */
462    public static byte[] decodeBase64Standard(final byte[] base64Data) {
463        return builder().setDecodeTableFormat(DecodeTableFormat.STANDARD).get().decode(base64Data);
464    }
465
466    /**
467     * Decodes a standard Base64 string into octets using lenient decoding.
468     *
469     * <p>
470     * This method uses the standard alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical
471     * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input.
472     * </p>
473     *
474     * <p>
475     * For canonical decoding, use {@code Base64.builder().setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64String)}.
476     * See {@link BaseNCodec} for guidance on comparing encoded values.
477     * </p>
478     *
479     * @param base64String String containing Base64 data.
480     * @return New array containing decoded data.
481     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
482     * @since 1.21
483     */
484    public static byte[] decodeBase64Standard(final String base64String) {
485        return builder().setDecodeTableFormat(DecodeTableFormat.STANDARD).get().decode(base64String);
486    }
487
488    /**
489     * Decodes URL-safe Base64 data into octets using lenient decoding.
490     *
491     * <p>
492     * This method uses the URL-safe alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical
493     * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input.
494     * </p>
495     *
496     * <p>
497     * For canonical decoding, use {@code Base64.builder().setUrlSafe(true).setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64Data)}.
498     * See {@link BaseNCodec} for guidance on comparing encoded values.
499     * </p>
500     *
501     * @param base64Data Byte array containing Base64 data.
502     * @return New array containing decoded data.
503     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
504     * @since 1.21
505     */
506    public static byte[] decodeBase64UrlSafe(final byte[] base64Data) {
507        return builder().setDecodeTableFormat(DecodeTableFormat.URL_SAFE).get().decode(base64Data);
508    }
509
510    /**
511     * Decodes a URL-safe Base64 string into octets using lenient decoding.
512     *
513     * <p>
514     * This method uses the URL-safe alphabet. It skips unsupported input, discards data after the first padding character, and accepts noncanonical
515     * padding and trailing bits. Different encoded inputs can therefore produce the same decoded bytes. This method does not validate canonical input.
516     * </p>
517     *
518     * <p>
519     * For canonical decoding, use {@code Base64.builder().setUrlSafe(true).setDecodingPolicy(CodecPolicy.STRICT).get().decode(base64String)}.
520     * See {@link BaseNCodec} for guidance on comparing encoded values.
521     * </p>
522     *
523     * @param base64String String containing Base64 data.
524     * @return New array containing decoded data.
525     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
526     * @since 1.21
527     */
528    public static byte[] decodeBase64UrlSafe(final String base64String) {
529        return builder().setDecodeTableFormat(DecodeTableFormat.URL_SAFE).get().decode(base64String);
530    }
531
532    /**
533     * Decodes a byte64-encoded integer according to crypto standards such as W3C's XML-Signature.
534     *
535     * @param array A byte array containing base64 character data.
536     * @return A BigInteger.
537     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
538     * @since 1.4
539     */
540    public static BigInteger decodeInteger(final byte[] array) {
541        return new BigInteger(1, decodeBase64(array));
542    }
543
544    /**
545     * Encodes binary data using the base64 algorithm but does not chunk the output.
546     *
547     * @param binaryData binary data to encode.
548     * @return byte[] containing Base64 characters in their UTF-8 representation.
549     */
550    public static byte[] encodeBase64(final byte[] binaryData) {
551        return encodeBase64(binaryData, false);
552    }
553
554    /**
555     * Encodes binary data using the base64 algorithm, optionally chunking the output into 76 character blocks.
556     *
557     * @param binaryData Array containing binary data to encode.
558     * @param isChunked  if {@code true} this encoder will chunk the base64 output into 76 character blocks.
559     * @return Base64-encoded data.
560     * @throws IllegalArgumentException Thrown when the input array needs an output array bigger than {@link Integer#MAX_VALUE}.
561     */
562    public static byte[] encodeBase64(final byte[] binaryData, final boolean isChunked) {
563        return encodeBase64(binaryData, isChunked, false);
564    }
565
566    /**
567     * Encodes binary data using the base64 algorithm, optionally chunking the output into 76 character blocks.
568     *
569     * @param binaryData Array containing binary data to encode.
570     * @param isChunked  if {@code true} this encoder will chunk the base64 output into 76 character blocks.
571     * @param urlSafe    if {@code true} this encoder will emit - and _ instead of the usual + and / characters. <strong>No padding is added when encoding using
572     *                   the URL-safe alphabet.</strong>
573     * @return Base64-encoded data.
574     * @throws IllegalArgumentException Thrown when the input array needs an output array bigger than {@link Integer#MAX_VALUE}.
575     * @since 1.4
576     */
577    public static byte[] encodeBase64(final byte[] binaryData, final boolean isChunked, final boolean urlSafe) {
578        return encodeBase64(binaryData, isChunked, urlSafe, Integer.MAX_VALUE);
579    }
580
581    /**
582     * Encodes binary data using the base64 algorithm, optionally chunking the output into 76 character blocks.
583     *
584     * @param binaryData    Array containing binary data to encode.
585     * @param isChunked     if {@code true} this encoder will chunk the base64 output into 76 character blocks.
586     * @param urlSafe       if {@code true} this encoder will emit - and _ instead of the usual + and / characters. <strong>No padding is added when encoding
587     *                      using the URL-safe alphabet.</strong>
588     * @param maxResultSize The maximum result size to accept.
589     * @return Base64-encoded data.
590     * @throws IllegalArgumentException Thrown when the input array needs an output array bigger than maxResultSize.
591     * @since 1.4
592     */
593    public static byte[] encodeBase64(final byte[] binaryData, final boolean isChunked, final boolean urlSafe, final int maxResultSize) {
594        if (BinaryCodec.isEmpty(binaryData)) {
595            return binaryData;
596        }
597        // Create this so can use the super-class method
598        // Also ensures that the same roundings are performed by the ctor and the code
599        final Base64 b64 = isChunked ? new Base64(urlSafe) : new Base64(0, CHUNK_SEPARATOR, urlSafe);
600        final long len = b64.getEncodedLength(binaryData);
601        if (len > maxResultSize) {
602            throw new IllegalArgumentException(
603                    "Input array too big, the output array would be bigger (" + len + ") than the specified maximum size of " + maxResultSize);
604        }
605        return b64.encode(binaryData);
606    }
607
608    /**
609     * Encodes binary data using the base64 algorithm and chunks the encoded output into 76 character blocks
610     *
611     * @param binaryData binary data to encode.
612     * @return Base64 characters chunked in 76 character blocks.
613     */
614    public static byte[] encodeBase64Chunked(final byte[] binaryData) {
615        return encodeBase64(binaryData, true);
616    }
617
618    /**
619     * Encodes binary data using the base64 algorithm but does not chunk the output.
620     * <p>
621     * <strong> We changed the behavior of this method from multi-line chunking (1.4) to single-line non-chunking (1.5).</strong>
622     * </p>
623     *
624     * @param binaryData binary data to encode.
625     * @return String containing Base64 characters.
626     * @since 1.4 (NOTE: 1.4 chunked the output, whereas 1.5 does not).
627     */
628    public static String encodeBase64String(final byte[] binaryData) {
629        return StringUtils.newStringUsAscii(encodeBase64(binaryData, false));
630    }
631
632    /**
633     * Encodes binary data using a URL-safe variation of the base64 algorithm but does not chunk the output. The url-safe variation emits - and _ instead of +
634     * and / characters. <strong>No padding is added.</strong>
635     *
636     * @param binaryData binary data to encode.
637     * @return byte[] containing Base64 characters in their UTF-8 representation.
638     * @since 1.4
639     */
640    public static byte[] encodeBase64URLSafe(final byte[] binaryData) {
641        return encodeBase64(binaryData, false, true);
642    }
643
644    /**
645     * Encodes binary data using a URL-safe variation of the base64 algorithm but does not chunk the output. The url-safe variation emits - and _ instead of +
646     * and / characters. <strong>No padding is added.</strong>
647     *
648     * @param binaryData binary data to encode.
649     * @return String containing Base64 characters.
650     * @since 1.4
651     */
652    public static String encodeBase64URLSafeString(final byte[] binaryData) {
653        return StringUtils.newStringUsAscii(encodeBase64(binaryData, false, true));
654    }
655
656    /**
657     * Encodes to a byte64-encoded integer according to crypto standards such as W3C's XML-Signature.
658     *
659     * @param bigInteger A BigInteger.
660     * @return A byte array containing base64 character data.
661     * @throws NullPointerException Thrown if null is passed in.
662     * @since 1.4
663     */
664    public static byte[] encodeInteger(final BigInteger bigInteger) {
665        Objects.requireNonNull(bigInteger, "bigInteger");
666        return encodeBase64(toUnsignedBytes(bigInteger), false);
667    }
668
669    /**
670     * Tests a given byte array to see if it contains only valid characters within the Base64 alphabet. Currently the method treats whitespace as valid.
671     *
672     * @param arrayOctet byte array to test.
673     * @return {@code true} if all bytes are valid characters in the Base64 alphabet or if the byte array is empty; {@code false}, otherwise.
674     * @deprecated 1.5 Use {@link #isBase64(byte[])}, will be removed in 2.0.
675     */
676    @Deprecated
677    public static boolean isArrayByteBase64(final byte[] arrayOctet) {
678        return isBase64(arrayOctet);
679    }
680
681    /**
682     * Tests whether or not the {@code octet} is in the Base64 alphabet.
683     * <p>
684     * This method treats all characters included within standard base64 and base64url encodings as valid base64 characters. This includes the '+' and '/'
685     * (standard base64), as well as '-' and '_' (URL-safe base64) characters. To test membership in only the standard Base64 or Base64 URL-safe
686     * alphabet, use {@link #isBase64Standard(byte)} or {@link #isBase64Url(byte)} methods respectively.
687     * </p>
688     *
689     * @param octet The value to test.
690     * @return {@code true} if the value is defined in the Base64 alphabet, {@code false} otherwise.
691     * @since 1.4
692     */
693    public static boolean isBase64(final byte octet) {
694        return octet == PAD_DEFAULT || octet >= 0 && octet < DECODE_TABLE.length && DECODE_TABLE[octet] != -1;
695    }
696
697    /**
698     * Tests a given byte array to see if it contains only valid characters within the Base64 alphabet. Currently the method treats whitespace as valid.
699     * <p>
700     * This method treats all characters included within standard base64 and base64url encodings as valid base64 characters. This includes the '+' and '/'
701     * (standard base64), as well as '-' and '_' (URL-safe base64) characters. To test membership in only the standard Base64 or Base64 URL-safe
702     * alphabet, use {@link #isBase64Standard(byte[])} or {@link #isBase64Url(byte[])} methods respectively.
703     * </p>
704     *
705     * <p>
706     * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits.
707     * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input.
708     * </p>
709     *
710     * @param arrayOctet byte array to test.
711     * @return {@code true} if all bytes are valid characters in the Base64 alphabet or if the byte array is empty; {@code false}, otherwise.
712     * @since 1.5
713     */
714    public static boolean isBase64(final byte[] arrayOctet) {
715        for (final byte element : arrayOctet) {
716            if (!isBase64(element) && !Character.isWhitespace(element)) {
717                return false;
718            }
719        }
720        return true;
721    }
722
723    /**
724     * Tests a given String to see if it contains only valid characters within the Base64 alphabet. Currently the method treats whitespace as valid.
725     * <p>
726     * This method treats all characters included within standard base64 and base64url encodings as valid base64 characters. This includes the '+' and '/'
727     * (standard base64), as well as '-' and '_' (URL-safe base64) characters. To test membership in only the standard Base64 or Base64 URL-safe
728     * alphabet, use {@link #isBase64Standard(String)} or {@link #isBase64Url(String)} methods respectively.
729     * </p>
730     *
731     * <p>
732     * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits.
733     * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input.
734     * </p>
735     *
736     * @param base64 String to test.
737     * @return {@code true} if all characters in the String are valid characters in the Base64 alphabet or if the String is empty; {@code false}, otherwise.
738     * @since 1.5
739     */
740    public static boolean isBase64(final String base64) {
741        return isBase64(StringUtils.getBytesUtf8(base64));
742    }
743
744    /**
745     * Tests whether or not the {@code octet} is in the standard Base64 alphabet.
746     * <p>
747     * This implementation is aligned with <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The
748     * Base64 Alphabet</a>.
749     * </p>
750     *
751     * @param octet The value to test.
752     * @return {@code true} if the value is defined in the standard Base64 alphabet, {@code false} otherwise.
753     * @since 1.21
754     */
755    public static boolean isBase64Standard(final byte octet) {
756        return octet == PAD_DEFAULT || octet >= 0 && octet < STANDARD_DECODE_TABLE.length && STANDARD_DECODE_TABLE[octet] != -1;
757    }
758
759    /**
760     * Tests a given byte array to see if it contains only valid characters within the standard Base64 alphabet. The method treats whitespace as valid.
761     * <p>
762     * This implementation is aligned with <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The
763     * Base64 Alphabet</a>.
764     * </p>
765     *
766     * <p>
767     * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits.
768     * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input.
769     * </p>
770     *
771     * @param arrayOctet byte array to test.
772     * @return {@code true} if all bytes are valid characters in the standard Base64 alphabet. {@code false}, otherwise.
773     * @since 1.21
774     */
775    public static boolean isBase64Standard(final byte[] arrayOctet) {
776        for (final byte element : arrayOctet) {
777            if (!isBase64Standard(element) && !Character.isWhitespace(element)) {
778                return false;
779            }
780        }
781        return true;
782    }
783
784    /**
785     * Tests a given String to see if it contains only valid characters within the standard Base64 alphabet. The method treats whitespace as valid.
786     * <p>
787     * This implementation is aligned with <a href="https://www.ietf.org/rfc/rfc2045#:~:text=Table%201%3A%20The%20Base64%20Alphabet">RFC 2045 Table 1: The
788     * Base64 Alphabet</a>.
789     * </p>
790     *
791     * <p>
792     * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits.
793     * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input.
794     * </p>
795     *
796     * @param base64 String to test.
797     * @return {@code true} if all characters in the String are valid characters in the standard Base64 alphabet or if the String is empty; {@code false},
798     *         otherwise.
799     * @since 1.21
800     */
801    public static boolean isBase64Standard(final String base64) {
802        return isBase64Standard(StringUtils.getBytesUtf8(base64));
803    }
804
805    /**
806     * Tests whether or not the {@code octet} is in the URL-safe Base64 alphabet.
807     * <p>
808     * This implementation is aligned with
809     * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648
810     * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>.
811     * </p>
812     *
813     * @param octet The value to test.
814     * @return {@code true} if the value is defined in the URL-safe Base64 alphabet, {@code false} otherwise.
815     * @since 1.21
816     */
817    public static boolean isBase64Url(final byte octet) {
818        return octet == PAD_DEFAULT || octet >= 0 && octet < URL_SAFE_DECODE_TABLE.length && URL_SAFE_DECODE_TABLE[octet] != -1;
819    }
820
821    /**
822     * Tests a given byte array to see if it contains only valid characters within the URL-safe Base64 alphabet. The method treats whitespace as valid.
823     * <p>
824     * This implementation is aligned with
825     * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648
826     * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>.
827     * </p>
828     *
829     * <p>
830     * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits.
831     * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input.
832     * </p>
833     *
834     * @param arrayOctet byte array to test.
835     * @return {@code true} if all bytes are valid characters in the URL-safe Base64 alphabet, {@code false}, otherwise.
836     * @since 1.21
837     */
838    public static boolean isBase64Url(final byte[] arrayOctet) {
839        for (final byte element : arrayOctet) {
840            if (!isBase64Url(element) && !Character.isWhitespace(element)) {
841                return false;
842            }
843        }
844        return true;
845    }
846
847    /**
848     * Tests a given String to see if it contains only valid characters within the URL-safe Base64 alphabet. The method treats whitespace as valid.
849     * <p>
850     * This implementation is aligned with
851     * <a href="https://datatracker.ietf.org/doc/html/rfc4648#:~:text=Table%202%3A%20The%20%22URL%20and%20Filename%20safe%22%20Base%2064%20Alphabet">RFC 4648
852     * Table 2: The "URL and Filename safe" Base 64 Alphabet</a>.
853     * </p>
854     *
855     * <p>
856     * This is a character-membership check, not canonical validation. It permits whitespace and padding in any position and does not check trailing bits.
857     * Use an instance configured with {@link CodecPolicy#STRICT} to require canonical input.
858     * </p>
859     *
860     * @param base64 String to test.
861     * @return {@code true} if all characters in the String are valid characters in the URL-safe Base64 alphabet or if the String is empty; {@code false},
862     *         otherwise.
863     * @since 1.21
864     */
865    public static boolean isBase64Url(final String base64) {
866        return isBase64Url(StringUtils.getBytesUtf8(base64));
867    }
868
869    private static byte[] toDecodeTable(final byte[] encodeTable) {
870        final byte[] table = encodeTable != null ? encodeTable : STANDARD_ENCODE_TABLE;
871        if (Arrays.equals(table, STANDARD_ENCODE_TABLE) || Arrays.equals(table, URL_SAFE_ENCODE_TABLE)) {
872            return DECODE_TABLE;
873        }
874        return calculateDecodeTable(table);
875    }
876
877    static byte[] toUrlSafeEncodeTable(final boolean urlSafe) {
878        return urlSafe ? URL_SAFE_ENCODE_TABLE : STANDARD_ENCODE_TABLE;
879    }
880
881    /**
882     * Line separator for encoding and strict decoding. Only used if lineLength &gt; 0.
883     */
884    private final byte[] lineSeparator;
885
886    /**
887     * Convenience variable to help us determine when our buffer is going to run out of room and needs resizing. {@code encodeSize = 4 + lineSeparator.length;}
888     */
889    private final int encodeSize;
890    private final boolean isUrlSafe;
891    private final boolean isStandardEncodeTable;
892
893    /**
894     * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode.
895     * <p>
896     * When encoding the line length is 0 (no chunking), and the encoding table is STANDARD_ENCODE_TABLE.
897     * </p>
898     * <p>
899     * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout.
900     * </p>
901     */
902    public Base64() {
903        this(0);
904    }
905
906    /**
907     * Constructs a Base64 codec used for decoding (all modes) and encoding in the given URL-safe mode.
908     * <p>
909     * When encoding the line length is 76, the line separator is CRLF, and the encoding table is STANDARD_ENCODE_TABLE.
910     * </p>
911     * <p>
912     * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout.
913     * </p>
914     *
915     * @param urlSafe if {@code true}, URL-safe encoding is used. In most cases this should be set to {@code false}.
916     * @since 1.4
917     * @deprecated Use {@link #builder()} and {@link Builder}.
918     */
919    @Deprecated
920    public Base64(final boolean urlSafe) {
921        this(MIME_CHUNK_SIZE, CHUNK_SEPARATOR, urlSafe);
922    }
923
924    private Base64(final Builder builder) {
925        super(builder);
926        final byte[] encTable = builder.getEncodeTable();
927        if (encTable.length != STANDARD_ENCODE_TABLE.length) {
928            throw new IllegalArgumentException("encodeTable must have exactly 64 entries.");
929        }
930        if (contains(encTable, pad)) {
931            throw new IllegalArgumentException("encodeTable must not contain the padding byte.");
932        }
933        this.isStandardEncodeTable = Arrays.equals(encTable, STANDARD_ENCODE_TABLE);
934        this.isUrlSafe = Arrays.equals(encTable, URL_SAFE_ENCODE_TABLE);
935        // TODO could be simplified if there is no requirement to reject invalid line sep when length <=0
936        // @see test case Base64Test.testConstructors()
937        if (builder.getLineSeparator().length > 0) {
938            final byte[] lineSeparatorB = builder.getLineSeparator();
939            if (containsAlphabetOrPad(lineSeparatorB)) {
940                final String sep = StringUtils.newStringUtf8(lineSeparatorB);
941                throw new IllegalArgumentException("lineSeparator must not contain base64 characters: [" + sep + "]");
942            }
943            if (builder.getLineLength() > 0) { // null line-sep forces no chunking rather than throwing IAE
944                this.encodeSize = BYTES_PER_ENCODED_BLOCK + lineSeparatorB.length;
945                this.lineSeparator = lineSeparatorB;
946            } else {
947                this.encodeSize = BYTES_PER_ENCODED_BLOCK;
948                this.lineSeparator = null;
949            }
950        } else {
951            this.encodeSize = BYTES_PER_ENCODED_BLOCK;
952            this.lineSeparator = null;
953        }
954    }
955
956    /**
957     * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode.
958     * <p>
959     * When encoding the line length is given in the constructor, the line separator is CRLF, and the encoding table is STANDARD_ENCODE_TABLE.
960     * </p>
961     * <p>
962     * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data.
963     * </p>
964     * <p>
965     * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout.
966     * </p>
967     *
968     * @param lineLength Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength &lt;= 0, then
969     *                   the output will not be divided into lines (chunks). Ignored when decoding leniently.
970     * @since 1.4
971     * @deprecated Use {@link #builder()} and {@link Builder}.
972     */
973    @Deprecated
974    public Base64(final int lineLength) {
975        this(lineLength, CHUNK_SEPARATOR);
976    }
977
978    /**
979     * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode.
980     * <p>
981     * When encoding the line length and line separator are given in the constructor, and the encoding table is STANDARD_ENCODE_TABLE.
982     * </p>
983     * <p>
984     * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data.
985     * </p>
986     * <p>
987     * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout.
988     * </p>
989     *
990     * @param lineLength    Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength &lt;= 0,
991     *                      then the output will not be divided into lines (chunks). Ignored when decoding leniently.
992     * @param lineSeparator Each line of encoded data will end with this sequence of bytes.
993     * @throws IllegalArgumentException Thrown when the provided lineSeparator included some base64 characters.
994     * @since 1.4
995     * @deprecated Use {@link #builder()} and {@link Builder}.
996     */
997    @Deprecated
998    public Base64(final int lineLength, final byte[] lineSeparator) {
999        this(lineLength, lineSeparator, false);
1000    }
1001
1002    /**
1003     * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode.
1004     * <p>
1005     * When encoding the line length and line separator are given in the constructor, and the encoding table is STANDARD_ENCODE_TABLE.
1006     * </p>
1007     * <p>
1008     * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data.
1009     * </p>
1010     * <p>
1011     * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout.
1012     * </p>
1013     *
1014     * @param lineLength    Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength &lt;= 0,
1015     *                      then the output will not be divided into lines (chunks). Ignored when decoding leniently.
1016     * @param lineSeparator Each line of encoded data will end with this sequence of bytes.
1017     * @param urlSafe       Instead of emitting '+' and '/' we emit '-' and '_' respectively. urlSafe is only applied to encode operations. Decoding seamlessly
1018     *                      handles both modes. <strong>No padding is added when using the URL-safe alphabet.</strong>
1019     * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base64 characters.
1020     * @since 1.4
1021     * @deprecated Use {@link #builder()} and {@link Builder}.
1022     */
1023    @Deprecated
1024    public Base64(final int lineLength, final byte[] lineSeparator, final boolean urlSafe) {
1025        this(builder().setLineLength(lineLength).setLineSeparator(lineSeparator != null ? lineSeparator : EMPTY_BYTE_ARRAY).setPadding(PAD_DEFAULT)
1026                .setEncodeTableRaw(toUrlSafeEncodeTable(urlSafe)).setDecodingPolicy(DECODING_POLICY_DEFAULT));
1027    }
1028
1029    /**
1030     * Constructs a Base64 codec used for decoding (all modes) and encoding in URL-unsafe mode.
1031     * <p>
1032     * When encoding the line length and line separator are given in the constructor, and the encoding table is STANDARD_ENCODE_TABLE.
1033     * </p>
1034     * <p>
1035     * Line lengths that aren't multiples of 4 will still essentially end up being multiples of 4 in the encoded data.
1036     * </p>
1037     * <p>
1038     * When decoding leniently all variants are supported. Strict decoding requires the configured encoding alphabet and layout.
1039     * </p>
1040     *
1041     * @param lineLength     Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 4). If lineLength &lt;= 0,
1042     *                       then the output will not be divided into lines (chunks). Ignored when decoding leniently.
1043     * @param lineSeparator  Each line of encoded data will end with this sequence of bytes.
1044     * @param urlSafe        Instead of emitting '+' and '/' we emit '-' and '_' respectively. Strict decoding requires this alphabet. Lenient
1045     *                       decoding handles both modes. <strong>No padding is added when using the URL-safe alphabet.</strong>
1046     * @param decodingPolicy The decoding policy.
1047     * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base64 characters.
1048     * @since 1.15
1049     * @deprecated Use {@link #builder()} and {@link Builder}.
1050     */
1051    @Deprecated
1052    public Base64(final int lineLength, final byte[] lineSeparator, final boolean urlSafe, final CodecPolicy decodingPolicy) {
1053        this(builder().setLineLength(lineLength).setLineSeparator(lineSeparator).setPadding(PAD_DEFAULT).setEncodeTableRaw(toUrlSafeEncodeTable(urlSafe))
1054                .setDecodingPolicy(decodingPolicy));
1055    }
1056
1057    /**
1058     * <p>
1059     * Decodes all of the provided data, starting at inPos, for inAvail bytes. Should be called at least twice: once with the data to decode, and once with
1060     * inAvail set to "-1" to alert decoder that EOF has been reached. Strict decoding requires the "-1" call to validate the complete input.
1061     * </p>
1062     * <p>
1063     * Lenient decoding ignores non-alphabet characters and stops at the first padding byte. Strict decoding accepts only the canonical form produced by this
1064     * instance's encoder, including its alphabet, padding, and line separators.
1065     * </p>
1066     * <p>
1067     * Thanks to "commons" project in ws.apache.org for the bitwise operations, and general approach.
1068     * https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/
1069     * </p>
1070     *
1071     * @param input   byte[] array of ASCII data to base64 decode.
1072     * @param inPos   Position to start reading data from.
1073     * @param inAvail Amount of bytes available from input for decoding.
1074     * @param context The context to be used.
1075     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
1076     */
1077    @Override
1078    void decode(final byte[] input, int inPos, final int inAvail, final Context context) {
1079        if (context.eof) {
1080            return;
1081        }
1082        if (inAvail < 0) {
1083            context.eof = true;
1084            if (isStrictDecoding()) {
1085                validateCanonicalEnd(isStandardEncodeTable, context);
1086            }
1087        }
1088        final int decodeSize = this.encodeSize - 1;
1089        for (int i = 0; i < inAvail; i++) {
1090            final int b = input[inPos++] & 0xff;
1091            if (isStrictDecoding()) {
1092                if (!validateCanonicalByte(b, lineSeparator, isStandardEncodeTable, context)) {
1093                    continue;
1094                }
1095            } else if (b == (pad & 0xff)) {
1096                // We're done.
1097                context.eof = true;
1098                break;
1099            }
1100            final byte[] buffer = ensureBufferSize(decodeSize, context);
1101            if (b < decodeTable.length) {
1102                final int result = decodeTable[b];
1103                if (result >= 0) {
1104                    context.modulus = (context.modulus + 1) % BYTES_PER_ENCODED_BLOCK;
1105                    context.ibitWorkArea = (context.ibitWorkArea << BITS_PER_ENCODED_BYTE) + result;
1106                    if (context.modulus == 0) {
1107                        buffer[context.pos++] = (byte) (context.ibitWorkArea >> 16 & MASK_8BITS);
1108                        buffer[context.pos++] = (byte) (context.ibitWorkArea >> 8 & MASK_8BITS);
1109                        buffer[context.pos++] = (byte) (context.ibitWorkArea & MASK_8BITS);
1110                    }
1111                }
1112            }
1113        }
1114
1115        // Strict decoding waits for physical EOF to validate the complete input.
1116        // Lenient decoding also treats the first padding byte as EOF.
1117        if (context.eof && context.modulus != 0) {
1118            final byte[] buffer = ensureBufferSize(decodeSize, context);
1119
1120            // We have some spare bits remaining
1121            // Output all whole multiples of 8 bits and ignore the rest
1122            switch (context.modulus) {
1123//              case 0 : // impossible, as excluded above
1124                case 1 : // 6 bits - either ignore entirely, or raise an exception
1125                    validateTrailingCharacter();
1126                    break;
1127                case 2 : // 12 bits = 8 + 4
1128                    validateCharacter(MASK_4_BITS, context);
1129                    context.ibitWorkArea = context.ibitWorkArea >> 4; // dump the extra 4 bits
1130                    buffer[context.pos++] = (byte) (context.ibitWorkArea & MASK_8BITS);
1131                    break;
1132                case 3 : // 18 bits = 8 + 8 + 2
1133                    validateCharacter(MASK_2_BITS, context);
1134                    context.ibitWorkArea = context.ibitWorkArea >> 2; // dump 2 bits
1135                    buffer[context.pos++] = (byte) (context.ibitWorkArea >> 8 & MASK_8BITS);
1136                    buffer[context.pos++] = (byte) (context.ibitWorkArea & MASK_8BITS);
1137                    break;
1138                default:
1139                    throw new IllegalStateException("Impossible modulus " + context.modulus);
1140            }
1141        }
1142    }
1143
1144    /**
1145     * <p>
1146     * Encodes all of the provided data, starting at inPos, for inAvail bytes. Must be called at least twice: once with the data to encode, and once with
1147     * inAvail set to "-1" to alert encoder that EOF has been reached, to flush last remaining bytes (if not multiple of 3).
1148     * </p>
1149     * <p>
1150     * <strong>No padding is added when encoding using the URL-safe alphabet.</strong>
1151     * </p>
1152     * <p>
1153     * Thanks to "commons" project in ws.apache.org for the bitwise operations, and general approach.
1154     * https://svn.apache.org/repos/asf/webservices/commons/trunk/modules/util/
1155     * </p>
1156     *
1157     * @param in      byte[] array of binary data to base64 encode.
1158     * @param inPos   Position to start reading data from.
1159     * @param inAvail Amount of bytes available from input for encoding.
1160     * @param context The context to be used.
1161     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
1162     */
1163    @Override
1164    void encode(final byte[] in, int inPos, final int inAvail, final Context context) {
1165        if (context.eof) {
1166            return;
1167        }
1168        // inAvail < 0 is how we're informed of EOF in the underlying data we're
1169        // encoding.
1170        if (inAvail < 0) {
1171            context.eof = true;
1172            if (0 == context.modulus && lineLength == 0) {
1173                return; // no leftovers to process and not using chunking
1174            }
1175            final byte[] buffer = ensureBufferSize(encodeSize, context);
1176            final int savedPos = context.pos;
1177            switch (context.modulus) { // 0-2
1178                case 0 : // nothing to do here
1179                    break;
1180                case 1 : // 8 bits = 6 + 2
1181                    // top 6 bits:
1182                    buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 2 & MASK_6_BITS];
1183                    // remaining 2:
1184                    buffer[context.pos++] = encodeTable[context.ibitWorkArea << 4 & MASK_6_BITS];
1185                    // URL-SAFE skips the padding to further reduce size.
1186                    if (isStandardEncodeTable) {
1187                        buffer[context.pos++] = pad;
1188                        buffer[context.pos++] = pad;
1189                    }
1190                    break;
1191
1192                case 2 : // 16 bits = 6 + 6 + 4
1193                    buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 10 & MASK_6_BITS];
1194                    buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 4 & MASK_6_BITS];
1195                    buffer[context.pos++] = encodeTable[context.ibitWorkArea << 2 & MASK_6_BITS];
1196                    // URL-SAFE skips the padding to further reduce size.
1197                    if (isStandardEncodeTable) {
1198                        buffer[context.pos++] = pad;
1199                    }
1200                    break;
1201                default:
1202                    throw new IllegalStateException("Impossible modulus " + context.modulus);
1203            }
1204            context.currentLinePos += context.pos - savedPos; // keep track of current line position
1205            // if currentPos == 0 we are at the start of a line, so don't add CRLF
1206            if (lineLength > 0 && context.currentLinePos > 0) {
1207                System.arraycopy(lineSeparator, 0, buffer, context.pos, lineSeparator.length);
1208                context.pos += lineSeparator.length;
1209            }
1210        } else {
1211            for (int i = 0; i < inAvail; i++) {
1212                final byte[] buffer = ensureBufferSize(encodeSize, context);
1213                context.modulus = (context.modulus + 1) % BYTES_PER_UNENCODED_BLOCK;
1214                int b = in[inPos++];
1215                if (b < 0) {
1216                    b += 256;
1217                }
1218                context.ibitWorkArea = (context.ibitWorkArea << 8) + b; // BITS_PER_BYTE
1219                if (0 == context.modulus) { // 3 bytes = 24 bits = 4 * 6 bits to extract
1220                    buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 18 & MASK_6_BITS];
1221                    buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 12 & MASK_6_BITS];
1222                    buffer[context.pos++] = encodeTable[context.ibitWorkArea >> 6 & MASK_6_BITS];
1223                    buffer[context.pos++] = encodeTable[context.ibitWorkArea & MASK_6_BITS];
1224                    context.currentLinePos += BYTES_PER_ENCODED_BLOCK;
1225                    if (lineLength > 0 && lineLength <= context.currentLinePos) {
1226                        System.arraycopy(lineSeparator, 0, buffer, context.pos, lineSeparator.length);
1227                        context.pos += lineSeparator.length;
1228                        context.currentLinePos = 0;
1229                    }
1230                }
1231            }
1232        }
1233    }
1234
1235    /**
1236     * Gets the line separator (for testing only).
1237     *
1238     * @return The line separator.
1239     */
1240    byte[] getLineSeparator() {
1241        return lineSeparator;
1242    }
1243
1244    /**
1245     * Tests whether the {@code octet} is in the Base64 alphabet.
1246     *
1247     * @param octet The value to test.
1248     * @return {@code true} if the value is defined in the Base64 alphabet {@code false} otherwise.
1249     */
1250    @Override
1251    protected boolean isInAlphabet(final byte octet) {
1252        final int value = octet & 0xff;
1253        return value < decodeTable.length && decodeTable[value] != -1;
1254    }
1255
1256    /**
1257     * Tests whether the current encoding mode is URL-safe.
1258     *
1259     * @return true if we're in URL-safe mode, false otherwise.
1260     * @since 1.4
1261     */
1262    public boolean isUrlSafe() {
1263        return isUrlSafe;
1264    }
1265
1266    /**
1267     * Validates whether decoding the final trailing character is possible in the context of the set of possible Base64 values.
1268     * <p>
1269     * The character is valid if the lower bits within the provided mask are zero. This is used to test the final trailing base-64 digit is zero in the bits
1270     * that will be discarded.
1271     * </p>
1272     *
1273     * @param emptyBitsMask The mask of the lower bits that should be empty.
1274     * @param context       The context to be used.
1275     * @throws IllegalArgumentException Thrown if the bits being checked contain any non-zero value.
1276     */
1277    private void validateCharacter(final int emptyBitsMask, final Context context) {
1278        if (isStrictDecoding() && (context.ibitWorkArea & emptyBitsMask) != 0) {
1279            throw new IllegalArgumentException("Strict decoding: Last encoded character (before the paddings if any) is a valid " +
1280                    "Base64 alphabet but not a possible encoding. Expected the discarded bits from the character to be zero.");
1281        }
1282    }
1283
1284    /**
1285     * Validates whether decoding allows an entire final trailing character that cannot be used for a complete byte.
1286     *
1287     * @throws IllegalArgumentException Thrown if strict decoding is enabled.
1288     */
1289    private void validateTrailingCharacter() {
1290        if (isStrictDecoding()) {
1291            throw new IllegalArgumentException("Strict decoding: Last encoded character (before the paddings if any) is a valid " +
1292                    "Base64 alphabet but not a possible encoding. Decoding requires at least two trailing 6-bit characters to create bytes.");
1293        }
1294    }
1295
1296}