001/*
002 * Licensed to the Apache Software Foundation (ASF) under one or more
003 * contributor license agreements.  See the NOTICE file distributed with
004 * this work for additional information regarding copyright ownership.
005 * The ASF licenses this file to You under the Apache License, Version 2.0
006 * (the "License"); you may not use this file except in compliance with
007 * the License.  You may obtain a copy of the License at
008 *
009 *      https://www.apache.org/licenses/LICENSE-2.0
010 *
011 * Unless required by applicable law or agreed to in writing, software
012 * distributed under the License is distributed on an "AS IS" BASIS,
013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
014 * See the License for the specific language governing permissions and
015 * limitations under the License.
016 */
017
018package org.apache.commons.codec.binary;
019
020import java.util.Arrays;
021
022import org.apache.commons.codec.CodecPolicy;
023
024/**
025 * Provides Base32 encoding and decoding as defined by <a href="https://www.ietf.org/rfc/rfc4648.txt">RFC 4648</a>.
026 *
027 * <p>
028 * The class can be parameterized in the following manner with various constructors:
029 * </p>
030 * <ul>
031 * <li>Whether to use the "base32hex" variant instead of the default "base32"</li>
032 * <li>Line length: Default 76. Line length that aren't multiples of 8 will still essentially end up being multiples of 8 in the encoded data.
033 * <li>Line separator: Default is CRLF ("\r\n")</li>
034 * </ul>
035 * <p>
036 * This class operates directly on byte streams, and not character streams.
037 * </p>
038 * <p>
039 * This class is thread-safe.
040 * </p>
041 * <p>
042 * To configure a new instance, use a {@link Builder}. For example:
043 * </p>
044 * <pre>
045 * Base32 base32 = Base32.builder()
046 *   .setDecodingPolicy(DecodingPolicy.LENIENT) // default is lenient
047 *   .setLineLength(0)                          // default is none
048 *   .setLineSeparator('\r', '\n')              // default is CR LF
049 *   .setPadding('=')                           // default is '='
050 *   .setEncodeTable(customEncodeTable)         // default is RFC 4648 Section 6, Table 3: The Base 32 Alphabet
051 *   .get()
052 * </pre>
053 *
054 * @see Base32InputStream
055 * @see Base32OutputStream
056 * @see <a href="https://www.ietf.org/rfc/rfc4648.txt">RFC 4648</a>
057 * @since 1.5
058 */
059public class Base32 extends BaseNCodec {
060
061    /**
062     * Builds {@link Base32} instances.
063     *
064     * <p>
065     * To configure a new instance, use a {@link Builder}. For example:
066     * </p>
067     *
068     * <pre>
069     * Base32 base32 = Base32.builder()
070     *   .setDecodingPolicy(DecodingPolicy.LENIENT) // default is lenient
071     *   .setLineLength(0)                          // default is none
072     *   .setLineSeparator('\r', '\n')              // default is CR LF
073     *   .setPadding('=')                           // default is '='
074     *   .setEncodeTable(customEncodeTable)         // default is RFC 4648 Section 6, Table 3: The Base 32 Alphabet
075     *   .get()
076     * </pre>
077     *
078     * @since 1.17.0
079     */
080    public static class Builder extends AbstractBuilder<Base32, Builder> {
081
082        /**
083         * Constructs a new instance using <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
084         * Alphabet</a>.
085         */
086        public Builder() {
087            super(ENCODE_TABLE);
088            setDecodeTableRaw(DECODE_TABLE);
089            setEncodeTableRaw(ENCODE_TABLE);
090            setEncodedBlockSize(BYTES_PER_ENCODED_BLOCK);
091            setUnencodedBlockSize(BYTES_PER_UNENCODED_BLOCK);
092        }
093
094        @Override
095        public Base32 get() {
096            return new Base32(this);
097        }
098
099        /**
100         * Sets the encode table and derives the matching decode table.
101         * <p>
102         * The RFC 4648 Base32 and Base32 Hex tables keep their case-insensitive decoders in lenient mode. Strict decoding requires the encoding alphabet.
103         * </p>
104         *
105         * @param encodeTable The encode table with exactly 32 unique entries, null resets to the default.
106         * @return {@code this} instance.
107         * @throws IllegalArgumentException Thrown if the encode table does not contain exactly 32 unique entries.
108         */
109        @Override
110        public Builder setEncodeTable(final byte... encodeTable) {
111            super.setDecodeTableRaw(toDecodeTable(encodeTable));
112            return super.setEncodeTable(encodeTable);
113        }
114
115        /**
116         * Sets the encode and decode tables to use Base32 hexadecimal if {@code true}, otherwise use the Base32 alphabet.
117         * <p>
118         * This overrides a value previously set with {@link #setEncodeTable(byte...)}.
119         * </p>
120         *
121         * @param useHex use Base32 hexadecimal if {@code true}, otherwise use the Base32 alphabet.
122         * @return {@code this} instance.
123         * @since 1.18.0
124         */
125        public Builder setHexDecodeTable(final boolean useHex) {
126            return setEncodeTable(encodeTable(useHex));
127        }
128
129        /**
130         * Sets the encode table to use Base32 hexadecimal if {@code true}, otherwise use the Base32 alphabet.
131         * <p>
132         * This overrides a value previously set with {@link #setEncodeTable(byte...)}.
133         * </p>
134         *
135         * @param useHex
136         *               <ul>
137         *               <li>If true, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding
138         *               with Extended Hex Alphabet</a></li>
139         *               <li>If false, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
140         *               Alphabet</a></li>
141         *               </ul>
142         * @return {@code this} instance.
143         * @since 1.18.0
144         */
145        public Builder setHexEncodeTable(final boolean useHex) {
146            return setEncodeTable(encodeTable(useHex));
147        }
148    }
149
150    /**
151     * BASE32 characters are 5 bits in length. They are formed by taking a block of five octets to form a 40-bit string, which is converted into eight BASE32
152     * characters.
153     */
154    private static final int BITS_PER_ENCODED_BYTE = 5;
155
156    private static final int BYTES_PER_ENCODED_BLOCK = 8;
157    private static final int BYTES_PER_UNENCODED_BLOCK = 5;
158    private static final int DECODING_TABLE_LENGTH = 256;
159    private static final int ENCODING_TABLE_LENGTH = 1 << BITS_PER_ENCODED_BYTE;
160
161    /**
162     * This array is a lookup table that translates Unicode characters drawn from the "Base32 Alphabet" (as specified in Table 3 of RFC 4648) into their 5-bit
163     * positive integer equivalents. Characters that are not in the Base32 alphabet but fall within the bounds of the array are translated to -1.
164     */
165    // @formatter:off
166    private static final byte[] DECODE_TABLE = {
167         //  0   1   2   3   4   5   6   7   8   9   A   B   C   D   E   F
168            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f
169            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f
170            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 20-2f
171            -1, -1, 26, 27, 28, 29, 30, 31, -1, -1, -1, -1, -1, -1, -1, -1, // 30-3f 2-7
172            -1,  0,  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, // 40-4f A-O
173            15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25,                     // 50-5a P-Z
174                                                        -1, -1, -1, -1, -1, // 5b-5f
175            -1,  0,  1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, // 60-6f a-o
176            15, 16, 17, 18, 19, 20, 21, 22, 23, 24, 25,                     // 70-7a p-z
177    };
178    // @formatter:on
179
180    /**
181     * This array is a lookup table that translates 5-bit positive integer index values into their "Base32 Alphabet" equivalents as specified in
182     * <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32 Alphabet</a>.
183     *
184     * @see <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32 Alphabet</a>
185     */
186    // @formatter:off
187    private static final byte[] ENCODE_TABLE = {
188            'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
189            'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
190            '2', '3', '4', '5', '6', '7',
191    };
192    // @formatter:on
193
194    /**
195     * This array is a lookup table that translates Unicode characters drawn from the "Base32 Hex Alphabet" (as specified in Table 4 of RFC 4648) into their
196     * 5-bit positive integer equivalents. Characters that are not in the Base32 Hex alphabet but fall within the bounds of the array are translated to -1.
197     */
198    // @formatter:off
199    private static final byte[] HEX_DECODE_TABLE = {
200         //  0   1   2   3   4   5   6   7   8   9   A   B   C   D   E   F
201            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f
202            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f
203            -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 20-2f
204             0,  1,  2,  3,  4,  5,  6,  7,  8,  9, -1, -1, -1, -1, -1, -1, // 30-3f 0-9
205            -1, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, // 40-4f A-O
206            25, 26, 27, 28, 29, 30, 31,                                     // 50-56 P-V
207                                        -1, -1, -1, -1, -1, -1, -1, -1, -1, // 57-5f
208            -1, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, // 60-6f a-o
209            25, 26, 27, 28, 29, 30, 31                                      // 70-76 p-v
210    };
211    // @formatter:on
212
213    /**
214     * This array is a lookup table that translates 5-bit positive integer index values into their "Base 32 Encoding with Extended Hex Alphabet" equivalents as
215     * specified in <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with Extended Hex
216     * Alphabet</a>.
217     *
218     * @see <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with Extended Hex Alphabet</a>
219     */
220    // @formatter:off
221    private static final byte[] HEX_ENCODE_TABLE = {
222            '0', '1', '2', '3', '4', '5', '6', '7', '8', '9',
223            'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
224            'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V',
225    };
226    // @formatter:on
227
228    /** Mask used to extract 5 bits, used when encoding Base32 bytes */
229    private static final int MASK_5_BITS = 0x1f;
230
231    /** Mask used to extract 4 bits, used when decoding final trailing character. */
232    private static final long MASK_4_BITS = 0x0fL;
233
234    /** Mask used to extract 3 bits, used when decoding final trailing character. */
235    private static final long MASK_3_BITS = 0x07L;
236
237    /** Mask used to extract 2 bits, used when decoding final trailing character. */
238    private static final long MASK_2_BITS = 0x03L;
239
240    /** Mask used to extract 1 bits, used when decoding final trailing character. */
241    private static final long MASK_1_BITS = 0x01L;
242
243    // The static final fields above are used for the original static byte[] methods on Base32.
244    // The private member fields below are used with the new streaming approach, which requires
245    // some state be preserved between calls of encode() and decode().
246
247    /**
248     * Creates a new Builder.
249     *
250     * <p>
251     * To configure a new instance, use a {@link Builder}. For example:
252     * </p>
253     *
254     * <pre>
255     * Base32 base32 = Base32.builder()
256     *   .setDecodingPolicy(DecodingPolicy.LENIENT) // default is lenient
257     *   .setLineLength(0)                          // default is none
258     *   .setLineSeparator('\r', '\n')              // default is CR LF
259     *   .setPadding('=')                           // default is '='
260     *   .setEncodeTable(customEncodeTable)         // default is RFC 4648 Section 6, Table 3: The Base 32 Alphabet
261     *   .get()
262     * </pre>
263     *
264     * @return A new Builder.
265     * @since 1.17.0
266     */
267    public static Builder builder() {
268        return new Builder();
269    }
270
271    /**
272     * Calculates a decode table for a given encode table.
273     *
274     * @param encodeTable that is used to determine decode lookup table.
275     * @return A new decode table.
276     * @throws IllegalArgumentException Thrown if the encode table does not contain exactly 32 unique entries.
277     */
278    private static byte[] calculateDecodeTable(final byte[] encodeTable) {
279        if (encodeTable.length != ENCODING_TABLE_LENGTH) {
280            throw new IllegalArgumentException("encodeTable must have exactly 32 entries.");
281        }
282        final byte[] decodeTable = new byte[DECODING_TABLE_LENGTH];
283        Arrays.fill(decodeTable, (byte) -1);
284        for (int i = 0; i < encodeTable.length; i++) {
285            final int encodedByte = encodeTable[i] & 0xff;
286            if (decodeTable[encodedByte] != -1) {
287                throw new IllegalArgumentException("encodeTable must not contain duplicate entries.");
288            }
289            decodeTable[encodedByte] = (byte) i;
290        }
291        return decodeTable;
292    }
293
294    private static byte[] decodeTable(final boolean useHex) {
295        return useHex ? HEX_DECODE_TABLE : DECODE_TABLE;
296    }
297
298    /**
299     * Gets the encoding table that matches {@code useHex}.
300     *
301     * @param useHex
302     *               <ul>
303     *               <li>If true, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with
304     *               Extended Hex Alphabet</a></li>
305     *               <li>If false, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
306     *               Alphabet</a></li>
307     *               </ul>
308     * @return The encoding table that matches {@code useHex}.
309     */
310    private static byte[] encodeTable(final boolean useHex) {
311        return useHex ? HEX_ENCODE_TABLE : ENCODE_TABLE;
312    }
313
314    /**
315     * Gets the decode table that matches the given encode table.
316     *
317     * @param encodeTable that is used to determine decode lookup table.
318     * @return The matching decode table.
319     */
320    private static byte[] toDecodeTable(final byte[] encodeTable) {
321        final byte[] table = encodeTable != null ? encodeTable : ENCODE_TABLE;
322        if (Arrays.equals(table, ENCODE_TABLE)) {
323            return DECODE_TABLE;
324        }
325        if (Arrays.equals(table, HEX_ENCODE_TABLE)) {
326            return HEX_DECODE_TABLE;
327        }
328        return calculateDecodeTable(table);
329    }
330
331    /**
332     * Convenience variable to help us determine when our buffer is going to run out of room and needs resizing. {@code encodeSize = {@link
333     * #BYTES_PER_ENCODED_BLOCK} + lineSeparator.length;}
334     */
335    private final int encodeSize;
336
337    /**
338     * Line separator for encoding and strict decoding. Only used if lineLength &gt; 0.
339     */
340    private final byte[] lineSeparator;
341
342    /**
343     * Constructs a Base32 codec used for decoding and encoding.
344     * <p>
345     * When encoding the line length is 0 (no chunking).
346     * </p>
347     */
348    public Base32() {
349        this(false);
350    }
351
352    /**
353     * Constructs a Base32 codec used for decoding and encoding.
354     * <p>
355     * When encoding the line length is 0 (no chunking).
356     * </p>
357     *
358     * @param useHex
359     *               <ul>
360     *               <li>If true, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with
361     *               Extended Hex Alphabet</a></li>
362     *               <li>If false, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
363     *               Alphabet</a></li>
364     *               </ul>
365     * @deprecated Use {@link #builder()} and {@link Builder}.
366     */
367    @Deprecated
368    public Base32(final boolean useHex) {
369        this(0, null, useHex, PAD_DEFAULT);
370    }
371
372    /**
373     * Constructs a Base32 codec used for decoding and encoding.
374     * <p>
375     * When encoding the line length is 0 (no chunking).
376     * </p>
377     *
378     * @param useHex
379     *               <ul>
380     *               <li>If true, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with
381     *               Extended Hex Alphabet</a></li>
382     *               <li>If false, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
383     *               Alphabet</a></li>
384     *               </ul>
385     * @param padding byte used as padding byte.
386     * @deprecated Use {@link #builder()} and {@link Builder}.
387     */
388    @Deprecated
389    public Base32(final boolean useHex, final byte padding) {
390        this(0, null, useHex, padding);
391    }
392
393    private Base32(final Builder builder) {
394        super(builder);
395        if (builder.getLineLength() > 0) {
396            final byte[] lineSeparator = builder.getLineSeparator();
397            // Must be done after initializing the tables
398            if (containsAlphabetOrPad(lineSeparator)) {
399                final String sep = StringUtils.newStringUtf8(lineSeparator);
400                throw new IllegalArgumentException("lineSeparator must not contain Base32 characters: [" + sep + "]");
401            }
402            this.encodeSize = BYTES_PER_ENCODED_BLOCK + lineSeparator.length;
403            this.lineSeparator = lineSeparator;
404        } else {
405            this.encodeSize = BYTES_PER_ENCODED_BLOCK;
406            this.lineSeparator = null;
407        }
408        if (isInAlphabet(builder.getPadding()) || Character.isWhitespace(builder.getPadding())) {
409            throw new IllegalArgumentException("pad must not be in alphabet or whitespace");
410        }
411    }
412
413    /**
414     * Constructs a Base32 codec used for decoding and encoding.
415     * <p>
416     * When encoding the line length is 0 (no chunking).
417     * </p>
418     *
419     * @param pad byte used as padding byte.
420     * @deprecated Use {@link #builder()} and {@link Builder}.
421     */
422    @Deprecated
423    public Base32(final byte pad) {
424        this(false, pad);
425    }
426
427    /**
428     * Constructs a Base32 codec used for decoding and encoding.
429     * <p>
430     * When encoding the line length is given in the constructor, the line separator is CRLF.
431     * </p>
432     *
433     * @param lineLength Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 8). If lineLength &lt;= 0, then
434     *                   the output will not be divided into lines (chunks). Ignored when decoding leniently.
435     * @deprecated Use {@link #builder()} and {@link Builder}.
436     */
437    @Deprecated
438    public Base32(final int lineLength) {
439        this(lineLength, CHUNK_SEPARATOR);
440    }
441
442    /**
443     * Constructs a Base32 codec used for decoding and encoding.
444     * <p>
445     * When encoding the line length and line separator are given in the constructor.
446     * </p>
447     * <p>
448     * Line lengths that aren't multiples of 8 will still essentially end up being multiples of 8 in the encoded data.
449     * </p>
450     *
451     * @param lineLength    Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 8). If lineLength &lt;= 0,
452     *                      then the output will not be divided into lines (chunks). Ignored when decoding leniently.
453     * @param lineSeparator Each line of encoded data will end with this sequence of bytes.
454     * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base32 characters.
455     * @deprecated Use {@link #builder()} and {@link Builder}.
456     */
457    @Deprecated
458    public Base32(final int lineLength, final byte[] lineSeparator) {
459        this(lineLength, lineSeparator, false, PAD_DEFAULT);
460    }
461
462    /**
463     * Constructs a Base32 / Base32 Hex codec used for decoding and encoding.
464     * <p>
465     * When encoding the line length and line separator are given in the constructor.
466     * </p>
467     * <p>
468     * Line lengths that aren't multiples of 8 will still essentially end up being multiples of 8 in the encoded data.
469     * </p>
470     *
471     * @param lineLength    Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 8). If lineLength &lt;= 0,
472     *                      then the output will not be divided into lines (chunks). Ignored when decoding leniently.
473     * @param lineSeparator Each line of encoded data will end with this sequence of bytes.
474     * @param useHex
475     *               <ul>
476     *               <li>If true, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with
477     *               Extended Hex Alphabet</a></li>
478     *               <li>If false, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
479     *               Alphabet</a></li>
480     *               </ul>
481     * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base32 characters. Or the lineLength &gt; 0 and lineSeparator is null.
482     * @deprecated Use {@link #builder()} and {@link Builder}.
483     */
484    @Deprecated
485    public Base32(final int lineLength, final byte[] lineSeparator, final boolean useHex) {
486        this(lineLength, lineSeparator, useHex, PAD_DEFAULT);
487    }
488
489    /**
490     * Constructs a Base32 / Base32 Hex codec used for decoding and encoding.
491     * <p>
492     * When encoding the line length and line separator are given in the constructor.
493     * </p>
494     * <p>
495     * Line lengths that aren't multiples of 8 will still essentially end up being multiples of 8 in the encoded data.
496     * </p>
497     *
498     * @param lineLength    Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 8). If lineLength &lt;= 0,
499     *                      then the output will not be divided into lines (chunks). Ignored when decoding leniently.
500     * @param lineSeparator Each line of encoded data will end with this sequence of bytes.
501     * @param useHex
502     *               <ul>
503     *               <li>If true, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with
504     *               Extended Hex Alphabet</a></li>
505     *               <li>If false, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
506     *               Alphabet</a></li>
507     *               </ul>
508     * @param padding       padding byte.
509     * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base32 characters. Or the lineLength &gt; 0 and lineSeparator is null.
510     * @deprecated Use {@link #builder()} and {@link Builder}.
511     */
512    @Deprecated
513    public Base32(final int lineLength, final byte[] lineSeparator, final boolean useHex, final byte padding) {
514        this(lineLength, lineSeparator, useHex, padding, DECODING_POLICY_DEFAULT);
515    }
516
517    /**
518     * Constructs a Base32 / Base32 Hex codec used for decoding and encoding.
519     * <p>
520     * When encoding the line length and line separator are given in the constructor.
521     * </p>
522     * <p>
523     * Line lengths that aren't multiples of 8 will still essentially end up being multiples of 8 in the encoded data.
524     * </p>
525     *
526     * @param lineLength     Each line of encoded data will be at most of the given length (rounded down to the nearest multiple of 8). If lineLength &lt;= 0,
527     *                       then the output will not be divided into lines (chunks). Ignored when decoding leniently.
528     * @param lineSeparator  Each line of encoded data will end with this sequence of bytes.
529     * @param useHex
530     *               <ul>
531     *               <li>If true, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-7">RFC 4648 Section 7, Table 4: Base 32 Encoding with
532     *               Extended Hex Alphabet</a></li>
533     *               <li>If false, then use <a href="https://datatracker.ietf.org/doc/html/rfc4648#section-6">RFC 4648 Section 6, Table 3: The Base 32
534     *               Alphabet</a></li>
535     *               </ul>
536     * @param padding        padding byte.
537     * @param decodingPolicy The decoding policy.
538     * @throws IllegalArgumentException Thrown when the {@code lineSeparator} contains Base32 characters. Or the lineLength &gt; 0 and lineSeparator is null.
539     * @since 1.15
540     * @deprecated Use {@link #builder()} and {@link Builder}.
541     */
542    @Deprecated
543    public Base32(final int lineLength, final byte[] lineSeparator, final boolean useHex, final byte padding, final CodecPolicy decodingPolicy) {
544        // @formatter:off
545        this(builder()
546                .setLineLength(lineLength)
547                .setLineSeparator(lineSeparator != null ? lineSeparator : EMPTY_BYTE_ARRAY)
548                .setDecodeTable(decodeTable(useHex))
549                .setEncodeTableRaw(encodeTable(useHex))
550                .setPadding(padding)
551                .setDecodingPolicy(decodingPolicy));
552        // @formatter:on
553    }
554
555    /**
556     * <p>
557     * Decodes all of the provided data, starting at inPos, for inAvail bytes. Should be called at least twice: once with the data to decode, and once with
558     * inAvail set to "-1" to alert decoder that EOF has been reached. Strict decoding requires the "-1" call to validate the complete input.
559     * </p>
560     * <p>
561     * Lenient decoding ignores non-alphabet characters and stops at the first padding byte. Strict decoding accepts only the canonical form produced by this
562     * instance's encoder, including its alphabet, padding, and line separators.
563     * </p>
564     * <p>
565     * Output is written to {@link BaseNCodec.Context#buffer Context#buffer} as 8-bit octets, using
566     * {@link BaseNCodec.Context#pos Context#pos} as the buffer position
567     * </p>
568     *
569     * @param input   byte[] array of ASCII data to Base32 decode.
570     * @param inPos   Position to start reading data from.
571     * @param inAvail Amount of bytes available from input for decoding.
572     * @param context The context to be used.
573     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
574     */
575    @Override
576    void decode(final byte[] input, int inPos, final int inAvail, final Context context) {
577        // package protected for access from I/O streams
578        if (context.eof) {
579            return;
580        }
581        if (inAvail < 0) {
582            context.eof = true;
583            if (isStrictDecoding()) {
584                validateCanonicalEnd(true, context);
585            }
586        }
587        final int decodeSize = this.encodeSize - 1;
588        for (int i = 0; i < inAvail; i++) {
589            final int b = input[inPos++] & 0xff;
590            if (isStrictDecoding()) {
591                if (!validateCanonicalByte(b, lineSeparator, true, context)) {
592                    continue;
593                }
594            } else if (b == (pad & 0xff)) {
595                // We're done.
596                context.eof = true;
597                break;
598            }
599            final byte[] buffer = ensureBufferSize(decodeSize, context);
600            if (b < this.decodeTable.length) {
601                final int result = this.decodeTable[b];
602                if (result >= 0) {
603                    context.modulus = (context.modulus + 1) % BYTES_PER_ENCODED_BLOCK;
604                    // collect decoded bytes
605                    context.lbitWorkArea = (context.lbitWorkArea << BITS_PER_ENCODED_BYTE) + result;
606                    if (context.modulus == 0) { // we can output the 5 bytes
607                        buffer[context.pos++] = (byte) (context.lbitWorkArea >> 32 & MASK_8BITS);
608                        buffer[context.pos++] = (byte) (context.lbitWorkArea >> 24 & MASK_8BITS);
609                        buffer[context.pos++] = (byte) (context.lbitWorkArea >> 16 & MASK_8BITS);
610                        buffer[context.pos++] = (byte) (context.lbitWorkArea >> 8 & MASK_8BITS);
611                        buffer[context.pos++] = (byte) (context.lbitWorkArea & MASK_8BITS);
612                    }
613                }
614            }
615        }
616        // Strict decoding waits for physical EOF to validate the complete input.
617        // Lenient decoding also treats the first padding byte as EOF.
618        if (context.eof && context.modulus > 0) { // if modulus == 0, nothing to do
619            final byte[] buffer = ensureBufferSize(decodeSize, context);
620            // We ignore partial bytes, i.e. only multiples of 8 count.
621            // Any combination not part of a valid encoding is either partially decoded
622            // or will raise an exception. Possible trailing characters are 2, 4, 5, 7.
623            // It is not possible to encode with 1, 3, 6 trailing characters.
624            // For backwards compatibility 3 & 6 chars are decoded anyway rather than discarded.
625            // See the encode(byte[]) method EOF section.
626            switch (context.modulus) {
627//              case 0 : // impossible, as excluded above
628            case 1: // 5 bits - either ignore entirely, or raise an exception
629                validateTrailingCharacters();
630                // falls-through
631            case 2: // 10 bits, drop 2 and output one byte
632                validateCharacter(MASK_2_BITS, context);
633                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 2 & MASK_8BITS);
634                break;
635            case 3: // 15 bits, drop 7 and output 1 byte, or raise an exception
636                validateTrailingCharacters();
637                // Not possible from a valid encoding but decode anyway
638                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 7 & MASK_8BITS);
639                break;
640            case 4: // 20 bits = 2*8 + 4
641                validateCharacter(MASK_4_BITS, context);
642                context.lbitWorkArea = context.lbitWorkArea >> 4; // drop 4 bits
643                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 8 & MASK_8BITS);
644                buffer[context.pos++] = (byte) (context.lbitWorkArea & MASK_8BITS);
645                break;
646            case 5: // 25 bits = 3*8 + 1
647                validateCharacter(MASK_1_BITS, context);
648                context.lbitWorkArea = context.lbitWorkArea >> 1;
649                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 16 & MASK_8BITS);
650                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 8 & MASK_8BITS);
651                buffer[context.pos++] = (byte) (context.lbitWorkArea & MASK_8BITS);
652                break;
653            case 6: // 30 bits = 3*8 + 6, or raise an exception
654                validateTrailingCharacters();
655                // Not possible from a valid encoding but decode anyway
656                context.lbitWorkArea = context.lbitWorkArea >> 6;
657                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 16 & MASK_8BITS);
658                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 8 & MASK_8BITS);
659                buffer[context.pos++] = (byte) (context.lbitWorkArea & MASK_8BITS);
660                break;
661            case 7: // 35 bits = 4*8 +3
662                validateCharacter(MASK_3_BITS, context);
663                context.lbitWorkArea = context.lbitWorkArea >> 3;
664                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 24 & MASK_8BITS);
665                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 16 & MASK_8BITS);
666                buffer[context.pos++] = (byte) (context.lbitWorkArea >> 8 & MASK_8BITS);
667                buffer[context.pos++] = (byte) (context.lbitWorkArea & MASK_8BITS);
668                break;
669            default:
670                // modulus can be 0-7, and we excluded 0,1 already
671                throw new IllegalStateException("Impossible modulus " + context.modulus);
672            }
673        }
674    }
675
676    /**
677     * <p>
678     * Encodes all of the provided data, starting at inPos, for inAvail bytes. Must be called at least twice: once with the data to encode, and once with
679     * inAvail set to "-1" to alert encoder that EOF has been reached, so flush last remaining bytes (if not multiple of 5).
680     * </p>
681     *
682     * @param input   byte[] array of binary data to Base32 encode.
683     * @param inPos   Position to start reading data from.
684     * @param inAvail Amount of bytes available from input for encoding.
685     * @param context The context to be used.
686     * @throws IllegalArgumentException Thrown when a problem is detected processing data.
687     */
688    @Override
689    void encode(final byte[] input, int inPos, final int inAvail, final Context context) {
690        // package protected for access from I/O streams
691        if (context.eof) {
692            return;
693        }
694        // inAvail < 0 is how we're informed of EOF in the underlying data we're
695        // encoding.
696        if (inAvail < 0) {
697            context.eof = true;
698            if (0 == context.modulus && lineLength == 0) {
699                return; // no leftovers to process and not using chunking
700            }
701            final byte[] buffer = ensureBufferSize(encodeSize, context);
702            final int savedPos = context.pos;
703            switch (context.modulus) { // % 5
704            case 0:
705                break;
706            case 1: // Only 1 octet; take top 5 bits then remainder
707                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 3) & MASK_5_BITS]; // 8-1*5 = 3
708                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea << 2) & MASK_5_BITS]; // 5-3=2
709                buffer[context.pos++] = pad;
710                buffer[context.pos++] = pad;
711                buffer[context.pos++] = pad;
712                buffer[context.pos++] = pad;
713                buffer[context.pos++] = pad;
714                buffer[context.pos++] = pad;
715                break;
716            case 2: // 2 octets = 16 bits to use
717                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 11) & MASK_5_BITS]; // 16-1*5 = 11
718                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 6) & MASK_5_BITS]; // 16-2*5 = 6
719                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 1) & MASK_5_BITS]; // 16-3*5 = 1
720                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea << 4) & MASK_5_BITS]; // 5-1 = 4
721                buffer[context.pos++] = pad;
722                buffer[context.pos++] = pad;
723                buffer[context.pos++] = pad;
724                buffer[context.pos++] = pad;
725                break;
726            case 3: // 3 octets = 24 bits to use
727                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 19) & MASK_5_BITS]; // 24-1*5 = 19
728                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 14) & MASK_5_BITS]; // 24-2*5 = 14
729                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 9) & MASK_5_BITS]; // 24-3*5 = 9
730                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 4) & MASK_5_BITS]; // 24-4*5 = 4
731                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea << 1) & MASK_5_BITS]; // 5-4 = 1
732                buffer[context.pos++] = pad;
733                buffer[context.pos++] = pad;
734                buffer[context.pos++] = pad;
735                break;
736            case 4: // 4 octets = 32 bits to use
737                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 27) & MASK_5_BITS]; // 32-1*5 = 27
738                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 22) & MASK_5_BITS]; // 32-2*5 = 22
739                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 17) & MASK_5_BITS]; // 32-3*5 = 17
740                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 12) & MASK_5_BITS]; // 32-4*5 = 12
741                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 7) & MASK_5_BITS]; // 32-5*5 = 7
742                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 2) & MASK_5_BITS]; // 32-6*5 = 2
743                buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea << 3) & MASK_5_BITS]; // 5-2 = 3
744                buffer[context.pos++] = pad;
745                break;
746            default:
747                throw new IllegalStateException("Impossible modulus " + context.modulus);
748            }
749            context.currentLinePos += context.pos - savedPos; // keep track of current line position
750            // if currentPos == 0 we are at the start of a line, so don't add CRLF
751            if (lineLength > 0 && context.currentLinePos > 0) { // add chunk separator if required
752                System.arraycopy(lineSeparator, 0, buffer, context.pos, lineSeparator.length);
753                context.pos += lineSeparator.length;
754            }
755        } else {
756            for (int i = 0; i < inAvail; i++) {
757                final byte[] buffer = ensureBufferSize(encodeSize, context);
758                context.modulus = (context.modulus + 1) % BYTES_PER_UNENCODED_BLOCK;
759                int b = input[inPos++];
760                if (b < 0) {
761                    b += 256;
762                }
763                context.lbitWorkArea = (context.lbitWorkArea << 8) + b; // BITS_PER_BYTE
764                if (0 == context.modulus) { // we have enough bytes to create our output
765                    buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 35) & MASK_5_BITS];
766                    buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 30) & MASK_5_BITS];
767                    buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 25) & MASK_5_BITS];
768                    buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 20) & MASK_5_BITS];
769                    buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 15) & MASK_5_BITS];
770                    buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 10) & MASK_5_BITS];
771                    buffer[context.pos++] = encodeTable[(int) (context.lbitWorkArea >> 5) & MASK_5_BITS];
772                    buffer[context.pos++] = encodeTable[(int) context.lbitWorkArea & MASK_5_BITS];
773                    context.currentLinePos += BYTES_PER_ENCODED_BLOCK;
774                    if (lineLength > 0 && lineLength <= context.currentLinePos) {
775                        System.arraycopy(lineSeparator, 0, buffer, context.pos, lineSeparator.length);
776                        context.pos += lineSeparator.length;
777                        context.currentLinePos = 0;
778                    }
779                }
780            }
781        }
782    }
783
784    /**
785     * Gets the line separator (for testing only).
786     *
787     * @return The line separator.
788     */
789    byte[] getLineSeparator() {
790        return lineSeparator;
791    }
792
793    /**
794     * Tests whether the {@code octet} is in the Base32 alphabet.
795     *
796     * @param octet The value to test.
797     * @return {@code true} if the value is defined in the Base32 alphabet {@code false} otherwise.
798     */
799    @Override
800    public boolean isInAlphabet(final byte octet) {
801        final int value = octet & 0xff;
802        return value < decodeTable.length && decodeTable[value] != -1;
803    }
804
805    /**
806     * Validates whether decoding the final trailing character is possible in the context of the set of possible Base32 values.
807     * <p>
808     * The character is valid if the lower bits within the provided mask are zero. This is used to test the final trailing base-32 digit is zero in the bits
809     * that will be discarded.
810     * </p>
811     *
812     * @param emptyBitsMask The mask of the lower bits that should be empty.
813     * @param context       The context to be used.
814     * @throws IllegalArgumentException Thrown if the bits being checked contain any non-zero value.
815     */
816    private void validateCharacter(final long emptyBitsMask, final Context context) {
817        // Use the long bit work area
818        if (isStrictDecoding() && (context.lbitWorkArea & emptyBitsMask) != 0) {
819            throw new IllegalArgumentException("Strict decoding: Last encoded character (before the paddings if any) is a valid " +
820                    "Base32 alphabet but not a possible encoding. Expected the discarded bits from the character to be zero.");
821        }
822    }
823
824    /**
825     * Validates whether decoding allows final trailing characters that cannot be created during encoding.
826     *
827     * @throws IllegalArgumentException Thrown if strict decoding is enabled.
828     */
829    private void validateTrailingCharacters() {
830        if (isStrictDecoding()) {
831            throw new IllegalArgumentException("Strict decoding: Last encoded character(s) (before the paddings if any) are valid " +
832                    "Base32 alphabet but not a possible encoding. Decoding requires either 2, 4, 5, or 7 trailing 5-bit characters to create bytes.");
833        }
834    }
835}