001/* 002 * Licensed to the Apache Software Foundation (ASF) under one or more 003 * contributor license agreements. See the NOTICE file distributed with 004 * this work for additional information regarding copyright ownership. 005 * The ASF licenses this file to You under the Apache License, Version 2.0 006 * (the "License"); you may not use this file except in compliance with 007 * the License. You may obtain a copy of the License at 008 * 009 * https://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 014 * See the License for the specific language governing permissions and 015 * limitations under the License. 016 */ 017 018package org.apache.commons.codec.binary; 019 020import java.util.Arrays; 021 022import org.apache.commons.codec.CodecPolicy; 023 024/** 025 * Provides Base45 encoding and decoding as defined by <a href="https://datatracker.ietf.org/doc/html/rfc9285">RFC 9285</a>. 026 * <p> 027 * Base45 is designed for efficient encoding of binary data in environments where a subset of ASCII characters is available, specifically 45 characters chosen 028 * from the QR code alphanumeric mode character set. Base45 is used in European Union Digital COVID Certificates (EUDCC) and similar applications. 029 * </p> 030 * <p> 031 * The Base45 alphabet consists of 45 characters: 032 * </p> 033 * 034 * <pre> 035 * Value Encoding Value Encoding Value Encoding Value Encoding 036 * 0 0 12 C 24 O 36 Space 037 * 1 1 13 D 25 P 37 $ 038 * 2 2 14 E 26 Q 38 % 039 * 3 3 15 F 27 R 39 * 040 * 4 4 16 G 28 S 40 + 041 * 5 5 17 H 29 T 41 - 042 * 6 6 18 I 30 U 42 . 043 * 7 7 19 J 31 V 43 / 044 * 8 8 20 K 32 W 44 : 045 * 9 9 21 L 33 X 046 * 10 A 22 M 34 Y 047 * 11 B 23 N 35 Z 048 * </pre> 049 * 050 * <h2>Encoding</h2> 051 * <p> 052 * Input bytes are grouped in pairs (2 bytes). Each pair is encoded as 3 Base45 characters. A single remaining byte is encoded as 2 Base45 characters. There is 053 * no padding. 054 * </p> 055 * <ul> 056 * <li>For each 2-byte pair {@code (b0, b1)}: {@code n = b0 * 256 + b1}; output 3 characters {@code alphabet[n % 45]}, {@code alphabet[(n / 45) % 45]}, 057 * {@code alphabet[n / 2025]}</li> 058 * <li>For a final single byte {@code b0}: {@code n = b0}; output 2 characters {@code alphabet[n % 45]}, {@code alphabet[n / 45]}</li> 059 * </ul> 060 * <h2>Decoding</h2> 061 * <p> 062 * Input characters are grouped in triples (3 characters). Each triple decodes to 2 bytes. A pair of trailing characters decodes to 1 byte. An input whose 063 * length modulo 3 equals 1 is invalid. 064 * </p> 065 * <p> 066 * This class is thread-safe. 067 * </p> 068 * <p> 069 * To create an instance, use the default constructor or the builder: 070 * </p> 071 * 072 * <pre> 073 * Base45 codec = new Base45(); 074 * 075 * // Or, use the builder to customize the encode table: 076 * Base45 custom = Base45.builder().setEncodeTable(...).get(); 077 * </pre> 078 * 079 * @see <a href="https://datatracker.ietf.org/doc/html/rfc9285">RFC 9285 – The Base45 Data Encoding</a> 080 * @since 1.23.0 081 */ 082public class Base45 extends BaseNCodec { 083 084 /** 085 * Builds {@link Base45} instances. 086 * <p> 087 * To configure a new instance, use a {@link Builder}. For example: 088 * </p> 089 * 090 * <pre> 091 * 092 * Base45 base45 = Base45.builder().get(); 093 * </pre> 094 * 095 * @since 1.23.0 096 */ 097 public static class Builder extends AbstractBuilder<Base45, Builder> { 098 099 /** 100 * Constructs a new instance using the Base45 alphabet as defined by RFC 9285. 101 */ 102 public Builder() { 103 super(ENCODE_TABLE); 104 setDecodingPolicy(CodecPolicy.STRICT); 105 setDecodeTableRaw(DECODE_TABLE); 106 setEncodeTableRaw(ENCODE_TABLE); 107 setEncodedBlockSize(BYTES_PER_ENCODED_BLOCK); 108 setUnencodedBlockSize(BYTES_PER_UNENCODED_BLOCK); 109 } 110 111 @Override 112 public Base45 get() { 113 return new Base45(this); 114 } 115 116 /** 117 * Sets the decoding policy. {@link CodecPolicy#STRICT} is the only supported policy. 118 * 119 * @param decodingPolicy The decoding policy; {@code null} resets to the default ({@link CodecPolicy#STRICT}). 120 * @return {@code this} instance. 121 * @throws IllegalArgumentException Thrown if the given policy is {@link CodecPolicy#LENIENT}. 122 */ 123 @Override 124 public Builder setDecodingPolicy(final CodecPolicy decodingPolicy) { 125 if (decodingPolicy == CodecPolicy.LENIENT) { 126 throw new IllegalArgumentException("CodecPolicy.STRICT is the only supported policy."); 127 } 128 return super.setDecodingPolicy(decodingPolicy != null ? decodingPolicy : CodecPolicy.STRICT); 129 } 130 131 /** 132 * Sets the encode table and derives the matching decode table, so the codec can always decode its own output. 133 * 134 * @param encodeTable The encode table with exactly 45 unique entries, null resets to the default. 135 * @return {@code this} instance. 136 * @throws IllegalArgumentException Thrown if the encode table does not contain exactly 45 unique entries. 137 */ 138 @Override 139 public Builder setEncodeTable(final byte... encodeTable) { 140 super.setDecodeTableRaw(toDecodeTable(encodeTable)); 141 return super.setEncodeTable(encodeTable); 142 } 143 144 /** 145 * Always throws {@link UnsupportedOperationException}; unsupported by Base45 RFC 9285. 146 * 147 * @throws UnsupportedOperationException Thrown because Base45 RFC 9285 does not support this operation. 148 */ 149 @Override 150 public Builder setLineLength(final int lineLength) { 151 throw new UnsupportedOperationException("Unsupported by Base45 RFC 9285"); 152 } 153 154 /** 155 * Always throws {@link UnsupportedOperationException}; unsupported by Base45 RFC 9285. 156 * 157 * @throws UnsupportedOperationException Thrown because Base45 RFC 9285 does not support this operation. 158 */ 159 @Override 160 public Builder setLineSeparator(final byte... lineSeparator) { 161 throw new UnsupportedOperationException("Unsupported by Base45 RFC 9285"); 162 } 163 164 /** 165 * Always throws {@link UnsupportedOperationException}; unsupported by Base45 RFC 9285. 166 * 167 * @throws UnsupportedOperationException Thrown because Base45 RFC 9285 does not support this operation. 168 */ 169 @Override 170 public Builder setPadding(final byte padding) { 171 throw new UnsupportedOperationException("Unsupported by Base45 RFC 9285"); 172 } 173 } 174 175 /** 176 * The number of characters in the Base45 alphabet. 177 */ 178 private static final int BASE = 45; 179 180 /** 181 * The square of the Base45 alphabet size (45 * 45 = 2025), used during decoding. 182 */ 183 private static final int BASE_SQUARED = BASE * BASE; // 2025 184 185 /** 186 * Number of Base45 characters in an encoded block (encoding 2 unencoded bytes). 187 */ 188 static final int BYTES_PER_ENCODED_BLOCK = 3; 189 190 private static final int TAIL_ENCODED_BLOCK = BYTES_PER_ENCODED_BLOCK - 1; 191 192 /** 193 * Number of unencoded bytes per full encoding block. 194 */ 195 static final int BYTES_PER_UNENCODED_BLOCK = 2; 196 197 /** 198 * Lookup table translating ASCII character values (0–127) to their Base45 alphabet index (0–44), or -1 if the character is not in the Base45 alphabet. 199 */ 200 // @formatter:off 201 static final byte[] DECODE_TABLE = { 202 // 0 1 2 3 4 5 6 7 8 9 A B C D E F 203 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 00-0f 204 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 10-1f 205 36, -1, -1, -1, 37, 38, -1, -1, -1, -1, 39, 40, -1, 41, 42, 43, // 20-2f ' ','$','%','*','+','-','.','/ 206 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 44, -1, -1, -1, -1, -1, // 30-3f '0'-'9', ':' 207 -1, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23, 24, // 40-4f 'A'-'O' 208 25, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, -1, -1, -1, -1, -1, // 50-5f 'P'-'Z' 209 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 60-6f 210 -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, -1, // 70-7f 211 }; 212 213 // @formatter:on 214 /** 215 * Lookup table translating Base45 values (0–44) to their ASCII character equivalents. 216 * <p> 217 * As specified in <a href="https://datatracker.ietf.org/doc/html/rfc9285">RFC 9285</a>: {@code 0-9, A-Z, Space, $, %, *, +, -, ., /, :} 218 * </p> 219 */ 220 // @formatter:off 221 static final byte[] ENCODE_TABLE = { 222 '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', // 0-9 223 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', // 10-21 224 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', // 22-33 225 'Y', 'Z', // 34-35 226 ' ', '$', '%', '*', '+', '-', '.', '/', ':', // 36-44 227 }; 228 // @formatter:on 229 230 /** 231 * Creates a new {@link Builder} for configuring a {@link Base45} instance. 232 * 233 * @return A new {@link Builder}. 234 */ 235 public static Builder builder() { 236 return new Builder(); 237 } 238 239 /** 240 * Constructs the decode table matching the given encode table. 241 * 242 * @param encodeTable The encode table. 243 * @return A new decode table. 244 * @throws IllegalArgumentException Thrown if the encode table does not contain exactly 45 unique entries. 245 */ 246 private static byte[] calculateDecodeTable(final byte[] encodeTable) { 247 if (encodeTable.length != BASE) { 248 throw new IllegalArgumentException("encodeTable must have exactly " + BASE + " entries."); 249 } 250 final byte[] decodeTable = new byte[DECODE_TABLE.length]; 251 Arrays.fill(decodeTable, (byte) -1); 252 for (int i = 0; i < encodeTable.length; i++) { 253 final int encodedByte = encodeTable[i] & 0xff; 254 if (encodedByte >= decodeTable.length || decodeTable[encodedByte] != -1) { 255 throw new IllegalArgumentException("encodeTable entries must be unique values in the range 0-127."); 256 } 257 decodeTable[encodedByte] = (byte) i; 258 } 259 return decodeTable; 260 } 261 262 /** 263 * Gets the decode table that matches the given encode table. 264 * 265 * @param encodeTable The encode table used to determine the decode lookup table. 266 * @return The matching decode table. 267 */ 268 private static byte[] toDecodeTable(final byte[] encodeTable) { 269 final byte[] table = encodeTable != null ? encodeTable : ENCODE_TABLE; 270 if (Arrays.equals(table, ENCODE_TABLE)) { 271 return DECODE_TABLE; 272 } 273 return calculateDecodeTable(table); 274 } 275 276 /** 277 * Constructs a Base45 codec using the default settings (strict decoding policy, the only supported policy). 278 */ 279 public Base45() { 280 this(builder()); 281 } 282 283 /** 284 * Constructs a Base45 codec from a builder. 285 * 286 * @param builder The builder to configure this instance. 287 */ 288 private Base45(final Builder builder) { 289 super(builder); 290 } 291 292 /** 293 * Decodes all of the provided data, starting at {@code inPos}, for {@code inAvail} bytes. 294 * <p> 295 * This method must be called at least twice: once with the data to decode, and once with {@code inAvail} set to {@code -1} to notify the decoder that EOF 296 * has been reached. 297 * </p> 298 * <p> 299 * Input characters not in the Base45 alphabet, including CR, LF, and TAB, cause an {@link IllegalArgumentException}. Space {@code ' '} is part of the 300 * Base45 alphabet and is decoded as data. 301 * </p> 302 * 303 * @param input byte array of Base45-encoded character data to decode. 304 * @param inPos Position to start reading data from. 305 * @param inAvail Number of bytes available from {@code input} for decoding, or {@code -1} to signal EOF. 306 * @param context The context to be used. 307 * @throws IllegalArgumentException Thrown if the input contains an invalid character, if the encoded length modulo 3 equals 1, or if an encoded triple decodes to 308 * a value exceeding 65535, or if a trailing 2-character sequence decodes to a value greater than 255. 309 */ 310 @Override 311 void decode(final byte[] input, int inPos, final int inAvail, final Context context) { 312 // package-protected for access from I/O streams 313 if (context.eof) { 314 return; 315 } 316 if (inAvail < 0) { 317 context.eof = true; 318 switch (context.modulus) { 319 case 0: 320 // Nothing to do; input length is a multiple of 3. 321 break; 322 case 1: 323 // RFC 9285: "It is an error if the remaining string length is 1 character." 324 throw new IllegalArgumentException("Invalid Base45 encoding: encoded input length modulo 3 must not equal 1."); 325 case 2: 326 // Two trailing characters decode to one byte. 327 // Maximum decodable value from two Base45 characters: 44 + 44*45 = 2024. 328 // Valid single-byte encodings have a decoded value in [0, 255]. 329 if (context.ibitWorkArea > 0xFF) { 330 throw new IllegalArgumentException("Invalid Base45 encoding: trailing 2-character sequence decodes to " + context.ibitWorkArea + 331 ", which exceeds the valid byte range (0-255)."); 332 } 333 ensureBufferSize(1, context)[context.pos++] = (byte) context.ibitWorkArea; 334 break; 335 default: 336 throw new IllegalStateException("Impossible modulus " + context.modulus); 337 } 338 return; 339 } 340 for (int i = 0; i < inAvail; i++) { 341 final int b = input[inPos++] & 0xFF; 342 if (b >= decodeTable.length || decodeTable[b] < 0) { 343 throw new IllegalArgumentException("Invalid Base45 character '" + (char) b + "' (value " + b + ")."); 344 } 345 final int value = decodeTable[b]; 346 switch (context.modulus) { 347 case 0: 348 // First character of a 3-character group: initialize accumulator. 349 context.ibitWorkArea = value; 350 context.modulus = 1; 351 break; 352 case 1: 353 // Second character of a 3-character group. 354 context.ibitWorkArea += value * BASE; 355 context.modulus = 2; 356 break; 357 case 2: 358 // Third character of a 3-character group: compute value and output 2 bytes. 359 context.ibitWorkArea += value * BASE_SQUARED; 360 context.modulus = 0; 361 if (context.ibitWorkArea > 0xFFFF) { 362 throw new IllegalArgumentException("Invalid Base45 encoding: 3-character sequence decodes to " + context.ibitWorkArea + 363 ", which exceeds the valid 16-bit range (0-65535)."); 364 } 365 final byte[] buffer = ensureBufferSize(BYTES_PER_UNENCODED_BLOCK, context); 366 buffer[context.pos++] = (byte) (context.ibitWorkArea >> 8); 367 buffer[context.pos++] = (byte) (context.ibitWorkArea & 0xFF); 368 context.ibitWorkArea = 0; 369 break; 370 default: 371 throw new IllegalStateException("Impossible modulus " + context.modulus); 372 } 373 } 374 } 375 376 /** 377 * Encodes all of the provided data, starting at {@code inPos}, for {@code inAvail} bytes. 378 * <p> 379 * This method must be called at least twice: once with the data to encode, and once with {@code inAvail} set to {@code -1} to notify the encoder that EOF 380 * has been reached. 381 * </p> 382 * <p> 383 * Each pair of input bytes is encoded to 3 Base45 characters. A final single byte is encoded as 2 Base45 characters. No padding is used. 384 * </p> 385 * 386 * @param input byte array of binary data to Base45-encode. 387 * @param inPos Position to start reading data from. 388 * @param inAvail Number of bytes available from {@code input} for encoding, or {@code -1} to signal EOF. 389 * @param context The context to be used. 390 */ 391 @Override 392 void encode(final byte[] input, int inPos, final int inAvail, final Context context) { 393 // package-protected for access from I/O streams 394 if (context.eof) { 395 return; 396 } 397 if (inAvail < 0) { 398 context.eof = true; 399 if (context.modulus == 1) { 400 // One remaining byte: encode as 2 Base45 characters. 401 final byte[] buffer = ensureBufferSize(TAIL_ENCODED_BLOCK, context); 402 final int n = context.ibitWorkArea & 0xFF; 403 buffer[context.pos++] = encodeTable[n % BASE]; 404 buffer[context.pos++] = encodeTable[n / BASE]; 405 } 406 // If modulus == 0, all bytes have been encoded; nothing to flush. 407 return; 408 } 409 for (int i = 0; i < inAvail; i++) { 410 final int b = input[inPos++] & 0xFF; 411 // Accumulate byte into work area and advance modulus. 412 context.modulus = (context.modulus + 1) % BYTES_PER_UNENCODED_BLOCK; 413 // Shift the accumulated value left by 8 bits and add the new byte. 414 context.ibitWorkArea = (context.ibitWorkArea << 8) + b; 415 if (context.modulus == 0) { 416 // We have a complete 2-byte group; encode as 3 Base45 characters. 417 final byte[] buffer = ensureBufferSize(BYTES_PER_ENCODED_BLOCK, context); 418 // The work area holds: b0 * 256 + b1 (a 16-bit value, 0–65535). 419 int n = context.ibitWorkArea & 0xFFFF; 420 buffer[context.pos++] = encodeTable[n % BASE]; 421 n /= BASE; 422 buffer[context.pos++] = encodeTable[n % BASE]; 423 n /= BASE; 424 buffer[context.pos++] = encodeTable[n]; 425 context.ibitWorkArea = 0; 426 } 427 } 428 } 429 430 /** 431 * Gets the number of Base45-encoded characters needed to encode the given byte array, as specified by RFC 9285. 432 * <p> 433 * The formula is: {@code (n / 2) * 3 + (n % 2 != 0 ? 2 : 0)}, where {@code n} is the number of unencoded bytes. 434 * </p> 435 * 436 * @param array The byte array to encode (used only for its length). 437 * @return The number of Base45 characters that would be produced by encoding {@code array}. 438 */ 439 @Override 440 public long getEncodedLength(final byte[] array) { 441 final long n = array.length; 442 return n / 2 * 3 + (n % 2 != 0 ? 2 : 0); 443 } 444 445 /** 446 * Tests whether or not the {@code value} is a valid Base45 alphabet character. 447 * 448 * @param value The byte value to test. 449 * @return {@code true} if the byte corresponds to a character in the Base45 alphabet (RFC 9285); {@code false} otherwise. 450 */ 451 @Override 452 public boolean isInAlphabet(final byte value) { 453 final int v = value & 0xFF; 454 return v < decodeTable.length && decodeTable[v] >= 0; 455 } 456 457 /** 458 * Tests a given byte array to see if it contains only valid characters within the alphabet. The method optionally treats whitespace as valid. 459 * <p> 460 * Unlike the {@link BaseNCodec} implementation, the pad character is <em>not</em> considered valid, because Base45 (RFC 9285) has no padding. 461 * </p> 462 * 463 * @param arrayOctet byte array to test. 464 * @param allowWhitespacePad if {@code true}, then whitespace is also allowed. 465 * @return {@code true} if all bytes are valid characters in the alphabet or if the byte array is empty; {@code false}, otherwise. 466 */ 467 @Override 468 public boolean isInAlphabet(final byte[] arrayOctet, final boolean allowWhitespacePad) { 469 for (final byte octet : arrayOctet) { 470 if (!isInAlphabet(octet) && (!allowWhitespacePad || !Character.isWhitespace(octet))) { 471 return false; 472 } 473 } 474 return true; 475 } 476}