001/* 002 * Licensed to the Apache Software Foundation (ASF) under one or more 003 * contributor license agreements. See the NOTICE file distributed with 004 * this work for additional information regarding copyright ownership. 005 * The ASF licenses this file to You under the Apache License, Version 2.0 006 * (the "License"); you may not use this file except in compliance with 007 * the License. You may obtain a copy of the License at 008 * 009 * https://www.apache.org/licenses/LICENSE-2.0 010 * 011 * Unless required by applicable law or agreed to in writing, software 012 * distributed under the License is distributed on an "AS IS" BASIS, 013 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. 014 * See the License for the specific language governing permissions and 015 * limitations under the License. 016 */ 017 018package org.apache.commons.codec.net; 019 020import java.nio.ByteBuffer; 021import java.util.BitSet; 022 023import org.apache.commons.codec.BinaryDecoder; 024import org.apache.commons.codec.BinaryEncoder; 025import org.apache.commons.codec.DecoderException; 026import org.apache.commons.codec.EncoderException; 027 028/** 029 * Implements the Percent-Encoding scheme, as described in HTTP 1.1 specification. For extensibility, an array of 030 * special US-ASCII characters can be specified in order to perform proper URI encoding for the different parts 031 * of the URI. 032 * <p> 033 * This class is immutable. It is also thread-safe besides using BitSet which is not thread-safe, but its public 034 * interface only call the access 035 * </p> 036 * 037 * @see <a href="https://tools.ietf.org/html/rfc3986#section-2.1">Percent-Encoding</a> 038 * @since 1.12 039 */ 040public class PercentCodec implements BinaryEncoder, BinaryDecoder { 041 042 /** 043 * The escape character used by the Percent-Encoding in order to introduce an encoded character. 044 */ 045 private static final byte ESCAPE_CHAR = '%'; 046 047 /** 048 * The plus character used to encode spaces when plusForSpace is true. 049 */ 050 private static final byte PLUS_CHAR = '+'; 051 052 /** 053 * The bit set used to store the character that should be always encoded. 054 */ 055 private final BitSet alwaysEncodeChars = new BitSet(); 056 057 /** 058 * The flag defining if the space character should be encoded as '+'. 059 */ 060 private final boolean plusForSpace; 061 062 /** 063 * The minimum and maximum code of the bytes that is inserted in the bit set, used to prevent look-ups 064 */ 065 private int alwaysEncodeCharsMin = Integer.MAX_VALUE, alwaysEncodeCharsMax = Integer.MIN_VALUE; 066 067 /** 068 * Constructs a Percent coded that will encode all the non US-ASCII characters using the Percent-Encoding 069 * while it will not encode all the US-ASCII characters, except for character '%' that is used as escape 070 * character for Percent-Encoding. 071 */ 072 public PercentCodec() { 073 this.plusForSpace = false; 074 insertAlwaysEncodeChar(ESCAPE_CHAR); 075 } 076 077 /** 078 * Constructs a Percent codec by specifying the characters that belong to US-ASCII that should 079 * always be encoded. The rest US-ASCII characters will not be encoded, except for character '%' that 080 * is used as escape character for Percent-Encoding. 081 * 082 * @param alwaysEncodeChars The unsafe characters that should always be encoded. 083 * @param plusForSpace The flag defining if the space character should be encoded as '+'. 084 */ 085 public PercentCodec(final byte[] alwaysEncodeChars, final boolean plusForSpace) { 086 this.plusForSpace = plusForSpace; 087 insertAlwaysEncodeChars(alwaysEncodeChars); 088 if (plusForSpace) { 089 insertAlwaysEncodeChar(PLUS_CHAR); 090 } 091 } 092 093 private boolean canEncode(final byte c) { 094 return !isAsciiChar(c) || inAlwaysEncodeCharsRange(c) && alwaysEncodeChars.get(c); 095 } 096 097 private boolean containsSpace(final byte[] bytes) { 098 for (final byte b : bytes) { 099 if (b == ' ') { 100 return true; 101 } 102 } 103 return false; 104 } 105 106 /** 107 * Decodes bytes encoded with Percent-Encoding based on RFC 3986. The reverse process is performed in order to 108 * decode the encoded characters to Unicode. 109 * 110 * @throws DecoderException Thrown if an invalid percent-encoded sequence is found. 111 */ 112 @Override 113 public byte[] decode(final byte[] bytes) throws DecoderException { 114 if (bytes == null) { 115 return null; 116 } 117 final ByteBuffer buffer = ByteBuffer.allocate(expectedDecodingBytes(bytes)); 118 for (int i = 0; i < bytes.length; i++) { 119 final byte b = bytes[i]; 120 if (b == ESCAPE_CHAR) { 121 try { 122 final int u = Utils.digit16(bytes[++i]); 123 final int l = Utils.digit16(bytes[++i]); 124 buffer.put((byte) ((u << 4) + l)); 125 } catch (final ArrayIndexOutOfBoundsException e) { 126 throw new DecoderException("Invalid percent decoding: ", e); 127 } 128 } else if (plusForSpace && b == '+') { 129 buffer.put((byte) ' '); 130 } else { 131 buffer.put(b); 132 } 133 } 134 return buffer.array(); 135 } 136 137 /** 138 * Decodes a byte[] Object, whose bytes are encoded with Percent-Encoding. 139 * 140 * @param obj The object to decode. 141 * @return The decoding result byte[] as Object. 142 * @throws DecoderException Thrown if the object is not a byte array. 143 */ 144 @Override 145 public Object decode(final Object obj) throws DecoderException { 146 if (obj == null) { 147 return null; 148 } 149 if (obj instanceof byte[]) { 150 return decode((byte[]) obj); 151 } 152 throw new DecoderException("Objects of type " + obj.getClass().getName() + " cannot be Percent decoded"); 153 } 154 155 private byte[] doEncode(final byte[] bytes, final int expectedLength, final boolean willEncode) { 156 final ByteBuffer buffer = ByteBuffer.allocate(expectedLength); 157 for (final byte b : bytes) { 158 if (willEncode && canEncode(b)) { 159 byte bb = b; 160 if (bb < 0) { 161 bb = (byte) (256 + bb); 162 } 163 final char hex1 = Utils.hexChar(bb >> 4); 164 final char hex2 = Utils.hexChar(bb); 165 buffer.put(ESCAPE_CHAR); 166 buffer.put((byte) hex1); 167 buffer.put((byte) hex2); 168 } else if (plusForSpace && b == ' ') { 169 buffer.put((byte) '+'); 170 } else { 171 buffer.put(b); 172 } 173 } 174 return buffer.array(); 175 } 176 177 /** 178 * Percent-Encoding based on RFC 3986. The non US-ASCII characters are encoded, as well as the 179 * US-ASCII characters that are configured to be always encoded. 180 * 181 * @throws EncoderException Thrown if an encoding error occurs. 182 */ 183 @Override 184 public byte[] encode(final byte[] bytes) throws EncoderException { 185 if (bytes == null) { 186 return null; 187 } 188 final int expectedEncodingBytes = expectedEncodingBytes(bytes); 189 final boolean willEncode = expectedEncodingBytes != bytes.length; 190 if (willEncode || plusForSpace && containsSpace(bytes)) { 191 return doEncode(bytes, expectedEncodingBytes, willEncode); 192 } 193 return bytes; 194 } 195 196 /** 197 * Encodes an object into using the Percent-Encoding. Only byte[] objects are accepted. 198 * 199 * @param obj The object to encode. 200 * @return The encoding result byte[] as Object. 201 * @throws EncoderException Thrown if the object is not a byte array. 202 */ 203 @Override 204 public Object encode(final Object obj) throws EncoderException { 205 if (obj == null) { 206 return null; 207 } 208 if (obj instanceof byte[]) { 209 return encode((byte[]) obj); 210 } 211 throw new EncoderException("Objects of type " + obj.getClass().getName() + " cannot be Percent encoded"); 212 } 213 214 private int expectedDecodingBytes(final byte[] bytes) { 215 int byteCount = 0; 216 for (int i = 0; i < bytes.length;) { 217 final byte b = bytes[i]; 218 i += b == ESCAPE_CHAR ? 3 : 1; 219 byteCount++; 220 } 221 return byteCount; 222 } 223 224 private int expectedEncodingBytes(final byte[] bytes) { 225 int byteCount = 0; 226 for (final byte b : bytes) { 227 byteCount += canEncode(b) ? 3 : 1; 228 } 229 return byteCount; 230 } 231 232 private boolean inAlwaysEncodeCharsRange(final byte c) { 233 return c >= alwaysEncodeCharsMin && c <= alwaysEncodeCharsMax; 234 } 235 236 /** 237 * Inserts a single character into a BitSet and maintains the min and max of the characters of the 238 * {@code BitSet alwaysEncodeChars} in order to avoid look-ups when a byte is out of this range. 239 * 240 * @param b The byte that is candidate for min and max limit. 241 */ 242 private void insertAlwaysEncodeChar(final byte b) { 243 if (b < 0) { 244 throw new IllegalArgumentException("byte must be >= 0"); 245 } 246 this.alwaysEncodeChars.set(b); 247 if (b < alwaysEncodeCharsMin) { 248 alwaysEncodeCharsMin = b; 249 } 250 if (b > alwaysEncodeCharsMax) { 251 alwaysEncodeCharsMax = b; 252 } 253 } 254 255 /** 256 * Inserts the byte array into a BitSet for faster lookup. 257 * 258 * @param alwaysEncodeCharsArray The byte array into a BitSet for faster lookup. 259 */ 260 private void insertAlwaysEncodeChars(final byte[] alwaysEncodeCharsArray) { 261 if (alwaysEncodeCharsArray != null) { 262 for (final byte b : alwaysEncodeCharsArray) { 263 insertAlwaysEncodeChar(b); 264 } 265 } 266 insertAlwaysEncodeChar(ESCAPE_CHAR); 267 } 268 269 private boolean isAsciiChar(final byte c) { 270 return c >= 0; 271 } 272}