publicfinalclass URI implements Comparable<URI>, Serializable
{
// Note: Comments containing the word "ASSERT" indicate places where a // throw of an InternalError should be replaced by an appropriate assertion // statement once asserts are enabled in the build.
@java.io.Serial staticfinallong serialVersionUID = -6052424284110960213L;
// -- Properties and components of this instance --
// Components of all URIs: [<scheme>:]<scheme-specific-part>[#<fragment>] privatetransient String scheme; // null ==> relative URI privatetransient String fragment;
// Hierarchical URI components: [//<authority>]<path>[?<query>] privatetransient String authority; // Registry or server
// The remaining fields may be computed on demand, which is safe even in // the face of multiple threads racing to initialize them privatetransient String schemeSpecificPart; privatetransientint hash; // Zero ==> undefined
/** *AttemptstoparsethisURI'sauthoritycomponent,ifdefined,into *user-information,host,andportcomponents. * *<p>IfthisURI'sauthoritycomponenthasalreadybeenrecognizedas *beingserver-basedthenitwillalreadyhavebeenparsedinto *user-information,host,andportcomponents.Inthiscase,orifthis *URIhasnoauthoritycomponent,thismethodsimplyreturnsthisURI. * *<p>Otherwisethismethodattemptsoncemoretoparsetheauthority *componentintouser-information,host,andportcomponents,andthrows *anexceptiondescribingwhytheauthoritycomponentcouldnotbeparsed *inthatway. * *<p>ThismethodisprovidedbecausethegenericURIsyntaxspecifiedin *<ahref="http://www.ietf.org/rfc/rfc2396.txt">RFC 2396</a> *cannotalwaysdistinguishamalformedserver-basedauthorityfroma *legitimateregistry-basedauthority.Itmustthereforetreatsome *instancesoftheformerasinstancesofthelatter.Theauthority *componentintheURIstring{@code"//foo:bar"}, for example, is not a *legalserver-basedauthoritybutitislegalasaregistry-based *authority. * *<p>Inmanycommonsituations,forexamplewhenworkingURIsthatare *knowntobeeitherURNsorURLs,thehierarchicalURIsbeingusedwill *alwaysbeserver-based.Theythereforemusteitherbeparsedassuchor *treatedasanerror.Inthesecasesastatementsuchas * *<blockquote> *{@codeURI}<i>u</i>{@code=newURI(str).parseServerAuthority();} *</blockquote> * *<p>canbeusedtoensurethat<i>u</i>alwaysreferstoaURIthat,if *ithasanauthoritycomponent,hasaserver-basedauthoritywithproper *user-information,host,andportcomponents.Invokingthismethodalso *ensuresthatiftheauthoritycouldnotbeparsedinthatwaythenan *appropriatediagnosticmessagecanbeissuedbasedupontheexception *thatisthrown.</p> * *@returnAURIwhoseauthorityfieldhasbeenparsed *asaserver-basedauthority * *@throwsURISyntaxException *IftheauthoritycomponentofthisURIisdefined *butcannotbeparsedasaserver-basedauthority *accordingtoRFC 2396
*/ public URI parseServerAuthority() throws URISyntaxException
{ // We could be clever and cache the error message and index from the // exception thrown during the original parse, but that would require // either more fields or a more-obscure representation. if ((host != null) || (authority == null)) returnthis; new Parser(toString()).parse(true); returnthis;
}
/** *Returnstherawscheme-specificpartofthisURI.Thescheme-specific *partisneverundefined,thoughitmaybeempty. * *<p>Thescheme-specificpartofaURIonlycontainslegalURI *characters.</p> * *@returnTherawscheme-specificpartofthisURI *(never{@codenull})
*/ public String getRawSchemeSpecificPart() {
String part = schemeSpecificPart; if (part != null) { return part;
}
String s = string; if (s != null) { // if string is defined, components will have been parsed int start = 0; int end = s.length(); if (scheme != null) {
start = scheme.length() + 1;
} if (fragment != null) {
end -= fragment.length() + 1;
} if (path != null && path.length() == end - start) {
part = path;
} else {
part = s.substring(start, end);
}
} else {
StringBuilder sb = new StringBuilder();
appendSchemeSpecificPart(sb, null, getAuthority(), getUserInfo(),
host, port, getPath(), getQuery());
part = sb.toString();
} return schemeSpecificPart = part;
}
/** *Returnsthedecodedscheme-specificpartofthisURI. * *<p>Thestringreturnedbythismethodisequaltothatreturnedbythe *{@link#getRawSchemeSpecificPart()getRawSchemeSpecificPart}method *exceptthatallsequencesofescapedoctetsare<a *href="#decode">decoded</a>.</p> * *@returnThedecodedscheme-specificpartofthisURI *(never{@codenull})
*/ public String getSchemeSpecificPart() {
String part = decodedSchemeSpecificPart; if (part == null) {
decodedSchemeSpecificPart = part = decode(getRawSchemeSpecificPart());
} return part;
}
if ((c = compareIgnoringCase(this.scheme, that.scheme)) != 0) return c;
if (this.isOpaque()) { if (that.isOpaque()) { // Both opaque if ((c = compare(this.schemeSpecificPart,
that.schemeSpecificPart)) != 0) return c; return compare(this.fragment, that.fragment);
} return +1; // Opaque > hierarchical
} elseif (that.isOpaque()) { return -1; // Hierarchical < opaque
}
// Hierarchical if ((this.host != null) && (that.host != null)) { // Both server-based if ((c = compare(this.userInfo, that.userInfo)) != 0) return c; if ((c = compareIgnoringCase(this.host, that.host)) != 0) return c; if ((c = this.port - that.port) != 0) return c;
} else { // If one or both authorities are registry-based then we simply // compare them in the usual, case-sensitive way. If one is // registry-based and one is server-based then the strings are // guaranteed to be unequal, hence the comparison will never return // zero and the compareTo and equals methods will remain // consistent. if ((c = compare(this.authority, that.authority)) != 0) return c;
}
// -- Utility methods for string-field comparison and hashing --
// These methods return appropriate values for null string arguments, // thereby simplifying the equals, hashCode, and compareTo methods. // // The case-ignoring methods should only be applied to strings whose // characters are all known to be US-ASCII. Because of this restriction, // these methods are faster than the similar methods in the String class.
// US-ASCII only privatestaticint toLower(char c) { if ((c >= 'A') && (c <= 'Z')) return c + ('a' - 'A'); return c;
}
// US-ASCII only privatestaticint toUpper(char c) { if ((c >= 'a') && (c <= 'z')) return c - ('a' - 'A'); return c;
}
privatestaticboolean equal(String s, String t) { boolean testForEquality = true; int result = percentNormalizedComparison(s, t, testForEquality); return result == 0;
}
// US-ASCII only privatestaticboolean equalIgnoringCase(String s, String t) { if (s == t) returntrue; if ((s != null) && (t != null)) { int n = s.length(); if (t.length() != n) returnfalse; for (int i = 0; i < n; i++) { if (toLower(s.charAt(i)) != toLower(t.charAt(i))) returnfalse;
} returntrue;
} returnfalse;
}
privatestaticint normalizedHash(int hash, String s) { int h = 0; for (int index = 0; index < s.length(); index++) { char ch = s.charAt(index);
h = 31 * h + ch; if (ch == '%') { /* *Processthenexttwoencodedcharacters
*/ for (int i = index + 1; i < index + 3; i++)
h = 31 * h + toUpper(s.charAt(i));
index += 2;
}
} return hash * 127 + h;
}
// US-ASCII only privatestaticint hashIgnoringCase(int hash, String s) { if (s == null) return hash; int h = hash; int n = s.length(); for (int i = 0; i < n; i++)
h = 31 * h + toLower(s.charAt(i)); return h;
}
privatestaticint compare(String s, String t) { boolean testForEquality = false; int result = percentNormalizedComparison(s, t, testForEquality); return result;
}
// The percentNormalizedComparison method does not verify two // characters that follow the % sign are hexadecimal digits. // Reason being: // 1) percentNormalizedComparison method is not called with // 'decoded' strings // 2) The only place where a percent can be followed by anything // other than hexadecimal digits is in the authority component // (for a IPv6 scope) and the whole authority component is case // insensitive. privatestaticint percentNormalizedComparison(String s, String t, boolean testForEquality) {
if (s == t) return0; if (s != null) { if (t != null) { if (s.indexOf('%') < 0) { return s.compareTo(t);
} int sn = s.length(); int tn = t.length(); if ((sn != tn) && testForEquality) return sn - tn; int val = 0; int n = Math.min(sn, tn); for (int i = 0; i < n; ) { char c = s.charAt(i); char d = t.charAt(i);
val = c - d; if (c != '%') { if (val != 0) return val;
i++; continue;
} if (d != '%') { if (val != 0) return val;
}
i++;
val = toLower(s.charAt(i)) - toLower(t.charAt(i)); if (val != 0) return val;
i++;
val = toLower(s.charAt(i)) - toLower(t.charAt(i)); if (val != 0) return val;
i++;
} return sn - tn;
} else return +1;
} else { return -1;
}
}
// US-ASCII only privatestaticint compareIgnoringCase(String s, String t) { if (s == t) return0; if (s != null) { if (t != null) { int sn = s.length(); int tn = t.length(); int n = sn < tn ? sn : tn; for (int i = 0; i < n; i++) { int c = toLower(s.charAt(i)) - toLower(t.charAt(i)); if (c != 0) return c;
} return sn - tn;
} return +1;
} else { return -1;
}
}
// -- String construction --
// If a scheme is given then the path, if given, must be absolute // privatestaticvoid checkPath(String s, String scheme, String path) throws URISyntaxException
{ if (scheme != null) { if (path != null && !path.isEmpty() && path.charAt(0) != '/') thrownew URISyntaxException(s, "Relative path in absolute URI");
}
}
privatevoid appendAuthority(StringBuilder sb,
String authority,
String userInfo,
String host, int port)
{ if (host != null) {
sb.append("//"); if (userInfo != null) {
sb.append(quote(userInfo, L_USERINFO, H_USERINFO));
sb.append('@');
} boolean needBrackets = ((host.indexOf(':') >= 0)
&& !host.startsWith("[")
&& !host.endsWith("]")); if (needBrackets) sb.append('[');
sb.append(host); if (needBrackets) sb.append(']'); if (port != -1) {
sb.append(':');
sb.append(port);
}
} elseif (authority != null) {
sb.append("//"); if (authority.startsWith("[")) { // authority should (but may not) contain an embedded IPv6 address int end = authority.indexOf(']');
String doquote = authority; if (end != -1 && authority.indexOf(':') != -1) { // the authority contains an IPv6 address
sb.append(authority, 0, end + 1);
doquote = authority.substring(end + 1);
}
sb.append(quote(doquote,
L_REG_NAME | L_SERVER,
H_REG_NAME | H_SERVER));
} else {
sb.append(quote(authority,
L_REG_NAME | L_SERVER,
H_REG_NAME | H_SERVER));
}
}
}
privatevoid appendSchemeSpecificPart(StringBuilder sb,
String opaquePart,
String authority,
String userInfo,
String host, int port,
String path,
String query)
{ if (opaquePart != null) { /* check if SSP begins with an IPv6 address *becausewemustnotquotealiteralIPv6address
*/ if (opaquePart.startsWith("//[")) { int end = opaquePart.indexOf(']'); if (end != -1 && opaquePart.indexOf(':')!=-1) {
String doquote = opaquePart.substring(end + 1);
sb.append(opaquePart, 0, end + 1);
sb.append(quote(doquote, L_URIC, H_URIC));
}
} else {
sb.append(quote(opaquePart, L_URIC, H_URIC));
}
} else {
appendAuthority(sb, authority, userInfo, host, port); if (path != null)
sb.append(quote(path, L_PATH, H_PATH)); if (query != null) {
sb.append('?');
sb.append(quote(query, L_URIC, H_URIC));
}
}
}
// -- Normalization, resolution, and relativization --
// RFC2396 5.2 (6) privatestatic String resolvePath(String base, String child, boolean absolute)
{ int i = base.lastIndexOf('/'); int cn = child.length();
String path = "";
if (cn == 0) { // 5.2 (6a) if (i >= 0)
path = base.substring(0, i + 1);
} else { // 5.2 (6a-b) if (i >= 0 || !absolute) {
path = base.substring(0, i + 1).concat(child);
} else {
path = "/".concat(child);
}
}
// 5.2 (6c-f)
String np = normalize(path);
// 5.2 (6g): If the result is absolute but the path begins with "../", // then we simply leave the path as-is
return np;
}
// RFC2396 5.2 privatestatic URI resolve(URI base, URI child) { // check if child if opaque first so that NPE is thrown // if child is null. if (child.isOpaque() || base.isOpaque()) return child;
// 5.2 (7): Recombine (nothing to do here) return ru;
}
// If the given URI's path is normal then return the URI; // o.w., return a new URI containing the normalized path. // privatestatic URI normalize(URI u) { if (u.isOpaque() || u.path == null || u.path.isEmpty()) return u;
String np = normalize(u.path); if (np == u.path) return u;
// If both URIs are hierarchical, their scheme and authority components are // identical, and the base path is a prefix of the child's path, then // return a relative URI that, when resolved against the base, yields the // child; otherwise, return the child. // privatestatic URI relativize(URI base, URI child) { // check if child if opaque first so that NPE is thrown // if child is null. if (child.isOpaque() || base.isOpaque()) return child; if (!equalIgnoringCase(base.scheme, child.scheme)
|| !equal(base.authority, child.authority)) return child;
String bp = normalize(base.path);
String cp = normalize(child.path); if (!bp.equals(cp)) { if (!bp.endsWith("/"))
bp = bp + "/"; if (!cp.startsWith(bp)) return child;
}
URI v = new URI();
v.path = cp.substring(bp.length());
v.query = child.query;
v.fragment = child.fragment; return v;
}
// -- Path normalization --
// The following algorithm for path normalization avoids the creation of a // string object for each segment, as well as the use of a string buffer to // compute the final result, by using a single char array and editing it in // place. The array is first split into segments, replacing each slash // with '\0' and creating a segment-index array, each element of which is // the index of the first char in the corresponding segment. We then walk // through both arrays, removing ".", "..", and other segments as necessary // by setting their entries in the index array to -1. Finally, the two // arrays are used to rejoin the segments and compute the final result. // // This code is based upon src/solaris/native/java/io/canonicalize_md.c
// Check the given path to see if it might need normalization. A path // might need normalization if it contains duplicate slashes, a "." // segment, or a ".." segment. Return -1 if no further normalization is // possible, otherwise return the number of segments found. // // This method takes a string argument rather than a char array so that // this test can be performed without invoking path.toCharArray(). // privatestaticint needsNormalization(String path) { boolean normal = true; int ns = 0; // Number of segments int end = path.length() - 1; // Index of last char in path int p = 0; // Index of next char in path
// Skip initial slashes while (p <= end) { if (path.charAt(p) != '/') break;
p++;
} if (p > 1) normal = false;
// Find beginning of next segment while (p <= end) { if (path.charAt(p++) != '/') continue;
// Skip redundant slashes while (p <= end) { if (path.charAt(p) != '/') break;
normal = false;
p++;
}
break;
}
}
return normal ? -1 : ns;
}
// Split the given path into segments, replacing slashes with nulls and // filling in the given segment-index array. // // Preconditions: // segs.length == Number of segments in path // // Postconditions: // All slashes in path replaced by '\0' // segs[i] == Index of first char in segment i (0 <= i < segs.length) // privatestaticvoid split(char[] path, int[] segs) { int end = path.length - 1; // Index of last char in path int p = 0; // Index of next char in path int i = 0; // Index of current segment
// Skip initial slashes while (p <= end) { if (path[p] != '/') break;
path[p] = '\0';
p++;
}
while (p <= end) {
// Note start of segment
segs[i++] = p++;
// Find beginning of next segment while (p <= end) { if (path[p++] != '/') continue;
path[p - 1] = '\0';
if (i != segs.length) thrownew InternalError(); // ASSERT
}
// Join the segments in the given path according to the given segment-index // array, ignoring those segments whose index entries have been set to -1, // and inserting slashes as needed. Return the length of the resulting // path. // // Preconditions: // segs[i] == -1 implies segment i is to be ignored // path computed by split, as above, with '\0' having replaced '/' // // Postconditions: // path[0] .. path[return value] == Resulting path // privatestaticint join(char[] path, int[] segs) { int ns = segs.length; // Number of segments int end = path.length - 1; // Index of last char in path int p = 0; // Index of next path char to write
if (path[p] == '\0') { // Restore initial slash for absolute paths
path[p++] = '/';
}
for (int i = 0; i < ns; i++) { int q = segs[i]; // Current segment if (q == -1) // Ignore this segment continue;
if (p == q) { // We're already at this segment, so just skip to its end while ((p <= end) && (path[p] != '\0'))
p++; if (p <= end) { // Preserve trailing slash
path[p++] = '/';
}
} elseif (p < q) { // Copy q down to p while ((q <= end) && (path[q] != '\0'))
path[p++] = path[q++]; if (q <= end) { // Preserve trailing slash
path[p++] = '/';
}
} else thrownew InternalError(); // ASSERT false
}
return p;
}
// Remove "." segments from the given path, and remove segment pairs // consisting of a non-".." segment followed by a ".." segment. // privatestaticvoid removeDots(char[] path, int[] segs) { int ns = segs.length; int end = path.length - 1;
for (int i = 0; i < ns; i++) { int dots = 0; // Number of dots found (0, 1, or 2)
// Find next occurrence of "." or ".." do { int p = segs[i]; if (path[p] == '.') { if (p == end) {
dots = 1; break;
} elseif (path[p + 1] == '\0') {
dots = 1; break;
} elseif ((path[p + 1] == '.')
&& ((p + 1 == end)
|| (path[p + 2] == '\0'))) {
dots = 2; break;
}
}
i++;
} while (i < ns); if ((i > ns) || (dots == 0)) break;
if (dots == 1) { // Remove this occurrence of "."
segs[i] = -1;
} else { // If there is a preceding non-".." segment, remove both that // segment and this occurrence of ".."; otherwise, leave this // ".." segment as-is. int j; for (j = i - 1; j >= 0; j--) { if (segs[j] != -1) break;
} if (j >= 0) { int q = segs[j]; if (!((path[q] == '.')
&& (path[q + 1] == '.')
&& (path[q + 2] == '\0'))) {
segs[i] = -1;
segs[j] = -1;
}
}
}
}
}
// DEVIATION: If the normalized path is relative, and if the first // segment could be parsed as a scheme name, then prepend a "." segment // privatestaticvoid maybeAddLeadingDot(char[] path, int[] segs) {
if (path[0] == '\0') // The path is absolute return;
int ns = segs.length; int f = 0; // Index of first segment while (f < ns) { if (segs[f] >= 0) break;
f++;
} if ((f >= ns) || (f == 0)) // The path is empty, or else the original first segment survived, // in which case we already know that no leading "." is needed return;
int p = segs[f]; while ((p < path.length) && (path[p] != ':') && (path[p] != '\0')) p++; if (p >= path.length || path[p] == '\0') // No colon in first segment, so no "." needed return;
// At this point we know that the first segment is unused, // hence we can insert a "." segment at that position
path[0] = '.';
path[1] = '\0';
segs[0] = 0;
}
// Normalize the given path string. A normal path string has no empty // segments (i.e., occurrences of "//"), no segments equal to ".", and no // segments equal to ".." that are preceded by a segment not equal to "..". // In contrast to Unix-style pathname normalization, for URI paths we // always retain trailing slashes. // privatestatic String normalize(String ps) {
// Does this path need normalization? int ns = needsNormalization(ps); // Number of segments if (ns < 0) // Nope -- just return it return ps;
char[] path = ps.toCharArray(); // Path in char-array form
// Join the remaining segments and return the result
String s = new String(path, 0, join(path, segs)); if (s.equals(ps)) { // string was already normalized return ps;
} return s;
}
// -- Character classes for parsing --
// RFC2396 precisely specifies which characters in the US-ASCII charset are // permissible in the various components of a URI reference. We here // define a set of mask pairs to aid in enforcing these restrictions. Each // mask pair consists of two longs, a low mask and a high mask. Taken // together they represent a 128-bit mask, where bit i is set iff the // character with value i is permitted. // // This approach is more efficient than sequentially searching arrays of // permitted characters. It could be made still more efficient by // precompiling the mask information so that a character's presence in a // given mask could be determined by a single table lookup.
// To save startup time, we manually calculate the low-/highMask constants. // For reference, the following methods were used to calculate the values:
// Compute the low-order mask for the characters in the given string // private static long lowMask(String chars) { // int n = chars.length(); // long m = 0; // for (int i = 0; i < n; i++) { // char c = chars.charAt(i); // if (c < 64) // m |= (1L << c); // } // return m; // }
// Compute the high-order mask for the characters in the given string // private static long highMask(String chars) { // int n = chars.length(); // long m = 0; // for (int i = 0; i < n; i++) { // char c = chars.charAt(i); // if ((c >= 64) && (c < 128)) // m |= (1L << (c - 64)); // } // return m; // }
// Compute a low-order mask for the characters // between first and last, inclusive // private static long lowMask(char first, char last) { // long m = 0; // int f = Math.max(Math.min(first, 63), 0); // int l = Math.max(Math.min(last, 63), 0); // for (int i = f; i <= l; i++) // m |= 1L << i; // return m; // }
// Compute a high-order mask for the characters // between first and last, inclusive // private static long highMask(char first, char last) { // long m = 0; // int f = Math.max(Math.min(first, 127), 64) - 64; // int l = Math.max(Math.min(last, 127), 64) - 64; // for (int i = f; i <= l; i++) // m |= 1L << i; // return m; // }
// Tell whether the given character is permitted by the given mask pair privatestaticboolean match(char c, long lowMask, long highMask) { if (c == 0) // 0 doesn't have a slot in the mask. So, it never matches. returnfalse; if (c < 64) return ((1L << c) & lowMask) != 0; if (c < 128) return ((1L << (c - 64)) & highMask) != 0; returnfalse;
}
// Character-class masks, in reverse order from RFC2396 because // initializers for static fields cannot make forward references.
// The zero'th bit is used to indicate that escape pairs and non-US-ASCII // characters are allowed; this is handled by the scanEscape method below. privatestaticfinallong L_ESCAPED = 1L; privatestaticfinallong H_ESCAPED = 0L;
// Dash, for use in domainlabel and toplabel privatestaticfinallong L_DASH = 0x200000000000L; // lowMask("-"); privatestaticfinallong H_DASH = 0x0L; // highMask("-");
// Dot, for use in hostnames privatestaticfinallong L_DOT = 0x400000000000L; // lowMask("."); privatestaticfinallong H_DOT = 0x0L; // highMask(".");
// Special case of server authority that represents an IPv6 address // In this case, a % does not signify an escape sequence privatestaticfinallong L_SERVER_PERCENT
= L_SERVER | 0x2000000000L; // lowMask("%"); privatestaticfinallong H_SERVER_PERCENT
= H_SERVER; // | highMask("%") == 0L;
privatestaticvoid appendEncoded(CharsetEncoder encoder, StringBuilder sb, char c) {
ByteBuffer bb = null; try {
bb = encoder.encode(CharBuffer.wrap(newchar[]{c}));
} catch (CharacterCodingException x) { assertfalse;
} while (bb.hasRemaining()) { int b = bb.get() & 0xff; if (b >= 0x80)
appendEscape(sb, (byte)b); else
sb.append((char)b);
}
}
// Quote any characters in s that are not permitted // by the given mask pair // privatestatic String quote(String s, long lowMask, long highMask) {
StringBuilder sb = null;
CharsetEncoder encoder = null; boolean allowNonASCII = ((lowMask & L_ESCAPED) != 0); for (int i = 0; i < s.length(); i++) { char c = s.charAt(i); if (c < '\u0080') { if (!match(c, lowMask, highMask)) { if (sb == null) {
sb = new StringBuilder();
sb.append(s, 0, i);
}
appendEscape(sb, (byte)c);
} else { if (sb != null)
sb.append(c);
}
} elseif (allowNonASCII
&& (Character.isSpaceChar(c)
|| Character.isISOControl(c))) { if (encoder == null)
encoder = UTF_8.INSTANCE.newEncoder(); if (sb == null) {
sb = new StringBuilder();
sb.append(s, 0, i);
}
appendEncoded(encoder, sb, c);
} else { if (sb != null)
sb.append(c);
}
} return (sb == null) ? s : sb.toString();
}
// Encodes all characters >= \u0080 into escaped, normalized UTF-8 octets, // assuming that s is otherwise legal // privatestatic String encode(String s) { int n = s.length(); if (n == 0) return s;
// First check whether we actually need to encode for (int i = 0;;) { if (s.charAt(i) >= '\u0080') break; if (++i >= n) return s;
}
// Evaluates all escapes in s, applying UTF-8 decoding if needed. Assumes // that escapes are well-formed syntactically, i.e., of the form %XX. If a // sequence of escaped octets is not valid UTF-8 then the erroneous octets // are replaced with '\uFFFD'. // Exception: any "%" found between "[]" is left alone. It is an IPv6 literal // with a scope_id // privatestatic String decode(String s) { return decode(s, true);
}
// This method was introduced as a generalization of URI.decode method // to provide a fix for JDK-8037396 privatestatic String decode(String s, boolean ignorePercentInBrackets) { if (s == null) return s; int n = s.length(); if (n == 0) return s; if (s.indexOf('%') < 0) return s;
StringBuilder sb = new StringBuilder(n);
ByteBuffer bb = ByteBuffer.allocate(n);
CharBuffer cb = CharBuffer.allocate(n);
CharsetDecoder dec = UTF_8.INSTANCE.newDecoder()
.onMalformedInput(CodingErrorAction.REPLACE)
.onUnmappableCharacter(CodingErrorAction.REPLACE);
// This is not horribly efficient, but it will do for now char c = s.charAt(0); boolean betweenBrackets = false;
for (int i = 0; i < n;) { assert c == s.charAt(i); // Loop invariant if (c == '[') {
betweenBrackets = true;
} elseif (betweenBrackets && c == ']') {
betweenBrackets = false;
} if (c != '%' || (betweenBrackets && ignorePercentInBrackets)) {
sb.append(c); if (++i >= n) break;
c = s.charAt(i); continue;
}
bb.clear(); for (;;) { assert (n - i >= 2);
bb.put(decode(s.charAt(++i), s.charAt(++i))); if (++i >= n) break;
c = s.charAt(i); if (c != '%') break;
}
bb.flip();
cb.clear();
dec.reset();
CoderResult cr = dec.decode(bb, cb, true); assert cr.isUnderflow();
cr = dec.flush(cb); assert cr.isUnderflow();
sb.append(cb.flip().toString());
}
return sb.toString();
}
// -- Parsing --
// For convenience we wrap the input URI string in a new instance of the // following internal class. This saves always having to pass the input // string as an argument to each internal scan/parse method.
// Tells whether start < end and, if so, whether charAt(start) == c // privateboolean at(int start, int end, char c) { return (start < end) && (input.charAt(start) == c);
}
// Tells whether start + s.length() < end and, if so, // whether the chars at the start position match s exactly // privateboolean at(int start, int end, String s) { int p = start; int sn = s.length(); if (sn > end - p) returnfalse; int i = 0; while (i < sn) { if (input.charAt(p++) != s.charAt(i)) { break;
}
i++;
} return (i == sn);
}
// -- Scanning --
// The various scan and parse methods that follow use a uniform // convention of taking the current start position and end index as // their first two arguments. The start is inclusive while the end is // exclusive, just as in the String class, i.e., a start/end pair // denotes the left-open interval [start, end) of the input string. // // These methods never proceed past the end position. They may return // -1 to indicate outright failure, but more often they simply return // the position of the first char after the last char scanned. Thus // a typical idiom is // // int p = start; // int q = scan(p, end, ...); // if (q > p) // // We scanned something // ...; // else if (q == p) // // We scanned nothing // ...; // else if (q == -1) // // Something went wrong // ...;
// Scan a specific char: If the char at the given start position is // equal to c, return the index of the next char; otherwise, return the // start position. // privateint scan(int start, int end, char c) { if ((start < end) && (input.charAt(start) == c)) return start + 1; return start;
}
// Scan forward from the given start position. Stop at the first char // in the err string (in which case -1 is returned), or the first char // in the stop string (in which case the index of the preceding char is // returned), or the end of the input string (in which case the length // of the input string is returned). May return the start position if // nothing matches. // privateint scan(int start, int end, String err, String stop) { int p = start; while (p < end) { char c = input.charAt(p); if (err.indexOf(c) >= 0) return -1; if (stop.indexOf(c) >= 0) break;
p++;
} return p;
}
// Scan forward from the given start position. Stop at the first char // in the stop string (in which case the index of the preceding char is // returned), or the end of the input string (in which case the length // of the input string is returned). May return the start position if // nothing matches. // privateint scan(int start, int end, String stop) { int p = start; while (p < end) { char c = input.charAt(p); if (stop.indexOf(c) >= 0) break;
p++;
} return p;
}
// Scan a potential escape sequence, starting at the given position, // with the given first char (i.e., charAt(start) == c). // // This method assumes that if escapes are allowed then visible // non-US-ASCII chars are also allowed. // privateint scanEscape(int start, int n, char first) throws URISyntaxException
{ int p = start; char c = first; if (c == '%') { // Process escape pair if ((p + 3 <= n)
&& match(input.charAt(p + 1), L_HEX, H_HEX)
&& match(input.charAt(p + 2), L_HEX, H_HEX)) { return p + 3;
}
fail("Malformed escape pair", p);
} elseif ((c > 128)
&& !Character.isSpaceChar(c)
&& !Character.isISOControl(c)) { // Allow unescaped but visible non-US-ASCII chars return p + 1;
} return p;
}
// Scan chars that match the given mask pair // privateint scan(int start, int n, long lowMask, long highMask) throws URISyntaxException
{ int p = start; while (p < n) { char c = input.charAt(p); if (match(c, lowMask, highMask)) {
p++; continue;
} if ((lowMask & L_ESCAPED) != 0) { int q = scanEscape(p, n, c); if (q > p) {
p = q; continue;
}
} break;
} return p;
}
// Check that each of the chars in [start, end) matches the given mask // privatevoid checkChars(int start, int end, long lowMask, long highMask,
String what) throws URISyntaxException
{ int p = scan(start, end, lowMask, highMask); if (p < end)
fail("Illegal character in " + what, p);
}
// Check that the char at position p matches the given mask // privatevoid checkChar(int p, long lowMask, long highMask,
String what) throws URISyntaxException
{
checkChars(p, p + 1, lowMask, highMask, what);
}
// -- Parsing --
// [<scheme>:]<scheme-specific-part>[#<fragment>] // void parse(boolean rsa) throws URISyntaxException {
requireServerAuthority = rsa; int n = input.length(); int p = scan(0, n, "/?#", ":"); if ((p >= 0) && at(p, n, ':')) { if (p == 0)
failExpecting("scheme name", 0);
checkChar(0, L_ALPHA, H_ALPHA, "scheme name");
checkChars(1, p, L_SCHEME, H_SCHEME, "scheme name");
scheme = input.substring(0, p);
p++; // Skip ':' if (at(p, n, '/')) {
p = parseHierarchical(p, n);
} else { // opaque; need to create the schemeSpecificPart int q = scan(p, n, "#"); if (q <= p)
failExpecting("scheme-specific part", p);
checkChars(p, q, L_URIC, H_URIC, "opaque part");
schemeSpecificPart = input.substring(p, q);
p = q;
}
} else {
p = parseHierarchical(0, n);
} if (at(p, n, '#')) {
checkChars(p + 1, n, L_URIC, H_URIC, "fragment");
fragment = input.substring(p + 1, n);
p = n;
} if (p < n)
fail("end of URI", p);
}
// [//authority]<path>[?<query>] // // DEVIATION from RFC2396: We allow an empty authority component as // long as it's followed by a non-empty path, query component, or // fragment component. This is so that URIs such as "file:///foo/bar" // will parse. This seems to be the intent of RFC2396, though the // grammar does not permit it. If the authority is empty then the // userInfo, host, and port components are undefined. // // DEVIATION from RFC2396: We allow empty relative paths. This seems // to be the intent of RFC2396, but the grammar does not permit it. // The primary consequence of this deviation is that "#f" parses as a // relative URI with an empty path. // privateint parseHierarchical(int start, int n) throws URISyntaxException
{ int p = start; if (at(p, n, '/') && at(p + 1, n, '/')) {
p += 2; int q = scan(p, n, "/?#"); if (q > p) {
p = parseAuthority(p, q);
} elseif (q < n) { // DEVIATION: Allow empty authority prior to non-empty // path, query component or fragment identifier
} else
failExpecting("authority", p);
} int q = scan(p, n, "?#"); // DEVIATION: May be empty
checkChars(p, q, L_PATH, H_PATH, "path");
path = input.substring(p, q);
p = q; if (at(p, n, '?')) {
p++;
q = scan(p, n, "#");
checkChars(p, q, L_URIC, H_URIC, "query");
query = input.substring(p, q);
p = q;
} return p;
}
// authority = server | reg_name // // Ambiguity: An authority that is a registry name rather than a server // might have a prefix that parses as a server. We use the fact that // the authority component is always followed by '/' or the end of the // input string to resolve this: If the complete authority did not // parse as a server then we try to parse it as a registry name. // privateint parseAuthority(int start, int n) throws URISyntaxException
{ int p = start; int q = p; int qreg = p;
URISyntaxException ex = null;
boolean serverChars; boolean regChars;
if (scan(p, n, "]") > p) { // contains a literal IPv6 address, therefore % is allowed
serverChars = (scan(p, n, L_SERVER_PERCENT, H_SERVER_PERCENT) == n);
} else {
serverChars = (scan(p, n, L_SERVER, H_SERVER) == n);
}
regChars = ((qreg = scan(p, n, L_REG_NAME, H_REG_NAME)) == n);
if (regChars && !serverChars) { // Must be a registry-based authority
authority = input.substring(p, n); return n;
}
if (serverChars) { // Might be (probably is) a server-based authority, so attempt // to parse it as such. If the attempt fails, try to treat it // as a registry-based authority. try {
q = parseServer(p, n); if (q < n)
failExpecting("end of authority", q);
authority = input.substring(p, n);
} catch (URISyntaxException x) { // Undo results of failed parse
userInfo = null;
host = null;
port = -1; if (requireServerAuthority) { // If we're insisting upon a server-based authority, // then just re-throw the exception throw x;
} else { // Save the exception in case it doesn't parse as a // registry either
ex = x;
q = p;
}
}
}
if (q < n) { if (regChars) { // Registry-based authority
authority = input.substring(p, n);
} elseif (ex != null) { // Re-throw exception; it was probably due to // a malformed IPv6 address throw ex;
} else {
fail("Illegal character in authority", serverChars ? q : qreg);
}
}
return n;
}
// [<userinfo>@]<host>[:<port>] // privateint parseServer(int start, int n) throws URISyntaxException
{ int p = start; int q;
// userinfo
q = scan(p, n, "/?#", "@"); if ((q >= p) && at(q, n, '@')) {
checkChars(p, q, L_USERINFO, H_USERINFO, "user info");
userInfo = input.substring(p, q);
p = q + 1; // Skip '@'
}
// hostname, IPv4 address, or IPv6 address if (at(p, n, '[')) { // DEVIATION from RFC2396: Support IPv6 addresses, per RFC2732
p++;
q = scan(p, n, "/?#", "]"); if ((q > p) && at(q, n, ']')) { // look for a "%" scope id int r = scan (p, q, "%"); if (r > p) {
parseIPv6Reference(p, r); if (r+1 == q) {
fail ("scope id expected");
}
checkChars (r+1, q, L_SCOPE_ID, H_SCOPE_ID, "scope id");
} else {
parseIPv6Reference(p, q);
}
host = input.substring(p-1, q+1);
p = q + 1;
} else {
failExpecting("closing bracket for IPv6 address", q);
}
} else {
q = parseIPv4Address(p, n); if (q <= p)
q = parseHostname(p, n);
p = q;
}
// port if (at(p, n, ':')) {
p++;
q = scan(p, n, "/"); if (q > p) {
checkChars(p, q, L_DIGIT, H_DIGIT, "port number"); try {
port = Integer.parseInt(input, p, q, 10);
} catch (NumberFormatException x) {
fail("Malformed port number", p);
}
p = q;
}
} if (p < n)
failExpecting("port number", p);
return p;
}
// Scan a string of decimal digits whose value fits in a byte // privateint scanByte(int start, int n) throws URISyntaxException
{ int p = start; int q = scan(p, n, L_DIGIT, H_DIGIT); if (q <= p) return q; if (Integer.parseInt(input, p, q, 10) > 255) return p; return q;
}
// Scan an IPv4 address. // // If the strict argument is true then we require that the given // interval contain nothing besides an IPv4 address; if it is false // then we only require that it start with an IPv4 address. // // If the interval does not contain or start with (depending upon the // strict argument) a legal IPv4 address characters then we return -1 // immediately; otherwise we insist that these characters parse as a // legal IPv4 address and throw an exception on failure. // // We assume that any string of decimal digits and dots must be an IPv4 // address. It won't parse as a hostname anyway, so making that // assumption here allows more meaningful exceptions to be thrown. // privateint scanIPv4Address(int start, int n, boolean strict) throws URISyntaxException
{ int p = start; int q; int m = scan(p, n, L_DIGIT | L_DOT, H_DIGIT | H_DOT); if ((m <= p) || (strict && (m != n))) return -1; for (;;) { // Per RFC2732: At most three digits per byte // Further constraint: Each element fits in a byte if ((q = scanByte(p, m)) <= p) break; p = q; if ((q = scan(p, m, '.')) <= p) break; p = q; if ((q = scanByte(p, m)) <= p) break; p = q; if ((q = scan(p, m, '.')) <= p) break; p = q; if ((q = scanByte(p, m)) <= p) break; p = q; if ((q = scan(p, m, '.')) <= p) break; p = q; if ((q = scanByte(p, m)) <= p) break; p = q; if (q < m) break; return q;
}
fail("Malformed IPv4 address", q); return -1;
}
// Take an IPv4 address: Throw an exception if the given interval // contains anything except an IPv4 address // privateint takeIPv4Address(int start, int n, String expected) throws URISyntaxException
{ int p = scanIPv4Address(start, n, true); if (p <= start)
failExpecting(expected, start); return p;
}
// Attempt to parse an IPv4 address, returning -1 on failure but // allowing the given interval to contain [:<characters>] after // the IPv4 address. // privateint parseIPv4Address(int start, int n) { int p;
try {
p = scanIPv4Address(start, n, false);
} catch (URISyntaxException | NumberFormatException x) { return -1;
}
if (p > start && p < n) { // IPv4 address is followed by something - check that // it's a ":" as this is the only valid character to // follow an address. if (input.charAt(p) != ':') {
p = -1;
}
}
if (p > start)
host = input.substring(start, p);
return p;
}
// hostname = domainlabel [ "." ] | 1*( domainlabel "." ) toplabel [ "." ] // domainlabel = alphanum | alphanum *( alphanum | "-" ) alphanum // toplabel = alpha | alpha *( alphanum | "-" ) alphanum // privateint parseHostname(int start, int n) throws URISyntaxException
{ int p = start; int q; int l = -1; // Start of last parsed label
do { // domainlabel = alphanum [ *( alphanum | "-" ) alphanum ]
q = scan(p, n, L_ALPHANUM, H_ALPHANUM); if (q <= p) break;
l = p;
p = q;
q = scan(p, n, L_ALPHANUM | L_DASH, H_ALPHANUM | H_DASH); if (q > p) { if (input.charAt(q - 1) == '-')
fail("Illegal character in hostname", q - 1);
p = q;
}
q = scan(p, n, '.'); if (q <= p) break;
p = q;
} while (p < n);
if ((p < n) && !at(p, n, ':'))
fail("Illegal character in hostname", p);
if (l < 0)
failExpecting("hostname", start);
// for a fully qualified hostname check that the rightmost // label starts with an alpha character. if (l > start && !match(input.charAt(l), L_ALPHA, H_ALPHA)) {
fail("Illegal character in hostname", l);
}
host = input.substring(start, p); return p;
}
// IPv6 address parsing, from RFC2373: IPv6 Addressing Architecture // // Bug: The grammar in RFC2373 Appendix B does not allow addresses of // the form ::12.34.56.78, which are clearly shown in the examples // earlier in the document. Here is the original grammar: // // IPv6address = hexpart [ ":" IPv4address ] // hexpart = hexseq | hexseq "::" [ hexseq ] | "::" [ hexseq ] // hexseq = hex4 *( ":" hex4) // hex4 = 1*4HEXDIG // // We therefore use the following revised grammar: // // IPv6address = hexseq [ ":" IPv4address ] // | hexseq [ "::" [ hexpost ] ] // | "::" [ hexpost ] // hexpost = hexseq | hexseq ":" IPv4address | IPv4address // hexseq = hex4 *( ":" hex4) // hex4 = 1*4HEXDIG // // This covers all and only the following cases: // // hexseq // hexseq : IPv4address // hexseq :: // hexseq :: hexseq // hexseq :: hexseq : IPv4address // hexseq :: IPv4address // :: hexseq // :: hexseq : IPv4address // :: IPv4address // :: // // Additionally we constrain the IPv6 address as follows :- // // i. IPv6 addresses without compressed zeros should contain // exactly 16 bytes. // // ii. IPv6 addresses with compressed zeros should contain // less than 16 bytes.
privateint ipv6byteCount = 0;
privateint parseIPv6Reference(int start, int n) throws URISyntaxException
{ int p = start; int q; boolean compressedZeros = false;
q = scanHexSeq(p, n);
if (q > p) {
p = q; if (at(p, n, "::")) {
compressedZeros = true;
p = scanHexPost(p + 2, n);
} elseif (at(p, n, ':')) {
p = takeIPv4Address(p + 1, n, "IPv4 address");
ipv6byteCount += 4;
}
} elseif (at(p, n, "::")) {
compressedZeros = true;
p = scanHexPost(p + 2, n);
} if (p < n)
fail("Malformed IPv6 address", start); if (ipv6byteCount > 16)
fail("IPv6 address too long", start); if (!compressedZeros && ipv6byteCount < 16)
fail("IPv6 address too short", start); if (compressedZeros && ipv6byteCount == 16)
fail("Malformed IPv6 address", start);
return p;
}
privateint scanHexPost(int start, int n) throws URISyntaxException
{ int p = start; int q;
if (p == n) return p;
q = scanHexSeq(p, n); if (q > p) {
p = q; if (at(p, n, ':')) {
p++;
p = takeIPv4Address(p, n, "hex digits or IPv4 address");
ipv6byteCount += 4;
}
} else {
p = takeIPv4Address(p, n, "hex digits or IPv4 address");
ipv6byteCount += 4;
} return p;
}
// Scan a hex sequence; return -1 if one could not be scanned // privateint scanHexSeq(int start, int n) throws URISyntaxException
{ int p = start; int q;
q = scan(p, n, L_HEX, H_HEX); if (q <= p) return -1; if (at(q, n, '.')) // Beginning of IPv4 address return -1; if (q > p + 4)
fail("IPv6 hexadecimal digit sequence too long", p);
ipv6byteCount += 2;
p = q; while (p < n) { if (!at(p, n, ':')) break; if (at(p + 1, n, ':')) break; // "::"
p++;
q = scan(p, n, L_HEX, H_HEX); if (q <= p)
failExpecting("digits for an IPv6 address", p); if (at(q, n, '.')) { // Beginning of IPv4 address
p--; break;
} if (q > p + 4)
fail("IPv6 hexadecimal digit sequence too long", p);
ipv6byteCount += 2;
p = q;
}
return p;
}
} static {
SharedSecrets.setJavaNetUriAccess( new JavaNetUriAccess() { public URI create(String scheme, String path) { returnnew URI(scheme, path);
}
}
);
}
}
Messung V0.5 in Prozent
¤ Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.0.535Bemerkung:
(vorverarbeitet am 2026-10-06)
¤
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.