vyommani commented on code in PR #1255: URL: https://github.com/apache/ranger/pull/1255#discussion_r4101554114
########## agents-common/src/main/java/org/apache/ranger/plugin/client/JdbcUrlValidator.java: ########## @@ -0,0 +1,237 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.ranger.plugin.client; + +import org.slf4j.Logger; +import org.slf4j.LoggerFactory; + +import java.net.URLDecoder; +import java.nio.charset.StandardCharsets; +import java.util.Arrays; +import java.util.Collection; +import java.util.Collections; +import java.util.HashSet; +import java.util.Set; + +public final class JdbcUrlValidator { + private static final Logger LOG = LoggerFactory.getLogger(JdbcUrlValidator.class); + private static final Set<String> BLOCKED_PARAMS = Collections.unmodifiableSet( + new HashSet<>(Arrays.asList( + "socketfactory", "socketfactoryarg", "sslfactory", "sslfactoryarg", + "sslhostnameverifier", "authenticationpluginclassname", "loggerclassname", + "kerberosservername", "gssdelegatecred", "sslpasswordcallback", "dnsresolver"))); + private static final String[] DANGEROUS_PATTERNS = {"socketfactory", "sslfactory", "autodeserialize"}; + + private JdbcUrlValidator() { + } + + static void validate(String jdbcUrl) throws HadoopException { + if (jdbcUrl == null || jdbcUrl.trim().isEmpty()) { + HadoopException e = new HadoopException("jdbc.url must not be null or empty"); + e.generateResponseDataMap(false, "Validation failed", "jdbc.url is required", + null, "jdbc.url"); + throw e; + } + String trimmed = jdbcUrl.trim(); + rejectBlockedParameters(trimmed, trimmed); + // Hive decodes percent-escapes before splitting session variables, so a + // separator written as %3B or %3F is invisible to a scan of the raw URL. + if (trimmed.indexOf('%') >= 0) { + rejectBlockedParameters(percentDecode(trimmed), trimmed); + } + LOG.debug("jdbc.url passed validation: {}", sanitizeForLog(trimmed)); + } + + /** + * Validates jdbc.url and requires it to start with one of the given prefixes, + * followed by a host. The prefix check uses the trimmed URL. Callers must pass + * that same trimmed string to DriverManager. + */ + public static void validate(String jdbcUrl, Collection<String> allowedUrlPrefixes) throws HadoopException { + validate(jdbcUrl); + + String candidate = jdbcUrl.trim(); + String matchedPrefix = null; + + if (allowedUrlPrefixes != null) { + for (String prefix : allowedUrlPrefixes) { + if (prefix != null && !prefix.isEmpty() && candidate.startsWith(prefix)) { + matchedPrefix = prefix; + break; + } + } + } + + if (matchedPrefix == null) { + LOG.warn("Rejected jdbc.url with unsupported scheme: {}", sanitizeForLog(candidate)); + + HadoopException e = new HadoopException("jdbc.url must start with one of " + allowedUrlPrefixes); + e.generateResponseDataMap(false, "Invalid jdbc.url", "jdbc.url must start with one of " + allowedUrlPrefixes, null, "jdbc.url"); + throw e; + } + + requireHost(candidate, matchedPrefix); + } + + public static void validateDriverClassName(String driverClassName, Collection<String> allowedDriverClassNames) throws HadoopException { + // Null skips registration. DriverManager then picks a driver that accepts the URL, + // which the prefix and host checks already constrained. + if (driverClassName == null) { + return; + } + + if (allowedDriverClassNames == null || !allowedDriverClassNames.contains(driverClassName)) { + LOG.warn("Rejected jdbc.driverClassName not in allowed list {}", allowedDriverClassNames); + + HadoopException e = new HadoopException("jdbc.driverClassName must be one of " + allowedDriverClassNames); + e.generateResponseDataMap(false, "Invalid jdbc.driverClassName", "jdbc.driverClassName must be one of " + allowedDriverClassNames, null, "jdbc.driverClassName"); + throw e; + } + } + + /** + * An empty host is Hive embedded mode (jdbc:hive2://, jdbc:hive2:///, jdbc:hive2://;...). + * That starts HiveServer2 inside the Admin JVM. The same host requirement applies to + * every allowed prefix. + */ + private static void requireHost(String url, String prefix) throws HadoopException { + boolean missingHost = url.length() == prefix.length(); + + if (!missingHost) { + char next = url.charAt(prefix.length()); + + missingHost = next == '/' || next == ';' || next == '?' || next == '#'; Review Comment: requireHost() did let jdbc:hive2://:10000/... through, because : was not treated as a missing host. The drivers already reject it: hive-jdbc 3.1.3 and 4.0.1 throw JdbcUriParseException ("Hostname not found") since URI.getHost() is null, and that path is not embedded mode. Trino 451 and Presto 333 throw "No host specified". The validator now rejects an empty host token itself, including :10000, %3A10000, user:pass@:10000, and an empty member of a ZooKeeper list. host:10000, [::1]:10000, and zk1:2181,zk2:2181 still pass. -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected]
