laskoviymishka commented on code in PR #3111: URL: https://github.com/apache/iceberg-rust/pull/3111#discussion_r4196324978
########## crates/storage/opendal/src/hdfs_native.rs: ########## @@ -0,0 +1,928 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! HDFS storage backend via OpenDAL's `services-hdfs-native` (pure Rust, no JNI). + +use std::collections::HashMap; +use std::collections::hash_map::Entry; +use std::sync::{Arc, RwLock, Weak}; + +use iceberg::io::{HDFS_HADOOP_CONF_PREFIX, HDFS_HOST, HDFS_NAME_NODE, HDFS_PORT}; +use iceberg::{Error, ErrorKind, Result}; +use opendal::Operator; +use opendal::services::HdfsNativeConfig; +use serde::{Deserialize, Serialize}; +use tokio::runtime::Handle; +use tokio::task::JoinHandle; +use url::Url; + +use crate::OpenDalClientConfig; +use crate::utils::from_opendal_error; + +/// Hadoop's default filesystem, which serves authority-less paths. +const FS_DEFAULT_FS: &str = "fs.defaultFS"; +const HDFS_DEFAULT_PORT: u16 = 8020; +/// PyIceberg keys with no equivalent in opendal's config. +const HDFS_UNSUPPORTED_KEYS: [&str; 2] = ["hdfs.user", "hdfs.kerberos_ticket"]; + +/// `hdfs-native` dials a NameNode as a socket address and has no default +/// port, so anything without one can only be a logical nameservice name. +fn hdfs_native_has_port(name_node: &str) -> bool { + name_node + .rsplit_once(':') + .is_some_and(|(_, port)| port.parse::<u16>().is_ok()) +} + +/// Parse iceberg properties to [`HdfsNativeConfig`]. +pub(crate) fn hdfs_native_config_parse(mut m: HashMap<String, String>) -> Result<HdfsNativeConfig> { + let mut cfg = HdfsNativeConfig::default(); + + // Entries are trimmed one by one: opendal splits the list on `,` as is, + // so a space after a comma would break failover to that NameNode. An + // empty result is dropped because `Operator::from_config` bypasses the + // builder's empty-string guard and `Some("")` would shadow the + // path-authority fallback below. + if let Some(name_node) = m.remove(HDFS_NAME_NODE) { + let entries: Vec<&str> = name_node + .split(',') + .map(|entry| entry.trim().trim_end_matches('/')) + .filter(|entry| !entry.is_empty()) + .collect(); + // Each entry is dialed as `host:port` (`hdfs://` optional); a portless + // one would fail only at the first I/O. + if let Some(entry) = entries.iter().find(|entry| !hdfs_native_has_port(entry)) { + return Err(Error::new( + ErrorKind::DataInvalid, + format!("Invalid `{HDFS_NAME_NODE}` entry: {entry}, expected host:port"), + )); + } + if !entries.is_empty() { + cfg.name_node = Some(entries.join(",")); + } + } + + // A config carried over from PyIceberg would otherwise change identity + // silently; the client reads `HADOOP_USER_NAME` and the Kerberos cache. + for key in HDFS_UNSUPPORTED_KEYS { + if m.remove(key).is_some() { + tracing::warn!("`{key}` is not supported by the hdfs-native backend and is ignored"); + } + } + if m.contains_key(HDFS_HADOOP_CONF_PREFIX) { + return Err(Error::new( + ErrorKind::DataInvalid, + format!( + "Invalid property `{HDFS_HADOOP_CONF_PREFIX}`: a Hadoop key must follow the prefix" + ), + )); + } + + let host = m + .remove(HDFS_HOST) + .map(|s| s.trim().to_string()) + .filter(|s| !s.is_empty()); + let port = m + .remove(HDFS_PORT) + .map(|s| s.trim().to_string()) + .filter(|s| !s.is_empty()) + .map(|port| { + port.parse::<u16>().map_err(|e| { + Error::new( + ErrorKind::DataInvalid, + format!("Invalid `{HDFS_PORT}`: {port}: {e}"), + ) + }) + }) + .transpose()?; + + let mut options: HashMap<String, String> = m + .into_iter() + .filter_map(|(key, value)| { + key.strip_prefix(HDFS_HADOOP_CONF_PREFIX) + .map(|stripped| (stripped.to_string(), value)) + }) + .collect(); + // PyIceberg's `hdfs.host`/`hdfs.port` name the filesystem for + // authority-less paths, which is what Hadoop's `fs.defaultFS` means; an + // explicit `hadoop.fs.defaultFS` wins. + match host { + Some(host) => { + // An IPv6 literal needs brackets in a URI authority. + let host = if host.contains(':') && !host.starts_with('[') { + format!("[{host}]") + } else { + host + }; + let port = port.unwrap_or(HDFS_DEFAULT_PORT); + options + .entry(FS_DEFAULT_FS.to_string()) + .or_insert_with(|| format!("hdfs://{host}:{port}")); + } + None if port.is_some() => { + tracing::warn!("`{HDFS_PORT}` has no effect without `{HDFS_HOST}` and is ignored"); + } + None => {} + } + if !options.is_empty() { + cfg.options = Some(options); + } + + Ok(cfg) +} + +/// Parse an HDFS path into `Some("hdfs://<authority>")` (`None` when +/// authority-less) and the relative path (no leading `/`, opendal style). +pub(crate) fn hdfs_native_parse_path(path: &str) -> Result<(Option<String>, &str)> { + let url = Url::parse(path).map_err(|e| { + Error::new( + ErrorKind::DataInvalid, + format!("Invalid hdfs path: {path}: {e}"), + ) + })?; + // Non-special schemes parse even without `//` (e.g. `hdfs:x` is a valid + // non-hierarchical URL), so require the literal prefix before slicing. + let (Some(after_scheme), "hdfs") = (path.strip_prefix("hdfs://"), url.scheme()) else { + return Err(Error::new( + ErrorKind::DataInvalid, + format!("Invalid hdfs path: {path}, expected scheme `hdfs://`"), + )); + }; + + let name_node = url.host_str().filter(|h| !h.is_empty()).map(|host| { + url.port() + .map(|port| format!("hdfs://{host}:{port}")) + .unwrap_or_else(|| format!("hdfs://{host}")) + }); + + // `url.path()` borrows from `url` and can't be returned with the input's + // lifetime. Slice the path component out of the original input instead; + // it starts after the first `/` following the `hdfs://` prefix. Opendal + // paths must not start with `/` (`Deleter::delete` rejects them). + let rel = match after_scheme.find('/') { + Some(i) => after_scheme[i..].trim_start_matches('/'), + None => "", + }; + + Ok((name_node, rel)) +} + +/// Resolves the effective NameNode for a path, plus the relative path. As in +/// Hadoop, an authority with a port is used as is; a logical nameservice +/// authority (no port) resolves through `hdfs.name-node`, and an +/// authority-less path through `hdfs.name-node`, else `fs.defaultFS`. The +/// operator cache, `delete_stream` batching and `relativize_path` all go +/// through this, so they cannot drift apart. +pub(crate) fn hdfs_native_effective_name_node<'a>( + config: &HdfsNativeConfig, + path: &'a str, +) -> Result<(String, &'a str)> { + let (authority, relative_path) = hdfs_native_parse_path(path)?; + let invalid = |reason: String| { + Error::new( + ErrorKind::DataInvalid, + format!("Invalid hdfs path: {path}, {reason}"), + ) + }; + let name_node = match authority { + Some(authority) if hdfs_native_has_port(&authority) => authority, + Some(logical) => config.name_node.clone().ok_or_else(|| { + let logical = logical.trim_start_matches("hdfs://"); + invalid(format!( + "logical nameservice `{logical}` requires `{HDFS_NAME_NODE}`" + )) + })?, + None => match config + .name_node + .clone() + .or_else(|| hdfs_native_default_fs(config)) + { + Some(name_node) if hdfs_native_has_port(&name_node) => name_node, + Some(logical) => { + return Err(invalid(format!( + "`{FS_DEFAULT_FS}` {logical} has no port, a logical nameservice requires `{HDFS_NAME_NODE}`" + ))); + } + None => { + return Err(invalid(format!( + "authority-less paths require `{HDFS_NAME_NODE}` or `{HDFS_HOST}`" + ))); + } + }, + }; + Ok((name_node, relative_path)) +} + +/// `fs.defaultFS` from the forwarded options, when it is an HDFS URI. +fn hdfs_native_default_fs(config: &HdfsNativeConfig) -> Option<String> { + config + .options + .as_ref()? + .get(FS_DEFAULT_FS) + .map(|s| s.trim().trim_end_matches('/')) + .filter(|s| { + s.strip_prefix("hdfs://") + .is_some_and(|rest| !rest.is_empty()) + }) + .map(str::to_string) +} + +/// State of [`OpenDalStorage::HdfsNative`](crate::OpenDalStorage::HdfsNative): +/// the parsed configuration and the per-NameNode operator cache. Only the +/// storage factories build it. +#[derive(Clone, Debug, Serialize, Deserialize)] +pub struct HdfsNativeStorage { + pub(crate) config: Arc<HdfsNativeConfig>, + #[serde(skip, default)] + pub(crate) operators: HdfsNativeOperatorCache, + #[serde(default)] + pub(crate) client_config: OpenDalClientConfig, +} + +impl HdfsNativeStorage { + pub(crate) fn new(config: HdfsNativeConfig, client_config: OpenDalClientConfig) -> Self { + Self { + config: Arc::new(config), + operators: HdfsNativeOperatorCache::default(), + client_config, + } + } +} + +/// Operators cached per effective NameNode: each holds an `hdfs-native` +/// client with live RPC connections, whose tasks run on the tokio runtime +/// current when it was built (a private one when built outside any). An +/// entry is rebuilt once that runtime is gone, as `hdfs-native` panics when +/// it spawns onto a dead one. The cache lives as long as the storage that +/// owns it (clones share it). +#[derive(Clone, Debug, Default)] +pub(crate) struct HdfsNativeOperatorCache(Arc<RwLock<HashMap<String, CachedOperator>>>); + +#[derive(Debug)] +struct CachedOperator { + operator: Operator, + /// `None` when built outside any runtime. + sentinel: Option<RuntimeSentinel>, +} + +/// A task parked on the building runtime that owns the token, so the token +/// outlives it only while that runtime is alive. Aborted on drop so entries +/// do not leave parked tasks behind. +#[derive(Debug)] +struct RuntimeSentinel { + alive: Weak<()>, + task: JoinHandle<()>, +} + +impl RuntimeSentinel { + fn spawn(handle: &Handle) -> Self { + let token = Arc::new(()); + let alive = Arc::downgrade(&token); + let task = handle.spawn(async move { + let _token = token; + std::future::pending::<()>().await + }); + Self { alive, task } + } +} + +impl Drop for RuntimeSentinel { + fn drop(&mut self) { + self.task.abort(); + } +} + +impl CachedOperator { + fn new(operator: Operator) -> Self { + let sentinel = Handle::try_current() + .ok() + .map(|handle| RuntimeSentinel::spawn(&handle)); + Self { operator, sentinel } + } + + fn runtime_alive(&self) -> bool { + self.sentinel + .as_ref() + .is_none_or(|sentinel| sentinel.alive.strong_count() > 0) + } +} + +impl HdfsNativeOperatorCache { + pub(crate) fn get(&self, name_node: &str) -> Result<Option<Operator>> { + Ok(self + .0 + .read() + .map_err(poisoned)? + .get(name_node) + .filter(|cached| cached.runtime_alive()) + .map(|cached| cached.operator.clone())) + } + + /// Inserts `op` unless a concurrent caller got there first, returning + /// whichever operator the cache now holds; a stale entry is replaced. + fn insert(&self, name_node: String, op: Operator) -> Result<Operator> { + let mut operators = self.0.write().map_err(poisoned)?; + match operators.entry(name_node) { + Entry::Occupied(entry) if entry.get().runtime_alive() => { + Ok(entry.get().operator.clone()) + } + Entry::Occupied(mut entry) => { + entry.insert(CachedOperator::new(op.clone())); + Ok(op) + } + Entry::Vacant(entry) => { + entry.insert(CachedOperator::new(op.clone())); + Ok(op) + } + } + } + + #[cfg(test)] + fn len(&self) -> usize { + self.0.read().unwrap().len() + } +} + +fn poisoned<T>(_: T) -> Error { + Error::new(ErrorKind::Unexpected, "HDFS operator cache lock poisoned") +} + +/// Creates an operator for the path, reusing the cached one for its +/// effective NameNode. +pub(crate) async fn hdfs_native_create_operator<'a>( + path: &'a str, + config: &Arc<HdfsNativeConfig>, + operators: &HdfsNativeOperatorCache, +) -> Result<(Operator, &'a str)> { + let (name_node, relative_path) = hdfs_native_effective_name_node(config, path)?; + + if let Some(op) = operators.get(&name_node)? { + return Ok((op, relative_path)); + } + + // The build reads the Hadoop XML config synchronously, so it runs on a + // blocking thread and outside the lock. A racing first caller may build + // too; the loser is dropped before opening any connection. + let build_config = Arc::clone(config); + let build_name_node = name_node.clone(); + let op = tokio::task::spawn_blocking(move || { Review Comment: I'd gate this on `Handle::try_current()` — use `spawn_blocking` when there's a runtime, otherwise build inline (or return `FeatureUnsupported`) — because `spawn_blocking` panics outside a tokio runtime, where the rest of the crate stays executor-agnostic. So this fixes the executor-blocking I flagged last round but swaps it for a hard crash under `futures::executor::block_on` / smol, which is the kind of panic the crate otherwise avoids. It also leaves dead code: `CachedOperator::new`'s `sentinel: None` arm and the "a private one when built outside any" comment can't be reached on this path anymore — only the direct-insert test hits them. Worth dropping that branch if it's genuinely unreachable. ########## crates/storage/opendal/src/hdfs_native.rs: ########## @@ -0,0 +1,928 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! HDFS storage backend via OpenDAL's `services-hdfs-native` (pure Rust, no JNI). + +use std::collections::HashMap; +use std::collections::hash_map::Entry; +use std::sync::{Arc, RwLock, Weak}; + +use iceberg::io::{HDFS_HADOOP_CONF_PREFIX, HDFS_HOST, HDFS_NAME_NODE, HDFS_PORT}; +use iceberg::{Error, ErrorKind, Result}; +use opendal::Operator; +use opendal::services::HdfsNativeConfig; +use serde::{Deserialize, Serialize}; +use tokio::runtime::Handle; +use tokio::task::JoinHandle; +use url::Url; + +use crate::OpenDalClientConfig; +use crate::utils::from_opendal_error; + +/// Hadoop's default filesystem, which serves authority-less paths. +const FS_DEFAULT_FS: &str = "fs.defaultFS"; +const HDFS_DEFAULT_PORT: u16 = 8020; +/// PyIceberg keys with no equivalent in opendal's config. +const HDFS_UNSUPPORTED_KEYS: [&str; 2] = ["hdfs.user", "hdfs.kerberos_ticket"]; + +/// `hdfs-native` dials a NameNode as a socket address and has no default +/// port, so anything without one can only be a logical nameservice name. +fn hdfs_native_has_port(name_node: &str) -> bool { + name_node + .rsplit_once(':') + .is_some_and(|(_, port)| port.parse::<u16>().is_ok()) +} + +/// Parse iceberg properties to [`HdfsNativeConfig`]. +pub(crate) fn hdfs_native_config_parse(mut m: HashMap<String, String>) -> Result<HdfsNativeConfig> { + let mut cfg = HdfsNativeConfig::default(); + + // Entries are trimmed one by one: opendal splits the list on `,` as is, + // so a space after a comma would break failover to that NameNode. An + // empty result is dropped because `Operator::from_config` bypasses the + // builder's empty-string guard and `Some("")` would shadow the + // path-authority fallback below. + if let Some(name_node) = m.remove(HDFS_NAME_NODE) { + let entries: Vec<&str> = name_node + .split(',') + .map(|entry| entry.trim().trim_end_matches('/')) + .filter(|entry| !entry.is_empty()) + .collect(); + // Each entry is dialed as `host:port` (`hdfs://` optional); a portless + // one would fail only at the first I/O. + if let Some(entry) = entries.iter().find(|entry| !hdfs_native_has_port(entry)) { + return Err(Error::new( + ErrorKind::DataInvalid, + format!("Invalid `{HDFS_NAME_NODE}` entry: {entry}, expected host:port"), + )); + } + if !entries.is_empty() { + cfg.name_node = Some(entries.join(",")); + } + } + + // A config carried over from PyIceberg would otherwise change identity + // silently; the client reads `HADOOP_USER_NAME` and the Kerberos cache. + for key in HDFS_UNSUPPORTED_KEYS { + if m.remove(key).is_some() { + tracing::warn!("`{key}` is not supported by the hdfs-native backend and is ignored"); + } + } + if m.contains_key(HDFS_HADOOP_CONF_PREFIX) { + return Err(Error::new( + ErrorKind::DataInvalid, + format!( + "Invalid property `{HDFS_HADOOP_CONF_PREFIX}`: a Hadoop key must follow the prefix" + ), + )); + } + + let host = m + .remove(HDFS_HOST) + .map(|s| s.trim().to_string()) + .filter(|s| !s.is_empty()); + let port = m + .remove(HDFS_PORT) + .map(|s| s.trim().to_string()) + .filter(|s| !s.is_empty()) + .map(|port| { + port.parse::<u16>().map_err(|e| { + Error::new( + ErrorKind::DataInvalid, + format!("Invalid `{HDFS_PORT}`: {port}: {e}"), + ) + }) + }) + .transpose()?; + + let mut options: HashMap<String, String> = m + .into_iter() + .filter_map(|(key, value)| { + key.strip_prefix(HDFS_HADOOP_CONF_PREFIX) + .map(|stripped| (stripped.to_string(), value)) + }) + .collect(); + // PyIceberg's `hdfs.host`/`hdfs.port` name the filesystem for + // authority-less paths, which is what Hadoop's `fs.defaultFS` means; an + // explicit `hadoop.fs.defaultFS` wins. + match host { + Some(host) => { + // An IPv6 literal needs brackets in a URI authority. + let host = if host.contains(':') && !host.starts_with('[') { + format!("[{host}]") + } else { + host + }; + let port = port.unwrap_or(HDFS_DEFAULT_PORT); + options + .entry(FS_DEFAULT_FS.to_string()) + .or_insert_with(|| format!("hdfs://{host}:{port}")); + } + None if port.is_some() => { + tracing::warn!("`{HDFS_PORT}` has no effect without `{HDFS_HOST}` and is ignored"); + } + None => {} + } + if !options.is_empty() { + cfg.options = Some(options); + } + + Ok(cfg) +} + +/// Parse an HDFS path into `Some("hdfs://<authority>")` (`None` when +/// authority-less) and the relative path (no leading `/`, opendal style). +pub(crate) fn hdfs_native_parse_path(path: &str) -> Result<(Option<String>, &str)> { + let url = Url::parse(path).map_err(|e| { + Error::new( + ErrorKind::DataInvalid, + format!("Invalid hdfs path: {path}: {e}"), + ) + })?; + // Non-special schemes parse even without `//` (e.g. `hdfs:x` is a valid + // non-hierarchical URL), so require the literal prefix before slicing. + let (Some(after_scheme), "hdfs") = (path.strip_prefix("hdfs://"), url.scheme()) else { + return Err(Error::new( + ErrorKind::DataInvalid, + format!("Invalid hdfs path: {path}, expected scheme `hdfs://`"), + )); + }; + + let name_node = url.host_str().filter(|h| !h.is_empty()).map(|host| { + url.port() + .map(|port| format!("hdfs://{host}:{port}")) + .unwrap_or_else(|| format!("hdfs://{host}")) + }); + + // `url.path()` borrows from `url` and can't be returned with the input's + // lifetime. Slice the path component out of the original input instead; + // it starts after the first `/` following the `hdfs://` prefix. Opendal + // paths must not start with `/` (`Deleter::delete` rejects them). + let rel = match after_scheme.find('/') { + Some(i) => after_scheme[i..].trim_start_matches('/'), + None => "", + }; + + Ok((name_node, rel)) +} + +/// Resolves the effective NameNode for a path, plus the relative path. As in +/// Hadoop, an authority with a port is used as is; a logical nameservice +/// authority (no port) resolves through `hdfs.name-node`, and an +/// authority-less path through `hdfs.name-node`, else `fs.defaultFS`. The +/// operator cache, `delete_stream` batching and `relativize_path` all go +/// through this, so they cannot drift apart. +pub(crate) fn hdfs_native_effective_name_node<'a>( + config: &HdfsNativeConfig, + path: &'a str, +) -> Result<(String, &'a str)> { + let (authority, relative_path) = hdfs_native_parse_path(path)?; + let invalid = |reason: String| { + Error::new( + ErrorKind::DataInvalid, + format!("Invalid hdfs path: {path}, {reason}"), + ) + }; + let name_node = match authority { + Some(authority) if hdfs_native_has_port(&authority) => authority, + Some(logical) => config.name_node.clone().ok_or_else(|| { Review Comment: The sharp edge here is `delete_prefix`/`delete_stream`: with `hdfs.name-node` set, every portless authority resolves to that one configured list, so `hdfs://ns-a/x`, `hdfs://ns-b/x`, and a typo'd authority all route to the same cluster — an orphan-cleanup pass keyed on the wrong name deletes against the wrong cluster. Java would just fail on an unknown nameservice. The flip side is that a nameservice defined only in `$HADOOP_CONF_DIR` is rejected here even though the client could resolve it from XML. Both come down to whether opendal 0.58.1 accepts a bare logical `name_node` and resolves it from the Hadoop XML — I couldn't confirm that from the registry sources. If it does, I'd forward a portless authority through unchanged and let the client resolve it; if it doesn't, I'd key the override per-nameservice and document the single-cluster assumption loudly. Do you know how opendal treats a logical `name_node` here? ########## crates/storage/opendal/src/hdfs_native.rs: ########## @@ -0,0 +1,928 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +//! HDFS storage backend via OpenDAL's `services-hdfs-native` (pure Rust, no JNI). + +use std::collections::HashMap; +use std::collections::hash_map::Entry; +use std::sync::{Arc, RwLock, Weak}; + +use iceberg::io::{HDFS_HADOOP_CONF_PREFIX, HDFS_HOST, HDFS_NAME_NODE, HDFS_PORT}; +use iceberg::{Error, ErrorKind, Result}; +use opendal::Operator; +use opendal::services::HdfsNativeConfig; +use serde::{Deserialize, Serialize}; +use tokio::runtime::Handle; +use tokio::task::JoinHandle; +use url::Url; + +use crate::OpenDalClientConfig; +use crate::utils::from_opendal_error; + +/// Hadoop's default filesystem, which serves authority-less paths. +const FS_DEFAULT_FS: &str = "fs.defaultFS"; +const HDFS_DEFAULT_PORT: u16 = 8020; +/// PyIceberg keys with no equivalent in opendal's config. +const HDFS_UNSUPPORTED_KEYS: [&str; 2] = ["hdfs.user", "hdfs.kerberos_ticket"]; + +/// `hdfs-native` dials a NameNode as a socket address and has no default +/// port, so anything without one can only be a logical nameservice name. +fn hdfs_native_has_port(name_node: &str) -> bool { Review Comment: This only looks after the last `:` and takes any `u16`, so it waves through a few shapes that aren't a NameNode: `viewfs://x:1`, `foo://x:1`, `nn:0`, `nn:+80` (`parse` accepts the leading `+`), and bare `::1` reads as port 1. Separately, `nn:8020` and `hdfs://nn:8020` normalize to different cache keys for the same cluster. I'd strip an optional `hdfs://`, reject any other `scheme://`, and normalize before joining — tightens the validation and dedupes the keys in one pass. -- This is an automated message from the Apache Git Service. To respond to the message, please log on to GitHub and use the URL above to go to the specific comment. To unsubscribe, e-mail: [email protected] For queries about this service, please contact Infrastructure at: [email protected] --------------------------------------------------------------------- To unsubscribe, e-mail: [email protected] For additional commands, e-mail: [email protected]
