/
usr
/
local
/
lib64
/
python3.6
/
site-packages
/
pyarrow
/
/usr/local/lib64/python3.6/site-packages/pyarrow
mkdir
upload
Name
Size
Mode
Actions
include/
-
0755
rm
includes/
-
0755
rm
tensorflow/
-
0755
rm
tests/
-
0755
rm
vendored/
-
0755
rm
__pycache__/
-
0755
rm
array.pxi
79609
0644
edit
dl
rm
benchmark.pxi
869
0644
edit
dl
rm
benchmark.py
856
0644
edit
dl
rm
builder.pxi
2688
0644
edit
dl
rm
cffi.py
2178
0644
edit
dl
rm
compat.pxi
1810
0644
edit
dl
rm
compat.py
1076
0644
edit
dl
rm
compute.py
22837
0644
edit
dl
rm
config.pxi
2613
0644
edit
dl
rm
csv.py
962
0644
edit
dl
rm
cuda.py
1087
0644
edit
dl
rm
dataset.py
33333
0644
edit
dl
rm
error.pxi
7807
0644
edit
dl
rm
feather.py
9258
0644
edit
dl
rm
filesystem.py
14468
0644
edit
dl
rm
flight.py
1794
0644
edit
dl
rm
fs.py
13444
0644
edit
dl
rm
gandiva.pyx
18447
0644
edit
dl
rm
hdfs.py
7527
0644
edit
dl
rm
io.pxi
64495
0644
edit
dl
rm
ipc.pxi
28821
0644
edit
dl
rm
ipc.py
8029
0644
edit
dl
rm
json.py
858
0644
edit
dl
rm
jvm.py
9593
0644
edit
dl
rm
lib.cpython-36m-x86_64-linux-gnu.so
3760528
0755
edit
dl
rm
lib.pxd
14442
0644
edit
dl
rm
lib.pyx
4373
0644
edit
dl
rm
libarrow.so.600
48162600
0755
edit
dl
rm
libarrow_dataset.so.600
2541120
0755
edit
dl
rm
libarrow_flight.so.600
12999784
0755
edit
dl
rm
libarrow_python.so.600
1848920
0755
edit
dl
rm
libarrow_python_flight.so.600
112728
0755
edit
dl
rm
libparquet.so.600
4458552
0755
edit
dl
rm
libplasma.so.600
251296
0755
edit
dl
rm
lib_api.h
19153
0644
edit
dl
rm
memory.pxi
7451
0644
edit
dl
rm
orc.py
5269
0644
edit
dl
rm
pandas-shim.pxi
7987
0644
edit
dl
rm
pandas_compat.py
42243
0644
edit
dl
rm
parquet.py
86844
0644
edit
dl
rm
plasma-store-server
482776
0755
edit
dl
rm
plasma.py
6075
0644
edit
dl
rm
public-api.pxi
12814
0644
edit
dl
rm
scalar.pxi
29182
0644
edit
dl
rm
serialization.pxi
19086
0644
edit
dl
rm
serialization.py
18202
0644
edit
dl
rm
table.pxi
72277
0644
edit
dl
rm
tensor.pxi
34311
0644
edit
dl
rm
types.pxi
79432
0644
edit
dl
rm
types.py
10381
0644
edit
dl
rm
util.py
5001
0644
edit
dl
rm
_compute.cpython-36m-x86_64-linux-gnu.so
717312
0755
edit
dl
rm
_compute.pxd
1149
0644
edit
dl
rm
_compute.pyx
41759
0644
edit
dl
rm
_csv.cpython-36m-x86_64-linux-gnu.so
312168
0755
edit
dl
rm
_csv.pxd
1602
0644
edit
dl
rm
_csv.pyx
39297
0644
edit
dl
rm
_cuda.pxd
1922
0644
edit
dl
rm
_cuda.pyx
34731
0644
edit
dl
rm
_dataset.cpython-36m-x86_64-linux-gnu.so
1080168
0755
edit
dl
rm
_dataset.pxd
1644
0644
edit
dl
rm
_dataset.pyx
120054
0644
edit
dl
rm
_dataset_orc.cpython-36m-x86_64-linux-gnu.so
52536
0755
edit
dl
rm
_dataset_orc.pyx
1345
0644
edit
dl
rm
_feather.cpython-36m-x86_64-linux-gnu.so
94280
0755
edit
dl
rm
_feather.pyx
3623
0644
edit
dl
rm
_flight.cpython-36m-x86_64-linux-gnu.so
1124664
0755
edit
dl
rm
_flight.pyx
93532
0644
edit
dl
rm
_fs.cpython-36m-x86_64-linux-gnu.so
469992
0755
edit
dl
rm
_fs.pxd
2484
0644
edit
dl
rm
_fs.pyx
40017
0644
edit
dl
rm
_generated_version.py
142
0644
edit
dl
rm
_hdfs.cpython-36m-x86_64-linux-gnu.so
127352
0755
edit
dl
rm
_hdfs.pyx
5471
0644
edit
dl
rm
_hdfsio.cpython-36m-x86_64-linux-gnu.so
208096
0755
edit
dl
rm
_hdfsio.pyx
13681
0644
edit
dl
rm
_json.cpython-36m-x86_64-linux-gnu.so
90616
0755
edit
dl
rm
_json.pyx
8406
0644
edit
dl
rm
_orc.cpython-36m-x86_64-linux-gnu.so
109776
0755
edit
dl
rm
_orc.pxd
2350
0644
edit
dl
rm
_orc.pyx
5151
0644
edit
dl
rm
_parquet.cpython-36m-x86_64-linux-gnu.so
507792
0755
edit
dl
rm
_parquet.pxd
22084
0644
edit
dl
rm
_parquet.pyx
47997
0644
edit
dl
rm
_plasma.cpython-36m-x86_64-linux-gnu.so
259168
0755
edit
dl
rm
_plasma.pyx
29495
0644
edit
dl
rm
_s3fs.cpython-36m-x86_64-linux-gnu.so
199496
0755
edit
dl
rm
_s3fs.pyx
11662
0644
edit
dl
rm
__init__.pxd
2195
0644
edit
dl
rm
__init__.py
21263
0644
edit
dl
rm
Edit:
/usr/local/lib64/python3.6/site-packages/pyarrow/hdfs.py
(7527B)
# Licensed to the Apache Software Foundation (ASF) under one # or more contributor license agreements. See the NOTICE file # distributed with this work for additional information # regarding copyright ownership. The ASF licenses this file # to you under the Apache License, Version 2.0 (the # "License"); you may not use this file except in compliance # with the License. You may obtain a copy of the License at # # http://www.apache.org/licenses/LICENSE-2.0 # # Unless required by applicable law or agreed to in writing, # software distributed under the License is distributed on an # "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY # KIND, either express or implied. See the License for the # specific language governing permissions and limitations # under the License. import os import posixpath import sys import warnings from pyarrow.util import implements, _DEPR_MSG from pyarrow.filesystem import FileSystem import pyarrow._hdfsio as _hdfsio class HadoopFileSystem(_hdfsio.HadoopFileSystem, FileSystem): """ DEPRECATED: FileSystem interface for HDFS cluster. See pyarrow.hdfs.connect for full connection details .. deprecated:: 2.0 ``pyarrow.hdfs.HadoopFileSystem`` is deprecated, please use ``pyarrow.fs.HadoopFileSystem`` instead. """ def __init__(self, host="default", port=0, user=None, kerb_ticket=None, driver='libhdfs', extra_conf=None): warnings.warn( _DEPR_MSG.format( "hdfs.HadoopFileSystem", "2.0.0", "fs.HadoopFileSystem"), FutureWarning, stacklevel=2) if driver == 'libhdfs': _maybe_set_hadoop_classpath() self._connect(host, port, user, kerb_ticket, extra_conf) def __reduce__(self): return (HadoopFileSystem, (self.host, self.port, self.user, self.kerb_ticket, self.extra_conf)) def _isfilestore(self): """ Return True if this is a Unix-style file store with directories. """ return True @implements(FileSystem.isdir) def isdir(self, path): return super().isdir(path) @implements(FileSystem.isfile) def isfile(self, path): return super().isfile(path) @implements(FileSystem.delete) def delete(self, path, recursive=False): return super().delete(path, recursive) def mkdir(self, path, **kwargs): """ Create directory in HDFS. Parameters ---------- path : str Directory path to create, including any parent directories. Notes ----- libhdfs does not support create_parents=False, so we ignore this here """ return super().mkdir(path) @implements(FileSystem.rename) def rename(self, path, new_path): return super().rename(path, new_path) @implements(FileSystem.exists) def exists(self, path): return super().exists(path) def ls(self, path, detail=False): """ Retrieve directory contents and metadata, if requested. Parameters ---------- path : str HDFS path to retrieve contents of. detail : bool, default False If False, only return list of paths. Returns ------- result : list of dicts (detail=True) or strings (detail=False) """ return super().ls(path, detail) def walk(self, top_path): """ Directory tree generator for HDFS, like os.walk. Parameters ---------- top_path : str Root directory for tree traversal. Returns ------- Generator yielding 3-tuple (dirpath, dirnames, filename) """ contents = self.ls(top_path, detail=True) directories, files = _libhdfs_walk_files_dirs(top_path, contents) yield top_path, directories, files for dirname in directories: yield from self.walk(self._path_join(top_path, dirname)) def _maybe_set_hadoop_classpath(): import re if re.search(r'hadoop-common[^/]+.jar', os.environ.get('CLASSPATH', '')): return if 'HADOOP_HOME' in os.environ: if sys.platform != 'win32': classpath = _derive_hadoop_classpath() else: hadoop_bin = '{}/bin/hadoop'.format(os.environ['HADOOP_HOME']) classpath = _hadoop_classpath_glob(hadoop_bin) else: classpath = _hadoop_classpath_glob('hadoop') os.environ['CLASSPATH'] = classpath.decode('utf-8') def _derive_hadoop_classpath(): import subprocess find_args = ('find', '-L', os.environ['HADOOP_HOME'], '-name', '*.jar') find = subprocess.Popen(find_args, stdout=subprocess.PIPE) xargs_echo = subprocess.Popen(('xargs', 'echo'), stdin=find.stdout, stdout=subprocess.PIPE) jars = subprocess.check_output(('tr', "' '", "':'"), stdin=xargs_echo.stdout) hadoop_conf = os.environ["HADOOP_CONF_DIR"] \ if "HADOOP_CONF_DIR" in os.environ \ else os.environ["HADOOP_HOME"] + "/etc/hadoop" return (hadoop_conf + ":").encode("utf-8") + jars def _hadoop_classpath_glob(hadoop_bin): import subprocess hadoop_classpath_args = (hadoop_bin, 'classpath', '--glob') return subprocess.check_output(hadoop_classpath_args) def _libhdfs_walk_files_dirs(top_path, contents): files = [] directories = [] for c in contents: scrubbed_name = posixpath.split(c['name'])[1] if c['kind'] == 'file': files.append(scrubbed_name) else: directories.append(scrubbed_name) return directories, files def connect(host="default", port=0, user=None, kerb_ticket=None, extra_conf=None): """ DEPRECATED: Connect to an HDFS cluster. All parameters are optional and should only be set if the defaults need to be overridden. Authentication should be automatic if the HDFS cluster uses Kerberos. However, if a username is specified, then the ticket cache will likely be required. .. deprecated:: 2.0 ``pyarrow.hdfs.connect`` is deprecated, please use ``pyarrow.fs.HadoopFileSystem`` instead. Parameters ---------- host : NameNode. Set to "default" for fs.defaultFS from core-site.xml. port : NameNode's port. Set to 0 for default or logical (HA) nodes. user : Username when connecting to HDFS; None implies login user. kerb_ticket : Path to Kerberos ticket cache. extra_conf : dict, default None extra Key/Value pairs for config; Will override any hdfs-site.xml properties Notes ----- The first time you call this method, it will take longer than usual due to JNI spin-up time. Returns ------- filesystem : HadoopFileSystem """ warnings.warn( _DEPR_MSG.format("hdfs.connect", "2.0.0", "fs.HadoopFileSystem"), FutureWarning, stacklevel=2 ) return _connect( host=host, port=port, user=user, kerb_ticket=kerb_ticket, extra_conf=extra_conf ) def _connect(host="default", port=0, user=None, kerb_ticket=None, extra_conf=None): with warnings.catch_warnings(): warnings.simplefilter("ignore") fs = HadoopFileSystem(host=host, port=port, user=user, kerb_ticket=kerb_ticket, extra_conf=extra_conf) return fs
Save
cmd:
run