/
usr
/
local
/
lib64
/
python3.6
/
site-packages
/
pyarrow
/
/usr/local/lib64/python3.6/site-packages/pyarrow
mkdir
upload
Name
Size
Mode
Actions
include/
-
0755
rm
includes/
-
0755
rm
tensorflow/
-
0755
rm
tests/
-
0755
rm
vendored/
-
0755
rm
__pycache__/
-
0755
rm
array.pxi
79609
0644
edit
dl
rm
benchmark.pxi
869
0644
edit
dl
rm
benchmark.py
856
0644
edit
dl
rm
builder.pxi
2688
0644
edit
dl
rm
cffi.py
2178
0644
edit
dl
rm
compat.pxi
1810
0644
edit
dl
rm
compat.py
1076
0644
edit
dl
rm
compute.py
22837
0644
edit
dl
rm
config.pxi
2613
0644
edit
dl
rm
csv.py
962
0644
edit
dl
rm
cuda.py
1087
0644
edit
dl
rm
dataset.py
33333
0644
edit
dl
rm
error.pxi
7807
0644
edit
dl
rm
feather.py
9258
0644
edit
dl
rm
filesystem.py
14468
0644
edit
dl
rm
flight.py
1794
0644
edit
dl
rm
fs.py
13444
0644
edit
dl
rm
gandiva.pyx
18447
0644
edit
dl
rm
hdfs.py
7527
0644
edit
dl
rm
io.pxi
64495
0644
edit
dl
rm
ipc.pxi
28821
0644
edit
dl
rm
ipc.py
8029
0644
edit
dl
rm
json.py
858
0644
edit
dl
rm
jvm.py
9593
0644
edit
dl
rm
lib.cpython-36m-x86_64-linux-gnu.so
3760528
0755
edit
dl
rm
lib.pxd
14442
0644
edit
dl
rm
lib.pyx
4373
0644
edit
dl
rm
libarrow.so.600
48162600
0755
edit
dl
rm
libarrow_dataset.so.600
2541120
0755
edit
dl
rm
libarrow_flight.so.600
12999784
0755
edit
dl
rm
libarrow_python.so.600
1848920
0755
edit
dl
rm
libarrow_python_flight.so.600
112728
0755
edit
dl
rm
libparquet.so.600
4458552
0755
edit
dl
rm
libplasma.so.600
251296
0755
edit
dl
rm
lib_api.h
19153
0644
edit
dl
rm
memory.pxi
7451
0644
edit
dl
rm
orc.py
5269
0644
edit
dl
rm
pandas-shim.pxi
7987
0644
edit
dl
rm
pandas_compat.py
42243
0644
edit
dl
rm
parquet.py
86844
0644
edit
dl
rm
plasma-store-server
482776
0755
edit
dl
rm
plasma.py
6075
0644
edit
dl
rm
public-api.pxi
12814
0644
edit
dl
rm
scalar.pxi
29182
0644
edit
dl
rm
serialization.pxi
19086
0644
edit
dl
rm
serialization.py
18202
0644
edit
dl
rm
table.pxi
72277
0644
edit
dl
rm
tensor.pxi
34311
0644
edit
dl
rm
types.pxi
79432
0644
edit
dl
rm
types.py
10381
0644
edit
dl
rm
util.py
5001
0644
edit
dl
rm
_compute.cpython-36m-x86_64-linux-gnu.so
717312
0755
edit
dl
rm
_compute.pxd
1149
0644
edit
dl
rm
_compute.pyx
41759
0644
edit
dl
rm
_csv.cpython-36m-x86_64-linux-gnu.so
312168
0755
edit
dl
rm
_csv.pxd
1602
0644
edit
dl
rm
_csv.pyx
39297
0644
edit
dl
rm
_cuda.pxd
1922
0644
edit
dl
rm
_cuda.pyx
34731
0644
edit
dl
rm
_dataset.cpython-36m-x86_64-linux-gnu.so
1080168
0755
edit
dl
rm
_dataset.pxd
1644
0644
edit
dl
rm
_dataset.pyx
120054
0644
edit
dl
rm
_dataset_orc.cpython-36m-x86_64-linux-gnu.so
52536
0755
edit
dl
rm
_dataset_orc.pyx
1345
0644
edit
dl
rm
_feather.cpython-36m-x86_64-linux-gnu.so
94280
0755
edit
dl
rm
_feather.pyx
3623
0644
edit
dl
rm
_flight.cpython-36m-x86_64-linux-gnu.so
1124664
0755
edit
dl
rm
_flight.pyx
93532
0644
edit
dl
rm
_fs.cpython-36m-x86_64-linux-gnu.so
469992
0755
edit
dl
rm
_fs.pxd
2484
0644
edit
dl
rm
_fs.pyx
40017
0644
edit
dl
rm
_generated_version.py
142
0644
edit
dl
rm
_hdfs.cpython-36m-x86_64-linux-gnu.so
127352
0755
edit
dl
rm
_hdfs.pyx
5471
0644
edit
dl
rm
_hdfsio.cpython-36m-x86_64-linux-gnu.so
208096
0755
edit
dl
rm
_hdfsio.pyx
13681
0644
edit
dl
rm
_json.cpython-36m-x86_64-linux-gnu.so
90616
0755
edit
dl
rm
_json.pyx
8406
0644
edit
dl
rm
_orc.cpython-36m-x86_64-linux-gnu.so
109776
0755
edit
dl
rm
_orc.pxd
2350
0644
edit
dl
rm
_orc.pyx
5151
0644
edit
dl
rm
_parquet.cpython-36m-x86_64-linux-gnu.so
507792
0755
edit
dl
rm
_parquet.pxd
22084
0644
edit
dl
rm
_parquet.pyx
47997
0644
edit
dl
rm
_plasma.cpython-36m-x86_64-linux-gnu.so
259168
0755
edit
dl
rm
_plasma.pyx
29495
0644
edit
dl
rm
_s3fs.cpython-36m-x86_64-linux-gnu.so
199496
0755
edit
dl
rm
_s3fs.pyx
11662
0644
edit
dl
rm
__init__.pxd
2195
0644
edit
dl
rm
__init__.py
21263
0644
edit
dl
rm
Edit:
/usr/local/lib64/python3.6/site-packages/pyarrow/feather.py
(9258B)
# Licensed to the Apache Software Foundation (ASF) under one # or more contributor license agreements. See the NOTICE file # distributed with this work for additional information # regarding copyright ownership. The ASF licenses this file # to you under the Apache License, Version 2.0 (the # "License"); you may not use this file except in compliance # with the License. You may obtain a copy of the License at # # http://www.apache.org/licenses/LICENSE-2.0 # # Unless required by applicable law or agreed to in writing, # software distributed under the License is distributed on an # "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY # KIND, either express or implied. See the License for the # specific language governing permissions and limitations # under the License. import os from pyarrow.pandas_compat import _pandas_api # noqa from pyarrow.lib import (Codec, Table, # noqa concat_tables, schema) import pyarrow.lib as ext from pyarrow import _feather from pyarrow._feather import FeatherError # noqa: F401 from pyarrow.vendored.version import Version def _check_pandas_version(): if _pandas_api.loose_version < Version('0.17.0'): raise ImportError("feather requires pandas >= 0.17.0") class FeatherDataset: """ Encapsulates details of reading a list of Feather files. Parameters ---------- path_or_paths : List[str] A list of file names validate_schema : bool, default True Check that individual file schemas are all the same / compatible """ def __init__(self, path_or_paths, validate_schema=True): self.paths = path_or_paths self.validate_schema = validate_schema def read_table(self, columns=None): """ Read multiple feather files as a single pyarrow.Table Parameters ---------- columns : List[str] Names of columns to read from the file Returns ------- pyarrow.Table Content of the file as a table (of columns) """ _fil = read_table(self.paths[0], columns=columns) self._tables = [_fil] self.schema = _fil.schema for path in self.paths[1:]: table = read_table(path, columns=columns) if self.validate_schema: self.validate_schemas(path, table) self._tables.append(table) return concat_tables(self._tables) def validate_schemas(self, piece, table): if not self.schema.equals(table.schema): raise ValueError('Schema in {!s} was different. \n' '{!s}\n\nvs\n\n{!s}' .format(piece, self.schema, table.schema)) def read_pandas(self, columns=None, use_threads=True): """ Read multiple Parquet files as a single pandas DataFrame Parameters ---------- columns : List[str] Names of columns to read from the file use_threads : bool, default True Use multiple threads when converting to pandas Returns ------- pandas.DataFrame Content of the file as a pandas DataFrame (of columns) """ _check_pandas_version() return self.read_table(columns=columns).to_pandas( use_threads=use_threads) def check_chunked_overflow(name, col): if col.num_chunks == 1: return if col.type in (ext.binary(), ext.string()): raise ValueError("Column '{}' exceeds 2GB maximum capacity of " "a Feather binary column. This restriction may be " "lifted in the future".format(name)) else: # TODO(wesm): Not sure when else this might be reached raise ValueError("Column '{}' of type {} was chunked on conversion " "to Arrow and cannot be currently written to " "Feather format".format(name, str(col.type))) _FEATHER_SUPPORTED_CODECS = {'lz4', 'zstd', 'uncompressed'} def write_feather(df, dest, compression=None, compression_level=None, chunksize=None, version=2): """ Write a pandas.DataFrame to Feather format. Parameters ---------- df : pandas.DataFrame or pyarrow.Table Data to write out as Feather format. dest : str Local destination path. compression : string, default None Can be one of {"zstd", "lz4", "uncompressed"}. The default of None uses LZ4 for V2 files if it is available, otherwise uncompressed. compression_level : int, default None Use a compression level particular to the chosen compressor. If None use the default compression level chunksize : int, default None For V2 files, the internal maximum size of Arrow RecordBatch chunks when writing the Arrow IPC file format. None means use the default, which is currently 64K version : int, default 2 Feather file version. Version 2 is the current. Version 1 is the more limited legacy format """ if _pandas_api.have_pandas: _check_pandas_version() if (_pandas_api.has_sparse and isinstance(df, _pandas_api.pd.SparseDataFrame)): df = df.to_dense() if _pandas_api.is_data_frame(df): table = Table.from_pandas(df, preserve_index=False) if version == 1: # Version 1 does not chunking for i, name in enumerate(table.schema.names): col = table[i] check_chunked_overflow(name, col) else: table = df if version == 1: if len(table.column_names) > len(set(table.column_names)): raise ValueError("cannot serialize duplicate column names") if compression is not None: raise ValueError("Feather V1 files do not support compression " "option") if chunksize is not None: raise ValueError("Feather V1 files do not support chunksize " "option") else: if compression is None and Codec.is_available('lz4_frame'): compression = 'lz4' elif (compression is not None and compression not in _FEATHER_SUPPORTED_CODECS): raise ValueError('compression="{}" not supported, must be ' 'one of {}'.format(compression, _FEATHER_SUPPORTED_CODECS)) try: _feather.write_feather(table, dest, compression=compression, compression_level=compression_level, chunksize=chunksize, version=version) except Exception: if isinstance(dest, str): try: os.remove(dest) except os.error: pass raise def read_feather(source, columns=None, use_threads=True, memory_map=True): """ Read a pandas.DataFrame from Feather format. To read as pyarrow.Table use feather.read_table. Parameters ---------- source : str file path, or file-like object columns : sequence, optional Only read a specific set of columns. If not provided, all columns are read. use_threads : bool, default True Whether to parallelize reading using multiple threads. If false the restriction is only used in the conversion to Pandas and not in the reading from Feather format. memory_map : boolean, default True Use memory mapping when opening file on disk Returns ------- df : pandas.DataFrame """ _check_pandas_version() return (read_table(source, columns=columns, memory_map=memory_map) .to_pandas(use_threads=use_threads)) def read_table(source, columns=None, memory_map=True): """ Read a pyarrow.Table from Feather format Parameters ---------- source : str file path, or file-like object columns : sequence, optional Only read a specific set of columns. If not provided, all columns are read. memory_map : boolean, default True Use memory mapping when opening file on disk Returns ------- table : pyarrow.Table """ reader = _feather.FeatherReader(source, use_memory_map=memory_map) if columns is None: return reader.read() column_types = [type(column) for column in columns] if all(map(lambda t: t == int, column_types)): table = reader.read_indices(columns) elif all(map(lambda t: t == str, column_types)): table = reader.read_names(columns) else: column_type_names = [t.__name__ for t in column_types] raise TypeError("Columns must be indices or names. " "Got columns {} of types {}" .format(columns, column_type_names)) # Feather v1 already respects the column selection if reader.version < 3: return table # Feather v2 reads with sorted / deduplicated selection elif sorted(set(columns)) == columns: return table else: # follow exact order / selection of names return table.select(columns)
Save
cmd:
run