/
usr
/
local
/
lib64
/
python3.6
/
site-packages
/
pyarrow
/
/usr/local/lib64/python3.6/site-packages/pyarrow
mkdir
upload
Name
Size
Mode
Actions
include/
-
0755
rm
includes/
-
0755
rm
tensorflow/
-
0755
rm
tests/
-
0755
rm
vendored/
-
0755
rm
__pycache__/
-
0755
rm
array.pxi
79609
0644
edit
dl
rm
benchmark.pxi
869
0644
edit
dl
rm
benchmark.py
856
0644
edit
dl
rm
builder.pxi
2688
0644
edit
dl
rm
cffi.py
2178
0644
edit
dl
rm
compat.pxi
1810
0644
edit
dl
rm
compat.py
1076
0644
edit
dl
rm
compute.py
22837
0644
edit
dl
rm
config.pxi
2613
0644
edit
dl
rm
csv.py
962
0644
edit
dl
rm
cuda.py
1087
0644
edit
dl
rm
dataset.py
33333
0644
edit
dl
rm
error.pxi
7807
0644
edit
dl
rm
feather.py
9258
0644
edit
dl
rm
filesystem.py
14468
0644
edit
dl
rm
flight.py
1794
0644
edit
dl
rm
fs.py
13444
0644
edit
dl
rm
gandiva.pyx
18447
0644
edit
dl
rm
hdfs.py
7527
0644
edit
dl
rm
io.pxi
64495
0644
edit
dl
rm
ipc.pxi
28821
0644
edit
dl
rm
ipc.py
8029
0644
edit
dl
rm
json.py
858
0644
edit
dl
rm
jvm.py
9593
0644
edit
dl
rm
lib.cpython-36m-x86_64-linux-gnu.so
3760528
0755
edit
dl
rm
lib.pxd
14442
0644
edit
dl
rm
lib.pyx
4373
0644
edit
dl
rm
libarrow.so.600
48162600
0755
edit
dl
rm
libarrow_dataset.so.600
2541120
0755
edit
dl
rm
libarrow_flight.so.600
12999784
0755
edit
dl
rm
libarrow_python.so.600
1848920
0755
edit
dl
rm
libarrow_python_flight.so.600
112728
0755
edit
dl
rm
libparquet.so.600
4458552
0755
edit
dl
rm
libplasma.so.600
251296
0755
edit
dl
rm
lib_api.h
19153
0644
edit
dl
rm
memory.pxi
7451
0644
edit
dl
rm
orc.py
5269
0644
edit
dl
rm
pandas-shim.pxi
7987
0644
edit
dl
rm
pandas_compat.py
42243
0644
edit
dl
rm
parquet.py
86844
0644
edit
dl
rm
plasma-store-server
482776
0755
edit
dl
rm
plasma.py
6075
0644
edit
dl
rm
public-api.pxi
12814
0644
edit
dl
rm
scalar.pxi
29182
0644
edit
dl
rm
serialization.pxi
19086
0644
edit
dl
rm
serialization.py
18202
0644
edit
dl
rm
table.pxi
72277
0644
edit
dl
rm
tensor.pxi
34311
0644
edit
dl
rm
types.pxi
79432
0644
edit
dl
rm
types.py
10381
0644
edit
dl
rm
util.py
5001
0644
edit
dl
rm
_compute.cpython-36m-x86_64-linux-gnu.so
717312
0755
edit
dl
rm
_compute.pxd
1149
0644
edit
dl
rm
_compute.pyx
41759
0644
edit
dl
rm
_csv.cpython-36m-x86_64-linux-gnu.so
312168
0755
edit
dl
rm
_csv.pxd
1602
0644
edit
dl
rm
_csv.pyx
39297
0644
edit
dl
rm
_cuda.pxd
1922
0644
edit
dl
rm
_cuda.pyx
34731
0644
edit
dl
rm
_dataset.cpython-36m-x86_64-linux-gnu.so
1080168
0755
edit
dl
rm
_dataset.pxd
1644
0644
edit
dl
rm
_dataset.pyx
120054
0644
edit
dl
rm
_dataset_orc.cpython-36m-x86_64-linux-gnu.so
52536
0755
edit
dl
rm
_dataset_orc.pyx
1345
0644
edit
dl
rm
_feather.cpython-36m-x86_64-linux-gnu.so
94280
0755
edit
dl
rm
_feather.pyx
3623
0644
edit
dl
rm
_flight.cpython-36m-x86_64-linux-gnu.so
1124664
0755
edit
dl
rm
_flight.pyx
93532
0644
edit
dl
rm
_fs.cpython-36m-x86_64-linux-gnu.so
469992
0755
edit
dl
rm
_fs.pxd
2484
0644
edit
dl
rm
_fs.pyx
40017
0644
edit
dl
rm
_generated_version.py
142
0644
edit
dl
rm
_hdfs.cpython-36m-x86_64-linux-gnu.so
127352
0755
edit
dl
rm
_hdfs.pyx
5471
0644
edit
dl
rm
_hdfsio.cpython-36m-x86_64-linux-gnu.so
208096
0755
edit
dl
rm
_hdfsio.pyx
13681
0644
edit
dl
rm
_json.cpython-36m-x86_64-linux-gnu.so
90616
0755
edit
dl
rm
_json.pyx
8406
0644
edit
dl
rm
_orc.cpython-36m-x86_64-linux-gnu.so
109776
0755
edit
dl
rm
_orc.pxd
2350
0644
edit
dl
rm
_orc.pyx
5151
0644
edit
dl
rm
_parquet.cpython-36m-x86_64-linux-gnu.so
507792
0755
edit
dl
rm
_parquet.pxd
22084
0644
edit
dl
rm
_parquet.pyx
47997
0644
edit
dl
rm
_plasma.cpython-36m-x86_64-linux-gnu.so
259168
0755
edit
dl
rm
_plasma.pyx
29495
0644
edit
dl
rm
_s3fs.cpython-36m-x86_64-linux-gnu.so
199496
0755
edit
dl
rm
_s3fs.pyx
11662
0644
edit
dl
rm
__init__.pxd
2195
0644
edit
dl
rm
__init__.py
21263
0644
edit
dl
rm
Edit:
/usr/local/lib64/python3.6/site-packages/pyarrow/_json.pyx
(8406B)
# Licensed to the Apache Software Foundation (ASF) under one # or more contributor license agreements. See the NOTICE file # distributed with this work for additional information # regarding copyright ownership. The ASF licenses this file # to you under the Apache License, Version 2.0 (the # "License"); you may not use this file except in compliance # with the License. You may obtain a copy of the License at # # http://www.apache.org/licenses/LICENSE-2.0 # # Unless required by applicable law or agreed to in writing, # software distributed under the License is distributed on an # "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY # KIND, either express or implied. See the License for the # specific language governing permissions and limitations # under the License. # cython: profile=False # distutils: language = c++ # cython: language_level = 3 from pyarrow.includes.common cimport * from pyarrow.includes.libarrow cimport * from pyarrow.lib cimport (check_status, _Weakrefable, Field, MemoryPool, ensure_type, maybe_unbox_memory_pool, get_input_stream, pyarrow_wrap_table, pyarrow_wrap_data_type, pyarrow_unwrap_data_type, pyarrow_wrap_schema, pyarrow_unwrap_schema) cdef class ReadOptions(_Weakrefable): """ Options for reading JSON files. Parameters ---------- use_threads : bool, optional (default True) Whether to use multiple threads to accelerate reading block_size : int, optional How much bytes to process at a time from the input stream. This will determine multi-threading granularity as well as the size of individual chunks in the Table. """ cdef: CJSONReadOptions options # Avoid mistakingly creating attributes __slots__ = () def __init__(self, use_threads=None, block_size=None): self.options = CJSONReadOptions.Defaults() if use_threads is not None: self.use_threads = use_threads if block_size is not None: self.block_size = block_size @property def use_threads(self): """ Whether to use multiple threads to accelerate reading. """ return self.options.use_threads @use_threads.setter def use_threads(self, value): self.options.use_threads = value @property def block_size(self): """ How much bytes to process at a time from the input stream. This will determine multi-threading granularity as well as the size of individual chunks in the Table. """ return self.options.block_size @block_size.setter def block_size(self, value): self.options.block_size = value cdef class ParseOptions(_Weakrefable): """ Options for parsing JSON files. Parameters ---------- explicit_schema : Schema, optional (default None) Optional explicit schema (no type inference, ignores other fields). newlines_in_values : bool, optional (default False) Whether objects may be printed across multiple lines (for example pretty printed). If false, input must end with an empty line. unexpected_field_behavior : str, default "infer" How JSON fields outside of explicit_schema (if given) are treated. Possible behaviors: - "ignore": unexpected JSON fields are ignored - "error": error out on unexpected JSON fields - "infer": unexpected JSON fields are type-inferred and included in the output """ cdef: CJSONParseOptions options __slots__ = () def __init__(self, explicit_schema=None, newlines_in_values=None, unexpected_field_behavior=None): self.options = CJSONParseOptions.Defaults() if explicit_schema is not None: self.explicit_schema = explicit_schema if newlines_in_values is not None: self.newlines_in_values = newlines_in_values if unexpected_field_behavior is not None: self.unexpected_field_behavior = unexpected_field_behavior @property def explicit_schema(self): """ Optional explicit schema (no type inference, ignores other fields) """ if self.options.explicit_schema.get() == NULL: return None else: return pyarrow_wrap_schema(self.options.explicit_schema) @explicit_schema.setter def explicit_schema(self, value): self.options.explicit_schema = pyarrow_unwrap_schema(value) @property def newlines_in_values(self): """ Whether newline characters are allowed in JSON values. Setting this to True reduces the performance of multi-threaded JSON reading. """ return self.options.newlines_in_values @newlines_in_values.setter def newlines_in_values(self, value): self.options.newlines_in_values = value @property def unexpected_field_behavior(self): """ How JSON fields outside of explicit_schema (if given) are treated. Possible behaviors: - "ignore": unexpected JSON fields are ignored - "error": error out on unexpected JSON fields - "infer": unexpected JSON fields are type-inferred and included in the output Set to "infer" by default. """ v = self.options.unexpected_field_behavior if v == CUnexpectedFieldBehavior_Ignore: return "ignore" elif v == CUnexpectedFieldBehavior_Error: return "error" elif v == CUnexpectedFieldBehavior_InferType: return "infer" else: raise ValueError('Unexpected value for unexpected_field_behavior') @unexpected_field_behavior.setter def unexpected_field_behavior(self, value): cdef CUnexpectedFieldBehavior v if value == "ignore": v = CUnexpectedFieldBehavior_Ignore elif value == "error": v = CUnexpectedFieldBehavior_Error elif value == "infer": v = CUnexpectedFieldBehavior_InferType else: raise ValueError( "Unexpected value `{}` for `unexpected_field_behavior`, pass " "either `ignore`, `error` or `infer`.".format(value) ) self.options.unexpected_field_behavior = v cdef _get_reader(input_file, shared_ptr[CInputStream]* out): use_memory_map = False get_input_stream(input_file, use_memory_map, out) cdef _get_read_options(ReadOptions read_options, CJSONReadOptions* out): if read_options is None: out[0] = CJSONReadOptions.Defaults() else: out[0] = read_options.options cdef _get_parse_options(ParseOptions parse_options, CJSONParseOptions* out): if parse_options is None: out[0] = CJSONParseOptions.Defaults() else: out[0] = parse_options.options def read_json(input_file, read_options=None, parse_options=None, MemoryPool memory_pool=None): """ Read a Table from a stream of JSON data. Parameters ---------- input_file : str, path or file-like object The location of JSON data. Currently only the line-delimited JSON format is supported. read_options : pyarrow.json.ReadOptions, optional Options for the JSON reader (see ReadOptions constructor for defaults). parse_options : pyarrow.json.ParseOptions, optional Options for the JSON parser (see ParseOptions constructor for defaults). memory_pool : MemoryPool, optional Pool to allocate Table memory from. Returns ------- :class:`pyarrow.Table` Contents of the JSON file as a in-memory table. """ cdef: shared_ptr[CInputStream] stream CJSONReadOptions c_read_options CJSONParseOptions c_parse_options shared_ptr[CJSONReader] reader shared_ptr[CTable] table _get_reader(input_file, &stream) _get_read_options(read_options, &c_read_options) _get_parse_options(parse_options, &c_parse_options) reader = GetResultValue( CJSONReader.Make(maybe_unbox_memory_pool(memory_pool), stream, c_read_options, c_parse_options)) with nogil: table = GetResultValue(reader.get().Read()) return pyarrow_wrap_table(table)
Save
cmd:
run