Skip to content
Roboto
Esc
↑↓navigate↵open⌘Jpreview
On this page

roboto.formats.parquet.parquet_parser

Module Contents

ParquetParser

class roboto.formats.parquet.parquet_parser.ParquetParser(source, min_required_row_group_size=100000, small_row_group_count_threshold=32)#View Source

Parameters

source pathlib.Path
min_required_row_group_size int
small_row_group_count_threshold int

Properties

ParquetParser.column_count

column_count int #
Return type: int

ParquetParser.extract_timestamp_info()

extract_timestamp_info(timestamp_column_name=None, timestamp_unit=None)#View Source

Parameters

timestamp_column_name Optional[str]
timestamp_unit Optional[Union[str, roboto.time.TimeUnit]]

Properties

ParquetParser.fields

fields Generator[pyarrow.Field, None, None] #
Return type: Generator[pyarrow.Field, None, None]

ParquetParser.find_timestamp_field_by_type()

find_timestamp_field_by_type()#View Source

Return type

pyarrow.Field

ParquetParser.get_data_for_column()

get_data_for_column(column_name)#View Source

Parameters

column_name str

Return type

pyarrow.Table

ParquetParser.get_timestamp_field_by_name()

get_timestamp_field_by_name(column_name)#View Source

Parameters

column_name str

Return type

pyarrow.Field

ParquetParser.is_parquet_file()

static is_parquet_file(path)#View Source

Parameters

path pathlib.Path

Return type

bool

ParquetParser.requires_rewrite()

requires_rewrite(timestamp)#View Source

Return type

bool

ParquetParser.rewrite()

rewrite(outfile, timestamp, target_row_group_size_bytes=100 * 1000 * 1000)#View Source

Parameters

outfile pathlib.Path
target_row_group_size_bytes int

Return type

None

Properties

ParquetParser.row_count

row_count int #
Return type: int

ParquetParser.row_group_count

row_group_count int #
Return type: int

ParquetParser.row_group_size

row_group_size int #
Return type: int

logger

roboto.formats.parquet.parquet_parser.logger#View Source

Was this page helpful?