Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
301 changes: 301 additions & 0 deletions python/pyarrow/_acero.pyi
Original file line number Diff line number Diff line change
@@ -0,0 +1,301 @@
# This file was generated by stubgen-pyx v0.2.23 from python/pyarrow/_acero.pyx

from pyarrow._compute import Expression
from pyarrow.includes.common import *
from pyarrow.includes.libarrow import *
from pyarrow.includes.libarrow_acero import *
from pyarrow.lib import *
from pyarrow.lib import RecordBatchReader, Table


class ExecNodeOptions(_Weakrefable):
"""
Base class for the node options.

Use one of the subclasses to construct an options object.
"""
__slots__ = ()

class TableSourceNodeOptions(_TableSourceNodeOptions):
"""
A Source node which accepts a table.

This is the option class for the "table_source" node factory.

Parameters
----------
table : pyarrow.Table
The table which acts as the data source.
"""
def __init__(self, table: Table): ...

class RecordBatchReaderSourceNodeOptions(_RecordBatchReaderSourceNodeOptions):
"""
A Source node which streams data from a RecordBatchReader.

This is the option class for the "record_batch_reader_source" node
factory.

Parameters
----------
reader : pyarrow.RecordBatchReader
The reader which acts as the data source.
"""
def __init__(self, reader: RecordBatchReader): ...

class FilterNodeOptions(_FilterNodeOptions):
"""
Make a node which excludes some rows from batches passed through it.

This is the option class for the "filter" node factory.

The "filter" operation provides an option to define data filtering
criteria. It selects rows where the given expression evaluates to true.
Filters can be written using pyarrow.compute.Expression, and the
expression must have a return type of boolean.

Parameters
----------
filter_expression : pyarrow.compute.Expression
"""
def __init__(self, filter_expression: Expression): ...

class ProjectNodeOptions(_ProjectNodeOptions):
"""
Make a node which executes expressions on input batches,
producing batches of the same length with new columns.

This is the option class for the "project" node factory.

The "project" operation rearranges, deletes, transforms, and
creates columns. Each output column is computed by evaluating
an expression against the source record batch. These must be
scalar expressions (expressions consisting of scalar literals,
field references and scalar functions, i.e. elementwise functions
that return one value for each input row independent of the value
of all other rows).

Parameters
----------
expressions : list of pyarrow.compute.Expression
List of expressions to evaluate against the source batch. This must
be scalar expressions.
names : list of str, optional
List of names for each of the output columns (same length as
`expressions`). If `names` is not provided, the string
representations of exprs will be used.
"""
def __init__(self, expressions, names=None): ...

class AggregateNodeOptions(_AggregateNodeOptions):
"""
Make a node which aggregates input batches, optionally grouped by keys.

This is the option class for the "aggregate" node factory.

Acero supports two types of aggregates: "scalar" aggregates,
and "hash" aggregates. Scalar aggregates reduce an array or scalar
input to a single scalar output (e.g. computing the mean of a column).
Hash aggregates act like GROUP BY in SQL and first partition data
based on one or more key columns, then reduce the data in each partition.
The aggregate node supports both types of computation, and can compute
any number of aggregations at once.

Parameters
----------
aggregates : list of tuples
Aggregations which will be applied to the targeted fields.
Specified as a list of tuples, where each tuple is one aggregation
specification and consists of: aggregation target column(s) followed
by function name, aggregation function options object and the
output field name.
The target column(s) specification can be a single field reference,
an empty list or a list of fields unary, nullary and n-ary aggregation
functions respectively. Each field reference can be a string
column name or expression.
keys : list of field references, optional
Keys by which aggregations will be grouped. Each key can reference
a field using a string name or expression.
"""
def __init__(self, aggregates, keys=None): ...

class OrderByNodeOptions(_OrderByNodeOptions):
"""
Make a node which applies a new ordering to the data.

Currently this node works by accumulating all data, sorting, and then
emitting the new data with an updated batch index.
Larger-than-memory sort is not currently supported.

This is the option class for the "order_by" node factory.

Parameters
----------
sort_keys : sequence of (name, order, null_placement="at_end") tuples
Names of field/column keys to sort the input on,
along with the order each field/column is sorted in.
Each field reference can be a string column name or expression.
Accepted values for `order` are "ascending", "descending".
Accepted values for `null_placement` are "at_start", "at_end".
null_placement : str, optional
Where nulls in input should be sorted, only applying to
columns/fields mentioned in `sort_keys`.
Accepted values are "at_start", "at_end",
"""
def __init__(self, sort_keys=(), *, null_placement=None): ...

class HashJoinNodeOptions(_HashJoinNodeOptions):
"""
Make a node which implements join operation using hash join strategy.

This is the option class for the "hashjoin" node factory.

Parameters
----------
join_type : str
Type of join. One of "left semi", "right semi", "left anti",
"right anti", "inner", "left outer", "right outer", "full outer".
left_keys : str, Expression or list
Key fields from left input. Each key can be a string column name
or a field expression, or a list of such field references.
right_keys : str, Expression or list
Key fields from right input. See `left_keys` for details.
left_output : list, optional
List of output fields passed from left input. If left and right
output fields are not specified, all valid fields from both left and
right input will be output. Each field can be a string column name
or a field expression.
right_output : list, optional
List of output fields passed from right input. If left and right
output fields are not specified, all valid fields from both left and
right input will be output. Each field can be a string column name
or a field expression.
output_suffix_for_left : str
Suffix added to names of output fields coming from left input
(used to distinguish, if necessary, between fields of the same
name in left and right input and can be left empty if there are
no name collisions).
output_suffix_for_right : str
Suffix added to names of output fields coming from right input,
see `output_suffix_for_left` for details.
filter_expression : pyarrow.compute.Expression
Residual filter which is applied to matching row.
"""
def __init__(self, join_type, left_keys, right_keys, left_output=None, right_output=None, output_suffix_for_left='', output_suffix_for_right='', filter_expression=None): ...

class AsofJoinNodeOptions(_AsofJoinNodeOptions):
"""
Make a node which implements 'as of join' operation.

This is the option class for the "asofjoin" node factory.

Parameters
----------
left_on : str, Expression
The left key on which the join operation should be performed.
Can be a string column name or a field expression.

An inexact match is used on the "on" key, i.e. a row is considered a
match if and only if ``right.on - left.on`` is in the range
``[min(0, tolerance), max(0, tolerance)]``.

The input dataset must be sorted by the "on" key. Must be a single
field of a common type.

Currently, the "on" key must be an integer, date, or timestamp type.
left_by: str, Expression or list
The left keys on which the join operation should be performed.
Exact equality is used for each field of the "by" keys.
Each key can be a string column name or a field expression,
or a list of such field references.
right_on : str, Expression
The right key on which the join operation should be performed.
See `left_on` for details.
right_by: str, Expression or list
The right keys on which the join operation should be performed.
See `left_by` for details.
tolerance : int
The tolerance to use for the asof join. The tolerance is interpreted in
the same units as the "on" key.
"""
def __init__(self, left_on, left_by, right_on, right_by, tolerance): ...

class Declaration(_Weakrefable):
"""
Helper class for declaring the nodes of an ExecPlan.

A Declaration represents an unconstructed ExecNode, and potentially
more since its inputs may also be Declarations or when constructed
with ``from_sequence``.

The possible ExecNodes to use are registered with a name,
the "factory name", and need to be specified using this name, together
with its corresponding ExecNodeOptions subclass.

Parameters
----------
factory_name : str
The ExecNode factory name, such as "table_source", "filter",
"project" etc. See the ExecNodeOptions subclasses for the exact
factory names to use.
options : ExecNodeOptions
Corresponding ExecNodeOptions subclass (matching the factory name).
inputs : list of Declaration, optional
Input nodes for this declaration. Optional if the node is a source
node, or when the declaration gets combined later with
``from_sequence``.

Returns
-------
Declaration
"""
def __init__(self, factory_name, options: ExecNodeOptions, inputs=None): ...
@staticmethod
def from_sequence(decls):
"""
Convenience factory for the common case of a simple sequence of nodes.

Each of the declarations will be appended to the inputs of the
subsequent declaration, and the final modified declaration will
be returned.

Parameters
----------
decls : list of Declaration

Returns
-------
Declaration
"""
def __str__(self): ...
def __repr__(self): ...
def to_table(self, use_threads: bool=True):
"""
Run the declaration and collect the results into a table.

This method will implicitly add a sink node to the declaration
to collect results into a table. It will then create an ExecPlan
from the declaration, start the exec plan, block until the plan
has finished, and return the created table.

Parameters
----------
use_threads : bool, default True
If set to False, then all CPU work will be done on the calling
thread. I/O tasks will still happen on the I/O executor
and may be multi-threaded (but should not use significant CPU
resources).

Returns
-------
pyarrow.Table
"""
def to_reader(self, use_threads: bool=True):
"""Run the declaration and return results as a RecordBatchReader.

For details about the parameters, see `to_table`.

Returns
-------
pyarrow.RecordBatchReader
"""
86 changes: 86 additions & 0 deletions python/pyarrow/_azurefs.pyi
Original file line number Diff line number Diff line change
@@ -0,0 +1,86 @@
# This file was generated by stubgen-pyx v0.2.23 from python/pyarrow/_azurefs.pyx

from pyarrow._fs import FileSystem
from pyarrow.includes.libarrow_fs import *


class AzureFileSystem(FileSystem):
"""
Azure Blob Storage backed FileSystem implementation

This implementation supports flat namespace and hierarchical namespace (HNS) a.k.a.
Data Lake Gen2 storage accounts. HNS will be automatically detected and HNS specific
features will be used when they provide a performance advantage. Azurite emulator is
also supported. Note: `/` is the only supported delimiter.

The storage account is considered the root of the filesystem. When enabled, containers
will be created or deleted during relevant directory operations. Obviously, this also
requires authentication with the additional permissions.

By default `DefaultAzureCredential <https://github.com/Azure/azure-sdk-for-cpp/blob/main/sdk/identity/azure-identity/README.md#defaultazurecredential>`__
is used for authentication. This means it will try several types of authentication
and go with the first one that works. If any authentication parameters are provided when
initialising the FileSystem, they will be used instead of the default credential.

Parameters
----------
account_name : str
Azure Blob Storage account name. This is the globally unique identifier for the
storage account.
account_key : str, default None
Account key of the storage account. If sas_token and account_key are None the
default credential will be used. The parameters account_key and sas_token are
mutually exclusive.
blob_storage_authority : str, default None
hostname[:port] of the Blob Service. Defaults to `.blob.core.windows.net`. Useful
for connecting to a local emulator, like Azurite.
blob_storage_scheme : str, default None
Either `http` or `https`. Defaults to `https`. Useful for connecting to a local
emulator, like Azurite.
client_id : str, default None
The client ID (Application ID) for Azure Active Directory authentication.
Its interpretation depends on the credential type being used:

- For `ClientSecretCredential`: It is the Application (client) ID of your
registered Azure AD application (Service Principal). It must be provided
together with `tenant_id` and `client_secret` to use ClientSecretCredential.
- For `ManagedIdentityCredential`: It is the client ID of a specific
user-assigned managed identity. This is only necessary if you are using a
user-assigned managed identity and need to explicitly specify which one
(e.g., if the resource has multiple user-assigned identities). For
system-assigned managed identities, this parameter is typically not required.

client_secret : str, default None
Client secret for Azure Active Directory authentication. Must be provided together
with `tenant_id` and `client_id` to use ClientSecretCredential.
dfs_storage_authority : str, default None
hostname[:port] of the Data Lake Gen 2 Service. Defaults to
`.dfs.core.windows.net`. Useful for connecting to a local emulator, like Azurite.
dfs_storage_scheme : str, default None
Either `http` or `https`. Defaults to `https`. Useful for connecting to a local
emulator, like Azurite.
sas_token : str, default None
SAS token for the storage account, used as an alternative to account_key. If sas_token
and account_key are None the default credential will be used. The parameters
account_key and sas_token are mutually exclusive.
tenant_id : str, default None
Tenant ID for Azure Active Directory authentication. Must be provided together with
`client_id` and `client_secret` to use ClientSecretCredential.

Examples
--------
>>> from pyarrow import fs
>>> azure_fs = fs.AzureFileSystem(account_name='myaccount')
>>> azurite_fs = fs.AzureFileSystem(
... account_name='devstoreaccount1',
... account_key='Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==',
... blob_storage_authority='127.0.0.1:10000',
... dfs_storage_authority='127.0.0.1:10000',
... blob_storage_scheme='http',
... dfs_storage_scheme='http',
... )

For usage of the methods see examples for :func:`~pyarrow.fs.LocalFileSystem`.
"""
def __init__(self, account_name, *, account_key=None, blob_storage_authority=None, blob_storage_scheme=None, client_id=None, client_secret=None, dfs_storage_authority=None, dfs_storage_scheme=None, sas_token=None, tenant_id=None): ...
def __reduce__(self): ...
Loading
Loading