from __future__ import annotations

import contextlib
from collections.abc import Sequence
from pathlib import Path
from typing import IO, TYPE_CHECKING, Any, Literal

import polars._reexport as pl
import polars.functions as F
from polars._utils.expired import RemovedParameter, RenamedParameter, removed_parameters
from polars._utils.unstable import issue_unstable_warning
from polars._utils.various import (
    _process_null_values,
    is_path_or_str_sequence,
    normalize_filepath,
    qualified_type_name,
)
from polars._utils.wrap import wrap_ldf
from polars.datatypes import N_INFER_DEFAULT, parse_into_dtype
from polars.io._utils import (
    parse_columns_arg,
    parse_row_index_args,
    prepare_file_arg,
)
from polars.io.cloud.credential_provider._builder import (
    _init_credential_provider_builder,
)
from polars.io.csv._utils import _check_arg_is_1byte, _update_columns

with contextlib.suppress(ImportError):  # Module not available when building docs
    from polars._plr import PyLazyFrame

if TYPE_CHECKING:
    from collections.abc import Callable

    from polars import DataFrame, LazyFrame
    from polars._typing import (
        CsvEncoding,
        PolarsDataType,
        SchemaDict,
        StorageOptionsDict,
    )
    from polars.io.cloud import CredentialProviderFunction


_N_INFER_FILES_DEFAULT = 10


@removed_parameters(
    RenamedParameter(
        name="dtypes",
        new_name="schema_overrides",
        deprecated_in="0.20.31",
        removed_in="2.0",
    ),
    RenamedParameter(
        name="row_count_name",
        new_name="row_index_name",
        deprecated_in="0.20.4",
        removed_in="2.0",
    ),
    RenamedParameter(
        name="row_count_offset",
        new_name="row_index_offset",
        deprecated_in="0.20.4",
        removed_in="2.0",
    ),
    RemovedParameter(
        name="rechunk",
        removed_in="2.0",
        hint="call `rechunk()` on the resulting Dataframe if you need contiguous memory.",
    ),
)
def read_csv(
    source: (
        str
        | Path
        | IO[str]
        | IO[bytes]
        | bytes
        | list[str]
        | list[Path]
        | list[IO[str]]
        | list[IO[bytes]]
        | list[bytes]
    ),
    *,
    columns: Sequence[int] | Sequence[str] | None = None,
    has_header: bool = True,
    separator: str = ",",
    comment_prefix: str | None = None,
    quote_char: str | None = '"',
    skip_rows: int = 0,
    skip_lines: int = 0,
    schema: SchemaDict | None = None,
    schema_overrides: SchemaDict | Sequence[PolarsDataType] | None = None,
    null_values: str | Sequence[str] | dict[str, str] | None = None,
    empty_string_is_null: bool = True,
    ignore_errors: bool = False,
    with_column_names: Callable[[list[str]], list[str]] | None = None,
    infer_schema: bool = True,
    infer_schema_length: int | None = N_INFER_DEFAULT,
    infer_schema_files: int = _N_INFER_FILES_DEFAULT,
    n_rows: int | None = None,
    encoding: CsvEncoding | str = "utf8",
    low_memory: bool = False,
    skip_rows_after_header: int = 0,
    row_index_name: str | None = None,
    row_index_offset: int = 0,
    try_parse_dates: bool = False,
    eol_char: str = "\n",
    new_columns: Sequence[str] | None = None,
    truncate_ragged_lines: bool | None = None,
    raise_if_empty: bool | None = None,
    decimal_comma: bool = False,
    glob: bool = True,
    storage_options: StorageOptionsDict | None = None,
    credential_provider: CredentialProviderFunction | Literal["auto"] | None = "auto",
    include_file_paths: str | None = None,
    extra_columns: Literal["ignore", "raise"] | None = None,
    missing_columns: Literal["insert", "raise"] | None = None,
    use_pyarrow: bool = False,
) -> DataFrame:
    r"""
    Read a CSV file into a DataFrame.

    Polars expects CSV data to strictly conform to RFC 4180, unless documented
    otherwise. Malformed data, though common, may lead to undefined behavior.

    .. versionchanged:: 0.20.31
        The `dtypes` parameter was renamed `schema_overrides`.
    .. versionchanged:: 0.20.4
        * The `row_count_name` parameter was renamed `row_index_name`.
        * The `row_count_offset` parameter was renamed `row_index_offset`.

    Parameters
    ----------
    source
        Path(s) to a file or a file-like object (by "file-like object" we refer to
        objects that have a `read()` method, such as a file handler like the builtin
        `open` function, or a `BytesIO` instance). If `fsspec` is installed, it might be
        used to open remote files. Compressed files (gzip and zstd) are supported when
        reading from a path or a file-like object. For file-like objects, the stream
        position may not be updated accordingly after reading.
    columns
        Columns to select. Accepts a list of column indices (starting
        at zero) or a list of column names.
    has_header
        Indicate if the first row of the dataset is a header or not. If set to False,
        column names will be autogenerated in the following format: `column_x`, with
        `x` being an enumeration over every column in the dataset, starting at 0.
    separator
        Single byte character to use as separator in the file.
    comment_prefix
        A string used to indicate the start of a comment line. Comment lines are skipped
        during parsing. Common examples of comment prefixes are `#` and `//`.
    quote_char
        Single byte character used for csv quoting, default = `"`.
        Set to None to turn off special handling and escaping of quotes.
    skip_rows
        Start reading after ``skip_rows`` rows. The header will be parsed at this
        offset. Note that we respect CSV escaping/comments when skipping rows.
        If you want to skip by newline char only, use `skip_lines`.
    skip_lines
        Start reading after `skip_lines` lines. The header will be parsed at this
        offset. Note that CSV escaping will not be respected when skipping lines.
        If you want to skip valid CSV rows, use ``skip_rows``.
    schema
        Provide the schema. This means that polars doesn't do schema inference.
        This argument expects the complete schema, whereas `schema_overrides` can be
        used to partially overwrite a schema. Note that the order of the columns in
        the provided `schema` must match the order of the columns in the CSV being read.
    schema_overrides
        Overwrite dtypes during inference; should be a {colname:dtype,} dict or,
        if providing a list of strings to `new_columns`, a list of dtypes of
        the same length.
    null_values
        Values to interpret as null values. You can provide a:

        - `str`: All values equal to this string will be null.
        - `List[str]`: All values equal to any string in this list will be null.
        - `Dict[str, str]`: A dictionary that maps column name to a
          null value string.

    empty_string_is_null
        By default a missing string value is considered to be null. If
        `empty_string_is_null` is set to False, missing string values are considered to
        decoded as empty strings.
    ignore_errors
        Try to keep reading lines if some lines yield errors.
        First try `infer_schema=False` to read all columns as
        `pl.String` to check which values might cause an issue.
    with_column_names
        Apply a function over the column names just in time (when they are determined);
        this function will receive (and should return) a list of column names.
    infer_schema
        When `True`, the schema is inferred from the data using the first
        `infer_schema_length` rows.
        When `False`, the schema is not inferred and will be `pl.String` if not
        specified in `schema` or `schema_overrides`.
    infer_schema_length
        The maximum number of rows to scan for schema inference.
        If set to `None`, the full data will be scanned into memory
        **(this is slow)**.
        Alternatively set `infer_schema=False` to read all columns as
        `pl.String`.
    infer_schema_files
        How many files to use when inferring schema.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.
    n_rows
        Stop reading from CSV file after reading `n_rows`.
    encoding : {'utf8', 'utf8-lossy', 'windows-1252', 'windows-1252-lossy', ...}
        Lossy means that invalid utf8 values are replaced with `�`
        characters. Defaults to "utf8".
    low_memory
        Reduce memory pressure at the expense of performance.
    skip_rows_after_header
        Skip this number of rows when the header is parsed.
    row_index_name
        If not None, this will insert a row index column with the given name into
        the DataFrame.
    row_index_offset
        Offset to start the row index column (only used if the name is set).
    try_parse_dates
        Try to automatically parse dates. Most ISO8601-like formats
        can be inferred, as well as a handful of others. If this does not succeed,
        the column remains of data type `pl.String`.
    eol_char
        Single byte end of line character (default: ``\n``). When encountering a file
        with windows line endings (``\r\n``), one can go with the default ``\n``. The
        extra ``\r`` will be removed when processed.
    new_columns
        Provide an explicit list of string column names to use (for example, when
        scanning a headerless CSV file). If the given list is shorter than the width of
        the DataFrame the remaining columns will have their original name.
    raise_if_empty
        Control behavior when scanning empty files:

        * `True`: Raise a `NoDataError`.
        * `False`: Return 0 rows.

        This defaults to `False` if `has_header=False` and a `schema` is specified,
        and `True` otherwise.
    truncate_ragged_lines
        Truncate lines that are longer than the schema.
    decimal_comma
        Parse floats using a comma as the decimal separator instead of a period.
    glob
        Expand path given via globbing rules.
    storage_options
        Options that indicate how to connect to a cloud provider.

        The cloud providers currently supported are AWS, GCP, and Azure.
        See supported keys here:

        * `aws <https://docs.rs/object_store/latest/object_store/aws/enum.AmazonS3ConfigKey.html>`_
        * `gcp <https://docs.rs/object_store/latest/object_store/gcp/enum.GoogleConfigKey.html>`_
        * `azure <https://docs.rs/object_store/latest/object_store/azure/enum.AzureConfigKey.html>`_
        * Hugging Face (`hf://`): Accepts an API key under the `token` parameter: \
          `{'token': '...'}`, or by setting the `HF_TOKEN` environment variable.

        If `storage_options` is not provided, Polars will try to infer the information
        from environment variables.
    credential_provider
        Provide a function that can be called to provide cloud storage
        credentials. The function is expected to return a dictionary of
        credential keys along with an optional credential expiry time.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.
    include_file_paths
        Include the path of the source file(s) as a column with this name.
    extra_columns
        Configuration for behavior when extra columns outside of the
        defined schema are encountered in the data:

        * `ignore`: Silently ignores.
        * `raise`: Raises an error.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.

    missing_columns
        Configuration for behavior when columns defined in the schema are
        missing from the data:

        * ``"insert"``: Insert the missing columns with NULL values.
        * ``"raise"``: Raise an error.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.
    use_pyarrow
        Try to use pyarrow's native CSV parser. This will always
        parse dates, even if `try_parse_dates=False`.
        This is not always possible. The set of arguments given to
        this function determines if it is possible to use pyarrow's
        native parser. Note that pyarrow and polars may have a
        different strategy regarding type inference.

    Returns
    -------
    DataFrame

    See Also
    --------
    scan_csv : Lazily read from a CSV file or multiple files via glob patterns.

    Warnings
    --------
    Calling `read_csv().lazy()` is an antipattern as this forces Polars to materialize
    a full csv file and therefore cannot push any optimizations into the reader.
    Therefore always prefer `scan_csv` if you want to work with `LazyFrame` s.

    Notes
    -----
    If the schema is inferred incorrectly (e.g. as `pl.Int64` instead of `pl.Float64`),
    try to increase the number of lines used to infer the schema with
    `infer_schema_length` or override the inferred dtype for those columns with
    `schema_overrides`.

    Examples
    --------
    >>> pl.read_csv("data.csv", separator="|")  # doctest: +SKIP

    Demonstrate use against a BytesIO object, parsing string dates.

    >>> from io import BytesIO
    >>> data = BytesIO(
    ...     b"ID,Name,Birthday\n"
    ...     b"1,Alice,1995-07-12\n"
    ...     b"2,Bob,1990-09-20\n"
    ...     b"3,Charlie,2002-03-08\n"
    ... )
    >>> pl.read_csv(data, try_parse_dates=True)
    shape: (3, 3)
    ┌─────┬─────────┬────────────┐
    │ ID  ┆ Name    ┆ Birthday   │
    │ --- ┆ ---     ┆ ---        │
    │ i64 ┆ str     ┆ date       │
    ╞═════╪═════════╪════════════╡
    │ 1   ┆ Alice   ┆ 1995-07-12 │
    │ 2   ┆ Bob     ┆ 1990-09-20 │
    │ 3   ┆ Charlie ┆ 2002-03-08 │
    └─────┴─────────┴────────────┘
    """
    projection, column_names = parse_columns_arg(columns)
    del columns
    row_index = parse_row_index_args(row_index_name, row_index_offset)
    del row_index_name
    del row_index_offset

    _check_arg_is_1byte("separator", separator, can_be_empty=False)
    _check_arg_is_1byte("quote_char", quote_char, can_be_empty=True)
    _check_arg_is_1byte("eol_char", eol_char, can_be_empty=False)

    storage_options = storage_options or {}

    if column_names and not has_header:
        for column in column_names:
            if not column.startswith("column_"):
                msg = (
                    "specified column names do not start with 'column_',"
                    " but autogenerated header names were requested"
                )
                raise ValueError(msg)

    if schema_overrides is not None and not isinstance(
        schema_overrides, (dict, Sequence)
    ):
        msg = "`schema_overrides` should be of type list or dict"
        raise TypeError(msg)

    if raise_if_empty is None:
        raise_if_empty = schema is None or has_header

    if (
        use_pyarrow
        and schema_overrides is None
        and n_rows is None
        and not low_memory
        and null_values is None
        and with_column_names is None
        and credential_provider == "auto"
        and include_file_paths is None
    ):
        include_columns: Sequence[str] | None = None
        if column_names:
            if not has_header:
                # Convert 'column_0', 'column_1', ... column names to 'f0', 'f1', ...
                # column names for pyarrow, if CSV file does not contain a header.
                include_columns = [f"f{int(column[7:])}" for column in column_names]
            else:
                include_columns = column_names

        if not column_names and projection:
            # User selected columns by positional index (e.g. `columns=[0]`).
            if not has_header:
                # pyarrow auto-generates names 'f0', 'f1', ... when there is no
                # header, so index N maps to name 'fN'.
                include_columns = [f"f{column_idx}" for column_idx in projection]
            else:
                # With a header, real names come from row 1 and aren't known
                # until after the read. Leave the filter off and slice the
                # Table by position later.
                include_columns = None

        with prepare_file_arg(  # type: ignore[no-matching-overload]
            source,  # type: ignore[arg-type]
            encoding=None,
            use_pyarrow=True,
            raise_if_empty=raise_if_empty,
            storage_options=storage_options,
        ) as data:
            import pyarrow as pa
            import pyarrow.csv

            try:
                tbl = pa.csv.read_csv(
                    data,
                    pa.csv.ReadOptions(
                        skip_rows=skip_rows,
                        skip_rows_after_names=skip_rows_after_header,
                        autogenerate_column_names=not has_header,
                        encoding=encoding,
                    ),
                    pa.csv.ParseOptions(
                        delimiter=separator,
                        quote_char=quote_char if quote_char else False,
                        double_quote=quote_char is not None and quote_char == '"',
                    ),
                    pa.csv.ConvertOptions(
                        column_types=None,
                        include_columns=include_columns,
                        include_missing_columns=ignore_errors,
                    ),
                )
            except pa.ArrowInvalid as err:
                if raise_if_empty or "Empty CSV" not in str(err):
                    raise
                return pl.DataFrame()

        if not has_header:
            # Rename 'f0', 'f1', ... columns names autogenerated by pyarrow
            # to 'column_0', 'column_1', ...
            tbl = tbl.rename_columns(
                [f"column_{int(column[1:])}" for column in tbl.column_names]
            )
        elif not column_names and projection:
            # User selected columns by positional index (e.g. `columns=[0, 2]`).
            # pyarrow's include_columns only accepts names, so the read above
            # fetched every column; pick out the requested positions now.
            tbl = tbl.select(list(projection))

        df = pl.DataFrame._from_arrow(tbl)

        if new_columns:
            return _update_columns(df, new_columns)

        if row_index is not None:
            name, offset = row_index
            df = df.with_row_index(name, offset)

        return df

    def _scan(source: Any, encoding: CsvEncoding | str) -> pl.LazyFrame:
        return scan_csv(
            source,
            has_header=has_header,
            separator=separator,
            comment_prefix=comment_prefix,
            quote_char=quote_char,
            skip_rows=skip_rows,
            skip_lines=skip_lines,
            schema=schema,
            schema_overrides=schema_overrides,  # type: ignore[arg-type]
            null_values=null_values,
            empty_string_is_null=empty_string_is_null,
            ignore_errors=ignore_errors,
            with_column_names=with_column_names,
            infer_schema=infer_schema,
            infer_schema_length=infer_schema_length,
            infer_schema_files=infer_schema_files,
            n_rows=n_rows,
            encoding=encoding,  # type: ignore[arg-type]
            low_memory=low_memory,
            skip_rows_after_header=skip_rows_after_header,
            try_parse_dates=try_parse_dates,
            eol_char=eol_char,
            new_columns=new_columns,
            truncate_ragged_lines=truncate_ragged_lines,
            raise_if_empty=raise_if_empty,
            decimal_comma=decimal_comma,
            glob=glob,
            storage_options=storage_options,
            credential_provider=credential_provider,
            include_file_paths=include_file_paths,
            extra_columns=extra_columns,
            missing_columns=missing_columns,
        )

    encoding_supported_in_lazy = encoding in {"utf8", "utf8-lossy"}

    if not encoding_supported_in_lazy:
        with prepare_file_arg(  # type: ignore[no-matching-overload]
            source,  # type: ignore[arg-type]
            encoding=encoding,
            use_pyarrow=False,
            raise_if_empty=raise_if_empty,
            storage_options=storage_options,
        ) as data:
            lf = _scan(data, encoding="utf8")
    else:
        lf = _scan(source, encoding=encoding)

    if column_names is not None:
        lf = lf.select(column_names)
    elif projection is not None:
        lf = lf.select(F.nth(projection))

    if row_index is not None:
        name, offset = row_index
        lf = lf.with_row_index(name, offset)

    return lf._collect_eager()


@removed_parameters(
    RenamedParameter(
        name="dtypes",
        new_name="schema_overrides",
        deprecated_in="0.20.31",
        removed_in="2.0",
    ),
    RenamedParameter(
        name="row_count_name",
        new_name="row_index_name",
        deprecated_in="0.20.4",
        removed_in="2.0",
    ),
    RenamedParameter(
        name="row_count_offset",
        new_name="row_index_offset",
        deprecated_in="0.20.4",
        removed_in="2.0",
    ),
    RemovedParameter(
        name="rechunk",
        removed_in="2.0",
        hint="call `rechunk()` on the resulting Dataframe if you need contiguous memory.",
    ),
)
def scan_csv(
    source: (
        str
        | Path
        | IO[str]
        | IO[bytes]
        | bytes
        | list[str]
        | list[Path]
        | list[IO[str]]
        | list[IO[bytes]]
        | list[bytes]
    ),
    *,
    has_header: bool = True,
    separator: str = ",",
    comment_prefix: str | None = None,
    quote_char: str | None = '"',
    skip_rows: int = 0,
    skip_lines: int = 0,
    schema: SchemaDict | None = None,
    schema_overrides: SchemaDict | Sequence[PolarsDataType] | None = None,
    null_values: str | Sequence[str] | dict[str, str] | None = None,
    empty_string_is_null: bool = True,
    ignore_errors: bool = False,
    with_column_names: Callable[[list[str]], list[str]] | None = None,
    infer_schema: bool = True,
    infer_schema_length: int | None = N_INFER_DEFAULT,
    infer_schema_files: int = _N_INFER_FILES_DEFAULT,
    n_rows: int | None = None,
    encoding: CsvEncoding = "utf8",
    low_memory: bool = False,
    skip_rows_after_header: int = 0,
    row_index_name: str | None = None,
    row_index_offset: int = 0,
    try_parse_dates: bool = False,
    eol_char: str = "\n",
    new_columns: Sequence[str] | None = None,
    truncate_ragged_lines: bool | None = None,
    raise_if_empty: bool | None = None,
    decimal_comma: bool = False,
    glob: bool = True,
    storage_options: StorageOptionsDict | None = None,
    credential_provider: CredentialProviderFunction | Literal["auto"] | None = "auto",
    include_file_paths: str | None = None,
    extra_columns: Literal["ignore", "raise"] | None = None,
    missing_columns: Literal["insert", "raise"] | None = None,
) -> LazyFrame:
    r"""
    Lazily read from a CSV file or multiple files via glob patterns.

    This allows the query optimizer to push down predicates and
    projections to the scan level, thereby potentially reducing
    memory overhead.

    .. engine-support:: in-memory, streaming, distributed

    Parameters
    ----------
    source
        Path(s) to a file or a file-like object (by "file-like object" we refer to
        objects that have a `read()` method, such as a file handler like the builtin
        `open` function, or a `BytesIO` instance). If `fsspec` is installed, it might be
        used to open remote files. Compressed files (gzip and zstd) are supported when
        reading from a path or a file-like object. For file-like objects, the stream
        position may not be updated accordingly after reading.
    has_header
        Indicate if the first row of the dataset is a header or not. If set to False,
        column names will be autogenerated in the following format: `column_x`, with
        `x` being an enumeration over every column in the dataset, starting at 0.
    separator
        Single byte character to use as separator in the file.
    comment_prefix
        A string used to indicate the start of a comment line. Comment lines are skipped
        during parsing. Common examples of comment prefixes are `#` and `//`.
    quote_char
        Single byte character used for csv quoting, default = `"`.
        Set to None to turn off special handling and escaping of quotes.
    skip_rows
        Start reading after ``skip_rows`` rows. The header will be parsed at this
        offset. Note that we respect CSV escaping/comments when skipping rows.
        If you want to skip by newline char only, use `skip_lines`.
    skip_lines
        Start reading after `skip_lines` lines. The header will be parsed at this
        offset. Note that CSV escaping will not be respected when skipping lines.
        If you want to skip valid CSV rows, use ``skip_rows``.
    schema
        Provide the schema. This means that polars doesn't do schema inference.
        This argument expects the complete schema, whereas `schema_overrides` can be
        used to partially overwrite a schema. Note that the order of the columns in
        the provided `schema` must match the order of the columns in the CSV being read.
    schema_overrides
        Overwrite dtypes during inference; should be a {colname:dtype,} dict or,
        if providing a list of strings to `new_columns`, a list of dtypes of
        the same length.
    null_values
        Values to interpret as null values. You can provide a:

        - `str`: All values equal to this string will be null.
        - `List[str]`: All values equal to any string in this list will be null.
        - `Dict[str, str]`: A dictionary that maps column name to a
          null value string.

    empty_string_is_null
        By default a missing string value is considered to be null. If
        `empty_string_is_null` is set to False, missing string values are considered to
        decoded as empty strings.
    ignore_errors
        Try to keep reading lines if some lines yield errors.
        First try `infer_schema=False` to read all columns as
        `pl.String` to check which values might cause an issue.
    with_column_names
        Apply a function over the column names just in time (when they are determined);
        this function will receive (and should return) a list of column names.
    infer_schema
        When `True`, the schema is inferred from the data using the first
        `infer_schema_length` rows.
        When `False`, the schema is not inferred and will be `pl.String` if not
        specified in `schema` or `schema_overrides`.
    infer_schema_length
        The maximum number of rows to scan for schema inference. This applies
        individually to each file included according to `infer_schema_files`.
        If set to `None`, the full data will be scanned into memory
        **(this is slow)**.
        Alternatively set `infer_schema=False` to read all columns as
        `pl.String`.
    infer_schema_files
        How many files to use when inferring schema.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.
    n_rows
        Stop reading from CSV file after reading `n_rows`.
    encoding : {'utf8', 'utf8-lossy'}
        Lossy means that invalid utf8 values are replaced with `�`
        characters. Defaults to "utf8".
    low_memory
        Reduce memory pressure at the expense of performance.
    skip_rows_after_header
        Skip this number of rows when the header is parsed.
    row_index_name
        If not None, this will insert a row index column with the given name into
        the DataFrame.
    row_index_offset
        Offset to start the row index column (only used if the name is set).
    try_parse_dates
        Try to automatically parse dates. Most ISO8601-like formats
        can be inferred, as well as a handful of others. If this does not succeed,
        the column remains of data type `pl.String`.
    eol_char
        Single byte end of line character (default: ``\n``). When encountering a file
        with windows line endings (``\r\n``), one can go with the default ``\n``. The
        extra ``\r`` will be removed when processed.
    new_columns
        Provide an explicit list of string column names to use (for example, when
        scanning a headerless CSV file). If the given list is shorter than the width of
        the DataFrame the remaining columns will have their original name.
    raise_if_empty
        Control behavior when scanning empty files:

        * `True`: Raise a `NoDataError`.
        * `False`: Return 0 rows.

        This defaults to `False` if `has_header=False` and a `schema` is specified,
        and `True` otherwise.
    truncate_ragged_lines
        Truncate lines that are longer than the schema.
    decimal_comma
        Parse floats using a comma as the decimal separator instead of a period.
    glob
        Expand path given via globbing rules.
    storage_options
        Options that indicate how to connect to a cloud provider.

        The cloud providers currently supported are AWS, GCP, and Azure.
        See supported keys here:

        * `aws <https://docs.rs/object_store/latest/object_store/aws/enum.AmazonS3ConfigKey.html>`_
        * `gcp <https://docs.rs/object_store/latest/object_store/gcp/enum.GoogleConfigKey.html>`_
        * `azure <https://docs.rs/object_store/latest/object_store/azure/enum.AzureConfigKey.html>`_
        * Hugging Face (`hf://`): Accepts an API key under the `token` parameter: \
          `{'token': '...'}`, or by setting the `HF_TOKEN` environment variable.

        If `storage_options` is not provided, Polars will try to infer the information
        from environment variables.
    credential_provider
        Provide a function that can be called to provide cloud storage
        credentials. The function is expected to return a dictionary of
        credential keys along with an optional credential expiry time.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.
    include_file_paths
        Include the path of the source file(s) as a column with this name.
    extra_columns
        Configuration for behavior when extra columns outside of the
        defined schema are encountered in the data:

        * `ignore`: Silently ignores.
        * `raise`: Raises an error.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.

    missing_columns
        Configuration for behavior when columns defined in the schema are
        missing from the data:

        * ``"insert"``: Insert the missing columns with NULL values.
        * ``"raise"``: Raise an error.

        .. warning::
            This functionality is considered **unstable**. It may be changed
            at any point without it being considered a breaking change.

    Returns
    -------
    LazyFrame

    See Also
    --------
    read_csv : Read a CSV file into a DataFrame.

    Examples
    --------
    >>> import pathlib
    >>>
    >>> (
    ...     pl.scan_csv("my_long_file.csv")  # lazy, doesn't do a thing
    ...     .select(
    ...         ["a", "c"]
    ...     )  # select only 2 columns (other columns will not be read)
    ...     .filter(
    ...         pl.col("a") > 10
    ...     )  # the filter is pushed down the scan, so less data is read into memory
    ...     .head(100)  # constrain number of returned results to 100
    ... )  # doctest: +SKIP

    We can use `with_column_names` to modify the header before scanning:

    >>> df = pl.DataFrame(
    ...     {"BrEeZaH": [1, 2, 3, 4], "LaNgUaGe": ["is", "hard", "to", "read"]}
    ... )
    >>> path: pathlib.Path = dirpath / "mydf.csv"
    >>> df.write_csv(path)
    >>> pl.scan_csv(
    ...     path, with_column_names=lambda cols: [col.lower() for col in cols]
    ... ).collect()
    shape: (4, 2)
    ┌─────────┬──────────┐
    │ breezah ┆ language │
    │ ---     ┆ ---      │
    │ i64     ┆ str      │
    ╞═════════╪══════════╡
    │ 1       ┆ is       │
    │ 2       ┆ hard     │
    │ 3       ┆ to       │
    │ 4       ┆ read     │
    └─────────┴──────────┘

    You can also simply replace column names (or provide them if the file has none)
    by passing a list of new column names to the `new_columns` parameter:

    >>> df.write_csv(path)
    >>> pl.scan_csv(
    ...     path,
    ...     new_columns=["idx", "txt"],
    ...     schema_overrides=[pl.UInt16, pl.String],
    ... ).collect()
    shape: (4, 2)
    ┌─────┬──────┐
    │ idx ┆ txt  │
    │ --- ┆ ---  │
    │ u16 ┆ str  │
    ╞═════╪══════╡
    │ 1   ┆ is   │
    │ 2   ┆ hard │
    │ 3   ┆ to   │
    │ 4   ┆ read │
    └─────┴──────┘
    """
    if schema_overrides is not None and not isinstance(
        schema_overrides, (dict, Sequence)
    ):
        msg = "`schema_overrides` should be of type list or dict"
        raise TypeError(msg)

    if new_columns is not None and with_column_names is not None:
        msg = (
            "cannot set both `with_column_names` and `new_columns`; mutually exclusive"
        )
        raise ValueError(msg)

    _check_arg_is_1byte("separator", separator, can_be_empty=False)
    _check_arg_is_1byte("quote_char", quote_char, can_be_empty=True)

    if raise_if_empty is None:
        raise_if_empty = schema is None or has_header

    if isinstance(source, (str, Path)):
        source = normalize_filepath(source, check_not_directory=False)
    elif is_path_or_str_sequence(source, allow_str=False):
        source = [
            normalize_filepath(source, check_not_directory=False) for source in source
        ]

    if not infer_schema:
        infer_schema_length = 0

    if extra_columns is not None:
        msg = "The `extra_columns` parameter of `scan_csv` is considered unstable."
        issue_unstable_warning(msg)
    else:
        extra_columns = "raise"

    if extra_columns == "ignore" and truncate_ragged_lines is False:
        msg = "cannot set truncate_ragged_lines=False with extra_columns='ignore'"
        raise ValueError(msg)

    if truncate_ragged_lines is None:
        truncate_ragged_lines = extra_columns == "ignore"

    if missing_columns is not None:
        msg = "The `missing_columns` parameter of `scan_csv` is considered unstable."
        issue_unstable_warning(msg)

    credential_provider_builder = _init_credential_provider_builder(
        credential_provider, source, storage_options, "scan_csv"
    )
    del credential_provider

    dtype_list: list[tuple[str, PolarsDataType]] | None = None
    dtype_slice: Sequence[PolarsDataType] | None = None
    if schema_overrides is not None:
        if isinstance(schema_overrides, dict):
            dtype_list = []
            for k, v in schema_overrides.items():
                dtype_list.append((k, parse_into_dtype(v)))
        elif isinstance(schema_overrides, Sequence):
            dtype_slice = [parse_into_dtype(v) for v in schema_overrides]
        else:
            msg = f"`schema_overrides` should be of type list or dict, got {qualified_type_name(schema_overrides)!r}"
            raise TypeError(msg)
    processed_null_values = _process_null_values(null_values)

    if isinstance(source, list):
        sources = source
        source = None  # type: ignore[assignment]
    else:
        sources = []

    pylf = PyLazyFrame.new_from_csv(
        source,
        sources,
        separator=separator,
        has_header=has_header,
        ignore_errors=ignore_errors,
        skip_rows=skip_rows,
        skip_lines=skip_lines,
        n_rows=n_rows,
        overwrite_dtype=dtype_list,
        overwrite_dtype_slice=dtype_slice,
        low_memory=low_memory,
        comment_prefix=comment_prefix,
        quote_char=quote_char,
        null_values=processed_null_values,
        empty_string_is_null=empty_string_is_null,
        infer_schema_length=infer_schema_length,
        infer_schema_files=infer_schema_files,
        new_columns=new_columns,
        with_schema_modify=with_column_names,
        rechunk=False,
        skip_rows_after_header=skip_rows_after_header,
        encoding=encoding,
        row_index=parse_row_index_args(row_index_name, row_index_offset),
        try_parse_dates=try_parse_dates,
        eol_char=eol_char,
        raise_if_empty=raise_if_empty,
        truncate_ragged_lines=truncate_ragged_lines,
        decimal_comma=decimal_comma,
        glob=glob,
        schema=schema,
        cloud_options=storage_options,
        credential_provider=credential_provider_builder,
        include_file_paths=include_file_paths,
        extra_columns=extra_columns,
        missing_columns=missing_columns,
    )

    return wrap_ldf(pylf)
