From e1b7837e27aab300ebc78000d9fad9f9b35c1e26 Mon Sep 17 00:00:00 2001 From: Qinyi Ding Date: Thu, 1 Oct 2026 15:11:07 -0700 Subject: [PATCH 1/4] [DOC-5942] Consolidate API reference clarity and source-link fixes Restore missing reference coverage and pandas export documentation, and clarify API behavior with validated examples. Generated with [Snowflake CoCo](https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code) Co-authored-by: Snowflake CoCo --- docs/source/conf.py | 25 ++-- docs/source/modin/index.rst | 8 +- docs/source/modin/interoperability.rst | 16 +++ docs/source/modin/supported/index.rst | 9 +- docs/source/snowpark/dataframe.rst | 3 + docs/source/snowpark/files.rst | 3 +- docs/source/snowpark/io.rst | 2 +- docs/source/snowpark/window.rst | 1 + src/snowflake/snowpark/dataframe.py | 62 ++++++++-- src/snowflake/snowpark/dataframe_reader.py | 20 ++++ .../snowpark/dataframe_stat_functions.py | 4 +- src/snowflake/snowpark/exceptions.py | 17 +++ src/snowflake/snowpark/file_operation.py | 6 + src/snowflake/snowpark/files.py | 62 +++++++++- src/snowflake/snowpark/functions.py | 40 ++++--- .../modin/plugin/extensions/pd_extensions.py | 18 ++- src/snowflake/snowpark/session.py | 50 +++++++- tests/unit/modin/test_docstrings.py | 8 ++ tests/unit/test_documentation.py | 107 ++++++++++++++++++ 19 files changed, 408 insertions(+), 53 deletions(-) create mode 100644 tests/unit/test_documentation.py diff --git a/docs/source/conf.py b/docs/source/conf.py index 983d2136d8..c51793fe7d 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -12,6 +12,7 @@ # import os import sys +from pathlib import Path # -- Project information ----------------------------------------------------- @@ -21,7 +22,8 @@ author = "Snowflake Inc." # The full version, including alpha/beta/rc tags -SRC_DIR = "../../src" +REPO_ROOT = Path(__file__).resolve().parents[2] +SRC_DIR = str(REPO_ROOT / "src") sys.path.insert(0, os.path.abspath(SRC_DIR)) SNOWPARK_SRC_DIR = os.path.join(SRC_DIR, "snowflake", "snowpark") VERSION = (1, 1, 1, None) # Default, needed so code will compile @@ -353,22 +355,21 @@ def linkcode_resolve(domain, info): try: if isinstance(obj, property): - fn = inspect.getsourcefile(inspect.unwrap(obj.fget)) - else: - fn = inspect.getsourcefile(inspect.unwrap(obj)) - except TypeError as e: + obj = obj.fget + obj = inspect.unwrap(obj) + fn = inspect.getsourcefile(obj) + if fn is None: + return None + relative_path = Path(fn).resolve().relative_to(REPO_ROOT).as_posix() + except (TypeError, ValueError, OSError): return None try: - if isinstance(obj, property): - source, lineno = inspect.getsourcelines(obj.fget) - else: - source, lineno = inspect.getsourcelines(obj) + source, lineno = inspect.getsourcelines(obj) linespec = f"#L{lineno}-L{lineno + len(source) - 1}" - except TypeError: + except (TypeError, OSError): linespec = "" return ( f"https://github.com/snowflakedb/snowpark-python/blob/" - f"v{release}/{os.path.relpath(fn)}{linespec}" + f"v{release}/{relative_path}{linespec}" ) - diff --git a/docs/source/modin/index.rst b/docs/source/modin/index.rst index ef076871ab..e95f280a68 100644 --- a/docs/source/modin/index.rst +++ b/docs/source/modin/index.rst @@ -5,6 +5,12 @@ Snowpark pandas API This page gives an overview of all public Snowpark pandas objects, functions and methods. For your convenience, here is all the :doc:`Supported APIs ` +Use this reference for Snowflake-specific behavior rather than assuming that +upstream pandas or Modin documentation describes every supported parameter. +The :doc:`support matrices ` include partial implementations +and unsupported parameters; individual method pages provide further details. +See :doc:`hybrid_execution` for how the execution backend is selected. + .. toctree:: :maxdepth: 2 @@ -22,4 +28,4 @@ For your convenience, here is all the :doc:`Supported APIs ` interoperability hybrid_execution numpy - performance \ No newline at end of file + performance diff --git a/docs/source/modin/interoperability.rst b/docs/source/modin/interoperability.rst index a155a8a21e..3ca4c9c0ff 100644 --- a/docs/source/modin/interoperability.rst +++ b/docs/source/modin/interoperability.rst @@ -9,6 +9,22 @@ works in Snowpark pandas as well. Snowpark pandas supports the `dataframe interchange protocol `_, which some libraries use to interoperate with Snowpark pandas to the same level of support as pandas. +Scope of this reference +======================= + +The tables below describe specific Plotly and scikit-learn operations, not a +complete compatibility certification for every version or API of a library. +Accepting a native pandas object does not by itself establish that an API accepts +a Snowpark pandas object or executes its work in Snowflake. + +For a library or operation not covered here, such as an Altair, Seaborn, +Matplotlib, XGBoost, NLTK, or Streamlit operation, check that library's accepted +input types. You can explicitly convert a Snowpark pandas DataFrame or Series +with ``to_pandas()`` and then use APIs that accept native pandas objects. +Conversion materializes the data on the client; ensure the result fits in +memory. This is a local-pandas path, not a claim of direct Snowpark pandas +interoperability. For NumPy-specific behavior, see :doc:`numpy`. + plotly.express ============== diff --git a/docs/source/modin/supported/index.rst b/docs/source/modin/supported/index.rst index 2d7999c495..4ec5701070 100644 --- a/docs/source/modin/supported/index.rst +++ b/docs/source/modin/supported/index.rst @@ -8,6 +8,13 @@ of the most recent release. To view the docs for the most recent release, check that you’re viewing the stable version of the docs. +Read each table's legend and the missing-parameter and notes columns, not just +the method name. An entry can describe a partial implementation or an +unsupported operation. These tables describe Snowpark pandas support; they are +not a promise that every upstream pandas or Modin API is available with identical +behavior. Also consult :doc:`../hybrid_execution` when interpreting where an +operation runs. + .. toctree:: :maxdepth: 2 @@ -21,4 +28,4 @@ To view the docs for the most recent release, check that you’re viewing the st groupby_supported resampling_supported series_dt_supported - series_str_supported \ No newline at end of file + series_str_supported diff --git a/docs/source/snowpark/dataframe.rst b/docs/source/snowpark/dataframe.rst index f6083cff84..e5dee2dfb4 100644 --- a/docs/source/snowpark/dataframe.rst +++ b/docs/source/snowpark/dataframe.rst @@ -88,6 +88,8 @@ DataFrame DataFrame.toDF DataFrame.toLocalIterator DataFrame.toPandas + DataFrame.to_arrow + DataFrame.to_arrow_batches DataFrame.to_df DataFrame.to_local_iterator DataFrame.to_pandas @@ -151,6 +153,7 @@ DataFrame :toctree: api/ DataFrame.ai + DataFrame.analytics DataFrame.columns DataFrame.na DataFrame.queries diff --git a/docs/source/snowpark/files.rst b/docs/source/snowpark/files.rst index 45099a4dae..0cf9e7ead0 100644 --- a/docs/source/snowpark/files.rst +++ b/docs/source/snowpark/files.rst @@ -26,7 +26,9 @@ Files :toctree: api/ ~SnowflakeFile.close + ~SnowflakeFile.detach ~SnowflakeFile.fileno + ~SnowflakeFile.flush ~SnowflakeFile.isatty ~SnowflakeFile.open ~SnowflakeFile.open_new_result @@ -49,4 +51,3 @@ Files .. rubric:: Attributes None - diff --git a/docs/source/snowpark/io.rst b/docs/source/snowpark/io.rst index 453aa6df87..8cb636784d 100644 --- a/docs/source/snowpark/io.rst +++ b/docs/source/snowpark/io.rst @@ -46,6 +46,7 @@ Input/Output DataFrameWriter.option DataFrameWriter.options DataFrameWriter.parquet + DataFrameWriter.partition_by DataFrameWriter.save DataFrameWriter.saveAsTable DataFrameWriter.save_as_table @@ -85,4 +86,3 @@ Input/Output ListResult.md5 ListResult.sha1 ListResult.last_modified - diff --git a/docs/source/snowpark/window.rst b/docs/source/snowpark/window.rst index f265ba8dfc..5a3e136457 100644 --- a/docs/source/snowpark/window.rst +++ b/docs/source/snowpark/window.rst @@ -10,6 +10,7 @@ Window :toctree: api/ Window + WindowSpec .. rubric:: Methods diff --git a/src/snowflake/snowpark/dataframe.py b/src/snowflake/snowpark/dataframe.py index 4afde3b5f1..55b82ae8ec 100644 --- a/src/snowflake/snowpark/dataframe.py +++ b/src/snowflake/snowpark/dataframe.py @@ -750,6 +750,11 @@ def stat(self) -> DataFrameStatFunctions: @property def analytics(self) -> DataFrameAnalyticsFunctions: + """Returns the :class:`DataFrameAnalyticsFunctions` namespace for this + DataFrame. Access methods through ``df.analytics``, for example + :meth:`DataFrameAnalyticsFunctions.moving_agg` or + :meth:`DataFrameAnalyticsFunctions.compute_lag`. + """ return self._analytics @property @@ -1153,10 +1158,21 @@ def to_pandas( 2. If you use :func:`Session.sql` with this method, the input query of :func:`Session.sql` can only be a SELECT statement. - 3. For TIMESTAMP columns: - - TIMESTAMP_LTZ and TIMESTAMP_TZ are both converted to `datetime64[ns, tz]` in pandas, - as pandas cannot distinguish between the two. - - TIMESTAMP_NTZ is converted to `datetime64[ns]` (without timezone). + 3. For TIMESTAMP columns, TIMESTAMP_LTZ and TIMESTAMP_TZ are both + converted to ``datetime64[ns, tz]`` in pandas, as pandas cannot + distinguish between the two. TIMESTAMP_NTZ is converted to + ``datetime64[ns]`` (without timezone). + + 4. Snowflake SQL types and pandas dtypes are not interchangeable. + Conversion does not preserve every detail of the Snowflake schema, + such as NUMBER precision and scale. NULL values can also affect the + resulting pandas dtype. See the Python Connector's + `Snowflake to pandas data mapping + `_. + Inspect :attr:`schema` before conversion and the returned DataFrame's + ``dtypes`` afterwards. Cast columns in Snowpark before conversion if + you need a particular SQL type. A pandas ``astype`` conversion after + fetching cannot recover precision already lost during conversion. """ if _emit_ast: @@ -1319,11 +1335,11 @@ def to_arrow( ) -> Union["pyarrow.Table", AsyncJob]: """ Executes the query representing this DataFrame and returns the result as a - `pyarrow Table `. + `pyarrow Table `_. When the data is too large to fit into memory, you can use :meth:`to_arrow_batches`. - This function requires the optional dependenct snowflake-snowpark-python[pandas] be installed. + This function requires the optional dependency ``snowflake-snowpark-python[pandas]`` to be installed. Args: statement_params: Dictionary of statement level parameters to be set while executing this action. @@ -6200,14 +6216,33 @@ def sample( n: Optional[int] = None, _emit_ast: bool = True, ) -> "DataFrame": - """Samples rows based on either the number of rows to be returned or a - percentage of rows to be returned. + """Returns a random sample of rows using Snowflake's SQL SAMPLE clause. + + Specify either ``frac`` or ``n``. Fractional sampling includes each row + with the given probability, so the number of returned rows can vary. + Fixed-size sampling returns the requested number of rows, or all rows + if the input contains fewer rows. Neither form guarantees row order. + Repeated executions can return different samples; this method has no + seed parameter. For seeded sampling of a table, see :meth:`Table.sample`. + + See `SAMPLE `_ + for SQL sampling semantics. Args: - frac: the percentage of rows to be sampled. + frac: The probability of selecting each row, from 0.0 to 1.0 + inclusive. For example, 0.1 requests approximately 10 percent + of the rows, not exactly 10 percent. n: the number of rows to sample in the range of 0 to 1,000,000 (inclusive). + Returns: a :class:`DataFrame` containing the sample of rows. + + Examples:: + + >>> df = session.range(100) + >>> fractional_sample = df.sample(frac=0.1) + >>> fixed_sample = df.sample(n=5) + >>> assert fixed_sample.count() == 5 """ DataFrame._validate_sample_input(frac, n) @@ -6263,6 +6298,15 @@ def ai(self) -> DataFrameAIFunctions: """ Returns a :class:`DataFrameAIFunctions` object that provides AI-powered functions for the DataFrame. + + Access this namespace through an existing DataFrame, for example + ``df.ai``. It is not a column and accessing it does not itself execute + an AI function. Call a method on the namespace to build an AI operation. + + See :meth:`DataFrameAIFunctions.classify`, + :meth:`DataFrameAIFunctions.extract`, and + :meth:`DataFrameAIFunctions.sentiment` for parameters and examples. + The :class:`DataFrameAIFunctions` reference lists the available methods. """ return self._ai diff --git a/src/snowflake/snowpark/dataframe_reader.py b/src/snowflake/snowpark/dataframe_reader.py index 17cbc60e7e..fd618e149c 100644 --- a/src/snowflake/snowpark/dataframe_reader.py +++ b/src/snowflake/snowpark/dataframe_reader.py @@ -1026,6 +1026,26 @@ def load(self, path: Optional[str] = None, _emit_ast: bool = True) -> DataFrame: def csv(self, path: str, _emit_ast: bool = True) -> DataFrame: """Specify the path of the CSV file(s) to load. + Configure the reader with :meth:`option`, :meth:`options`, and + :meth:`schema` before calling ``csv``. Options are not keyword + arguments of this method. Common CSV format options include + ``FIELD_DELIMITER``, ``SKIP_HEADER``, and + ``FIELD_OPTIONALLY_ENCLOSED_BY``. See the + `CSV format options + `_ + for their accepted values. + + Provide a schema explicitly or enable the reader's ``INFER_SCHEMA`` + option. Schema inference can issue queries while constructing the + DataFrame and is not supported for CSV in local testing mode. + + Example (requires an existing stage containing a header row and two + comma-separated columns):: + + >>> from snowflake.snowpark.types import StructType, StructField, IntegerType, StringType + >>> schema = StructType([StructField("id", IntegerType()), StructField("name", StringType())]) + >>> df = session.read.schema(schema).option("SKIP_HEADER", 1).csv("@my_stage/people.csv") # doctest: +SKIP + Args: path: The stage location of a CSV file, or a stage location that has CSV files. diff --git a/src/snowflake/snowpark/dataframe_stat_functions.py b/src/snowflake/snowpark/dataframe_stat_functions.py index 306025f52d..b5b17dc350 100644 --- a/src/snowflake/snowpark/dataframe_stat_functions.py +++ b/src/snowflake/snowpark/dataframe_stat_functions.py @@ -74,7 +74,9 @@ def approx_quantile( Args: col: The name of the numeric column. - percentile: A list of float values greater than or equal to 0.0 and less than 1.0. + percentile: A list of float values between 0.0 and 1.0, inclusive. + For example, 0.5 requests the approximate median and 1.0 requests + the approximate maximum. statement_params: Dictionary of statement level parameters to be set while executing this action. Returns: diff --git a/src/snowflake/snowpark/exceptions.py b/src/snowflake/snowpark/exceptions.py index 5ae5ace724..d3bbcb09bf 100644 --- a/src/snowflake/snowpark/exceptions.py +++ b/src/snowflake/snowpark/exceptions.py @@ -77,6 +77,23 @@ class SnowparkSQLException(SnowparkClientException): Includes all error codes in range 13XX (where XX is 0-9). This exception is specifically raised for error codes: 1300, 1304. + + Attributes: + error_code: Snowpark client error code. This is distinct from the + underlying Snowflake SQL error code. + sql_error_code: Snowflake SQL error number, when provided by the connector. + sfqid: Snowflake query ID, when available. Use this to locate the failed + statement in `Query History + `_. + query: SQL text associated with the error, when available. + raw_message: Underlying error message, when available. + conn_error: Original connector exception, when available. + + Read the error message first: a compilation error, missing object, and + insufficient privilege require different fixes even if they share a + Snowpark error code. Check the failed query's database, schema, role, and + referenced identifiers. When requesting help, include the query ID and + error codes; redact sensitive literals from SQL and error messages. """ def __init__( diff --git a/src/snowflake/snowpark/file_operation.py b/src/snowflake/snowpark/file_operation.py index 2a0f4fbefc..4415a11664 100644 --- a/src/snowflake/snowpark/file_operation.py +++ b/src/snowflake/snowpark/file_operation.py @@ -213,6 +213,12 @@ def get( The command lists all files in the specified path and applies the regular expression pattern on each of the files found. Default: ``None`` (all files in the specified stage are downloaded). statement_params: Dictionary of statement level parameters to be set while executing this action. + For example, ``{"QUERY_TAG": "download_files"}`` tags the GET + statement. See `Snowflake parameters + `_ for + parameter meanings and allowed values. Only parameters supported + at statement level apply; this is not a way to set account-only + parameters or GET options such as ``parallel`` and ``pattern``. Returns: A ``list`` of :class:`GetResult` instances, each of which represents the result of a downloaded file. diff --git a/src/snowflake/snowpark/files.py b/src/snowflake/snowpark/files.py index 3be46b45a8..a651cadea7 100644 --- a/src/snowflake/snowpark/files.py +++ b/src/snowflake/snowpark/files.py @@ -40,7 +40,10 @@ class SnowflakeFile(RawIOBase): SnowflakeFile supports most operations supported by Python IOBase objects. A SnowflakeFile object can be used as a Python IOBase object. - The constructor of this class is not supposed to be called directly. Call :meth:`~snowflake.snowpark.file.SnowflakeFile.open` to create a read-only SnowflakeFile object, and call :meth:`~snowflake.snowpark.file.SnowflakeFile.open_new_result` to create a write-only SnowflakeFile object. + Do not call the constructor directly. Call :meth:`open` to create a read-only + SnowflakeFile object, or :meth:`open_new_result` to create a write-only result + file. The constructor's ``from_result_api`` argument is for internal use. + The ``is_owner_file`` argument is deprecated; use ``require_scoped_url``. This class is used to read and write files in UDFs and stored procedures. On Snowflake, it is used to read and write stage files. It also supports Python IOBase and BufferedBase methods such as :meth:`read`, :meth:`write`, :meth:`close`. @@ -52,8 +55,8 @@ class SnowflakeFile(RawIOBase): >>> from snowflake.snowpark.functions import udf >>> @udf ... def read_file(url: str) -> str: - ... file = SnowflakeFile.open(url, "r") - ... return file.read() + ... with SnowflakeFile.open(url, "r") as file: + ... return file.read() To write to a staged file first write to a result file via the following example. The result file will return as a scoped URL which can be copied to a permanent stage @@ -76,6 +79,11 @@ class SnowflakeFile(RawIOBase): (sessions in local testing mode that aren't connected to a real stage), and Snowflake stages. Scoped and Stage URLs (https://) are not yet supported. + Not every IOBase operation is supported. :meth:`fileno` raises an OSError, + :meth:`detach` is unsupported, :meth:`flush` does not write buffered data, + and :meth:`isatty` returns False. Consult each method's reference for + environment-specific limitations. + Note: 1. All of the implementation in this file is for local testing purposes. @@ -156,7 +164,8 @@ def open( require_scoped_url: bool = True, ) -> SnowflakeFile: """ - Used to create a :class:`~snowflake.snowpark.file.SnowflakeFile` which can only be used for read-based IO operations on the file. + Opens a file for reading and returns a :class:`SnowflakeFile` stream. + Use a ``with`` statement to close the stream after reading. In UDFs and Stored Procedures, the object works like a read-only Python IOBase object and as a wrapper for an IO stream of remote files. @@ -171,7 +180,42 @@ def open( file_location: scoped URL, file URL, or string path for files located in a stage mode: A string used to mark the type of an IO stream. Supported modes are "r" for text read and "rb" for binary read. is_owner_file: (Deprecated) A boolean value, if True, the API is intended to access owner's files and all URI/URL are allowed. If False, the API is intended to access files passed into the function by the caller and only scoped URL is allowed. - require_scoped_url: A boolean value, if True, file_location must be a scoped URL. A scoped URL ensures that the caller cannot access the UDF owners files that the caller does not have access to. + require_scoped_url: Defaults to True, the recommended setting for + caller-supplied file locations. In Snowflake, file_location must + then be a scoped URL. Do not disable this requirement for + untrusted input. Local testing does not validate scoped URLs + or reproduce Snowflake access-control checks. + + Note: + Reading an entire file with ``read()`` and then parsing it can hold + both the file and the parsed objects in memory. For large files, + read bounded chunks or use a streaming parser; avoid accumulating + all chunks in a list. Available memory depends on the execution + environment, so there is no universal safe file-size threshold. + + For UTF-8 files that might begin with a byte order mark (BOM), binary + mode lets you control decoding explicitly. Python's ``utf-8-sig`` + incremental decoder removes an initial UTF-8 BOM while preserving + multibyte characters split across chunks. Do not decode each chunk + independently or assume a JSON parser will ignore a BOM. + + Example (handler code; ``url`` is a caller-provided scoped URL):: + + import codecs + from snowflake.snowpark.files import SnowflakeFile + + def decoded_chunks(url): + decoder = codecs.getincrementaldecoder("utf-8-sig")() + with SnowflakeFile.open(url, "rb") as source: + while True: + chunk = source.read(64 * 1024) + if not chunk: + break + yield decoder.decode(chunk) + yield decoder.decode(b"", final=True) + + Consume these chunks incrementally. This helper decodes text; it does + not by itself parse a JSON document or bound the memory used by a caller. """ if mode not in _READ_MODES: raise ValueError( @@ -184,7 +228,13 @@ def open( @classmethod def open_new_result(cls, mode: str = "w") -> SnowflakeFile: """ - Used to create a :class:`~snowflake.snowpark.file.SnowflakeFile` which can only be used for write-based IO operations. UDFs/Stored Procedures should return the file to materialize it, and it is then made accessible via a scoped URL returned in the query results. + Used to create a :class:`SnowflakeFile` which can only be used for write-based IO operations. UDFs/Stored Procedures should return the file to materialize it, and it is then made accessible via a scoped URL returned in the query results. + + This creates a result file, not a permanent named stage file. Return the + file object from the handler after writing; returning only its contents + does not expose the file. Copy results that must be retained to a stage + using `COPY FILES `_. + See the write example in :class:`SnowflakeFile`. In UDFs and Stored Procedures, the object works like a write-only Python IOBase object and as a wrapper for an IO stream of remote files. diff --git a/src/snowflake/snowpark/functions.py b/src/snowflake/snowpark/functions.py index 9f35db90f9..ca0b02c2e4 100644 --- a/src/snowflake/snowpark/functions.py +++ b/src/snowflake/snowpark/functions.py @@ -7356,7 +7356,10 @@ def array_insert( def array_position( variant: ColumnOrName, array: ColumnOrName, _emit_ast: bool = True ) -> Column: - """Returns the index of the first occurrence of an element in an ARRAY. + """Returns the zero-based index of the first occurrence of an element in an ARRAY. + + The first element has index 0. If the value is not present, returns SQL NULL + (represented by ``None`` in a collected Row), not -1. Args: variant: Column containing the VARIANT value that you want to find. The function @@ -7364,16 +7367,16 @@ def array_position( array: Column containing the ARRAY to be searched. Example:: - >>> from snowflake.snowpark import Row - >>> df = session.create_dataframe([Row([2, 1]), Row([1, 3])], schema=["a"]) - >>> df.select(array_position(lit(1), "a").alias("result")).show() - ------------ - |"RESULT" | - ------------ - |1 | - |0 | - ------------ - + >>> from snowflake.snowpark.functions import array_position, lit + >>> df = session.create_dataframe( + ... [(1, [2, 1, 1]), (2, [1, 3]), (3, [4, 5])], + ... schema=["id", "values"]) + >>> df.select("id", array_position(lit(1), "values").alias("position")).sort("id").collect() + [Row(ID=1, POSITION=1), Row(ID=2, POSITION=0), Row(ID=3, POSITION=None)] + + In the first row, 1 appears twice; the result is the position of its first + occurrence. Use :func:`lit` to search for a literal value rather than a + column name. """ v = _to_col_if_str(variant, "array_position") a = _to_col_if_str(array, "array_position") @@ -8834,13 +8837,20 @@ def iff( expr1: A :class:`Column` expression or a literal value, which will be returned if ``condition`` is true. expr2: A :class:`Column` expression or a literal value, which will be returned - if ``condition`` is false. + if ``condition`` is false or NULL. Examples:: - >>> df = session.create_dataframe([True, False, None], schema=["a"]) - >>> df.select(iff(df["a"], lit("true"), lit("false")).alias("iff")).collect() - [Row(IFF='true'), Row(IFF='false'), Row(IFF='false')] + >>> from snowflake.snowpark.functions import iff, lit + >>> df = session.create_dataframe( + ... [(1, True), (2, False), (3, None)], schema=["id", "approved"]) + >>> df.select( + ... "id", iff(df["approved"], lit("ship"), lit("hold")).alias("action") + ... ).sort("id").collect() + [Row(ID=1, ACTION='ship'), Row(ID=2, ACTION='hold'), Row(ID=3, ACTION='hold')] + + Only an approved row selects ``"ship"``. Both false and unknown (NULL) + approval select ``"hold"``. """ ast = build_function_expr("iff", [condition, expr1, expr2]) if _emit_ast else None return _call_function( diff --git a/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py b/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py index 049411b846..3f8ae847fd 100644 --- a/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py +++ b/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py @@ -417,19 +417,28 @@ def _check_obj_and_set_backend_to_snowflake( Save the Snowpark pandas DataFrame or Series as a Snowflake table. Args: - obj: Either a Snowpark pandas DataFrame or Series + obj: The Snowpark pandas DataFrame or Series to write. This function + executes the write and returns None, not a row-count report. name: - Name of the SQL table or fully-qualified object identifier + Destination table name or fully-qualified object identifier, such as + ``"MY_DB.MY_SCHEMA.MY_TABLE"`` or + ``["MY_DB", "MY_SCHEMA", "MY_TABLE"]``. Unqualified names use the + session's current database and schema. Use double-quoted identifier + components to preserve mixed case. if_exists: How to behave if table already exists. default 'fail' - fail: Raise ValueError. - replace: Drop the table before inserting new values. - append: Insert new values to the existing table. The order of insertion is not guaranteed. index: default True - If true, save DataFrame index columns as table columns. + If true, save the object's index levels as table columns in addition + to its data columns. Set False to omit the index. Row ordering is + not preserved by a Snowflake table. index_label: Column label for index column(s). If None is given (default) and index is True, - then the index names are used. A sequence should be given if the DataFrame uses MultiIndex. + then the index names are used. For an unnamed index, supply a label + when saving it. For MultiIndex, provide one label per level. + Labels must not duplicate data-column labels. Ignored when index=False. table_type: The table type of table to be created. The supported values are: ``temp``, ``temporary``, and ``transient``. An empty string means to create a permanent table. Learn more about table @@ -505,6 +514,7 @@ def _read_snowflake_ray_backend( return df.set_backend("Ray") +@doc(_TO_SNOWFLAKE_DOC) def to_snowflake( obj: Union[DataFrame, Series], name: Union[str, Iterable[str]], diff --git a/src/snowflake/snowpark/session.py b/src/snowflake/snowpark/session.py index bb83a230d8..4b44f8b503 100644 --- a/src/snowflake/snowpark/session.py +++ b/src/snowflake/snowpark/session.py @@ -1062,6 +1062,23 @@ def conf(self) -> RuntimeConfig: def sql_simplifier_enabled(self) -> bool: """Set to ``True`` to use the SQL simplifier (defaults to ``True``). The generated SQLs from ``DataFrame`` transformations would have fewer layers of nested queries if the SQL simplifier is enabled. + + Set this property before constructing the DataFrame whose SQL you want + to inspect. SQL text can change between library versions; compare + :attr:`DataFrame.queries` rather than relying on a specific SQL string. + + Example:: + + >>> original_setting = session.sql_simplifier_enabled + >>> try: + ... session.sql_simplifier_enabled = True + ... df = session.range(10).select("id").filter("id > 2") + ... simplified_queries = df.queries["queries"] + ... session.sql_simplifier_enabled = False + ... df = session.range(10).select("id").filter("id > 2") + ... unsimplified_queries = df.queries["queries"] + ... finally: + ... session.sql_simplifier_enabled = original_setting """ return self._sql_simplifier_enabled @@ -3156,6 +3173,19 @@ def sql( or :func:`DataFrame.to_pandas` evaluate the DataFrame. For **immediate execution**, chain the call with the collect method: `session.sql(query).collect()`. + SQL compilation and execution errors usually surface when an action + executes the query, not when this method creates the DataFrame. + Operations that request schema metadata can also contact Snowflake + before collection. Catch :class:`~snowflake.snowpark.exceptions.SnowparkSQLException` + around the operation that triggers evaluation; inspect its message + and query ID to diagnose the server error. This is not an exhaustive + list of possible client, connection, or argument errors. + + Raises: + NotImplementedError: SQL execution is not supported in local testing + mode. See `mocking SQL operations + `_. + Args: query: The SQL statement to execute. params: binding parameters. We only support qmark bind variables. For more information, check @@ -3515,12 +3545,24 @@ def write_pandas( **kwargs: Dict[str, Any], ) -> Table: """Writes a pandas DataFrame to a table in Snowflake and returns a - Snowpark :class:`DataFrame` object referring to the table where the + Snowpark :class:`Table` object referring to the table where the pandas DataFrame was written to. + The return value is not an insertion report. Calling ``count()`` on it + counts all rows currently in the target table, including pre-existing + rows when appending. The number of input rows is not necessarily the + number loaded if ``on_error`` permits skipping errors. For per-load + row counts, the Python Connector's + `write_pandas `_ + returns a result tuple containing ``num_rows``; that is a different API. + Args: df: The pandas DataFrame or Snowpark pandas DataFrame or Series we'd like to write back. - table_name: Name of the table we want to insert into. + table_name: Name of the table we want to insert into, without the + database or schema prefix. Pass those separately through + ``database`` and ``schema``. For example, use + ``table_name="MY_TABLE", database="MY_DB", schema="MY_SCHEMA"``, + not ``table_name="MY_DB.MY_SCHEMA.MY_TABLE"``. database: Database that the table is in. If not provided, the default one will be used. schema: Schema that the table is in. If not provided, the default one will be used. chunk_size: Number of rows to be inserted once. If not provided, all rows will be dumped once. @@ -3537,6 +3579,10 @@ def write_pandas( quote_identifiers: By default, identifiers, specifically database, schema, table and column names (from :attr:`DataFrame.columns`) will be quoted. If set to ``False``, identifiers are passed on to Snowflake without quoting, i.e. identifiers will be coerced to uppercase by Snowflake. + With the default ``True``, names must match the stored case: + an object created with an unquoted name normally has an uppercase + name, whereas a quoted mixed-case name must retain its case. + Do not uppercase names of quoted mixed-case objects. auto_create_table: When true, automatically creates a table to store the passed in pandas DataFrame using the passed in ``database``, ``schema``, and ``table_name``. Note: there are usually multiple table configurations that would allow you to upload a particular pandas DataFrame successfully. If you don't like the auto created diff --git a/tests/unit/modin/test_docstrings.py b/tests/unit/modin/test_docstrings.py index f9e858f9b3..4a1b81a943 100644 --- a/tests/unit/modin/test_docstrings.py +++ b/tests/unit/modin/test_docstrings.py @@ -2,12 +2,20 @@ # Copyright (c) 2012-2025 Snowflake Computing Inc. All rights reserved. # +import inspect + import modin.pandas as pd import pandas as native_pd import snowflake.snowpark.modin.plugin as plugin +def test_to_snowflake_has_parameter_documentation(): + documentation = inspect.getdoc(pd.to_snowflake) + for parameter in ("obj", "name", "if_exists", "index", "index_label", "table_type"): + assert f"{parameter}:" in documentation + + def test_base_property_snow_1305329(): assert "Snowpark pandas" in pd.base.BasePandasDataset.iloc.__doc__ diff --git a/tests/unit/test_documentation.py b/tests/unit/test_documentation.py new file mode 100644 index 0000000000..708a962669 --- /dev/null +++ b/tests/unit/test_documentation.py @@ -0,0 +1,107 @@ +import functools +import inspect +from pathlib import Path +import runpy +import sys +from unittest.mock import patch + +import pytest + + +REPOSITORY = Path(__file__).resolve().parents[2] + + +def documented_function(): + return None + + +@functools.wraps(documented_function) +def decorated_function(): + return documented_function() + + +class DocumentedClass: + @property + def value(self): + return None + + +@pytest.fixture +def documentation_config(): + original_path = sys.path[:] + try: + return runpy.run_path(str(REPOSITORY / "docs/source/conf.py")) + finally: + sys.path[:] = original_path + + +@pytest.mark.parametrize("working_directory", ["repository", "docs", "elsewhere"]) +@pytest.mark.parametrize( + "name,target", + [ + ("documented_function", documented_function), + ("decorated_function", documented_function), + ("DocumentedClass.value", DocumentedClass.value.fget), + ], +) +def test_source_links_are_independent_of_working_directory( + monkeypatch, tmp_path, documentation_config, working_directory, name, target +): + directory = { + "repository": REPOSITORY, + "docs": REPOSITORY / "docs/source", + "elsewhere": tmp_path, + }[working_directory] + monkeypatch.chdir(directory) + source, line = inspect.getsourcelines(target) + expected = ( + "https://github.com/snowflakedb/snowpark-python/blob/" + f"v{documentation_config['release']}/tests/unit/test_documentation.py" + f"#L{line}-L{line + len(source) - 1}" + ) + assert ( + documentation_config["linkcode_resolve"]( + "py", {"module": __name__, "fullname": name} + ) + == expected + ) + + +@pytest.mark.parametrize( + "domain,module,name", + [ + ("js", __name__, "documented_function"), + ("py", "missing_module", "missing"), + ("py", __name__, "missing"), + ("py", "builtins", "len"), + ("py", "inspect", "getsourcefile"), + ], +) +def test_source_links_skip_unavailable_or_external_objects( + documentation_config, domain, module, name +): + assert ( + documentation_config["linkcode_resolve"]( + domain, {"module": module, "fullname": name} + ) + is None + ) + + +def test_source_links_allow_unavailable_line_numbers(documentation_config): + with patch("inspect.getsourcelines", side_effect=OSError): + result = documentation_config["linkcode_resolve"]( + "py", {"module": __name__, "fullname": "documented_function"} + ) + assert result.endswith("/tests/unit/test_documentation.py") + + +def test_configuration_loads_outside_documentation_directory(monkeypatch, tmp_path): + monkeypatch.chdir(tmp_path) + original_path = sys.path[:] + try: + config = runpy.run_path(str(REPOSITORY / "docs/source/conf.py")) + assert config["REPO_ROOT"] == REPOSITORY + assert Path(config["SRC_DIR"]) == REPOSITORY / "src" + finally: + sys.path[:] = original_path From 0fb79548fd59167ab56de9de18342098f476992e Mon Sep 17 00:00:00 2001 From: Qinyi Ding Date: Thu, 1 Oct 2026 15:15:57 -0700 Subject: [PATCH 2/4] [DOC-5942] Align documentation tests with repository hooks Add the required test header and canonical import ordering. Generated with [Snowflake CoCo](https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code) Co-authored-by: Snowflake CoCo --- tests/unit/test_documentation.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/unit/test_documentation.py b/tests/unit/test_documentation.py index 708a962669..90124a0a89 100644 --- a/tests/unit/test_documentation.py +++ b/tests/unit/test_documentation.py @@ -1,13 +1,16 @@ +# +# Copyright (c) 2012-2025 Snowflake Computing Inc. All rights reserved. +# + import functools import inspect -from pathlib import Path import runpy import sys +from pathlib import Path from unittest.mock import patch import pytest - REPOSITORY = Path(__file__).resolve().parents[2] From 037be2d10a7fb3e5f9cd094e2b6d7406065fafc0 Mon Sep 17 00:00:00 2001 From: Qinyi Ding Date: Fri, 2 Oct 2026 10:05:54 -0700 Subject: [PATCH 3/4] [DOC-12260] Make compatibility guidance concise and neutral State support limits directly without defensive disclaimers or admonishing readers. Generated with [Snowflake CoCo](https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code) Co-authored-by: Snowflake CoCo --- docs/source/modin/index.rst | 7 ++----- docs/source/modin/interoperability.rst | 20 +++++--------------- docs/source/modin/supported/index.rst | 8 ++------ 3 files changed, 9 insertions(+), 26 deletions(-) diff --git a/docs/source/modin/index.rst b/docs/source/modin/index.rst index e95f280a68..10c4c4b777 100644 --- a/docs/source/modin/index.rst +++ b/docs/source/modin/index.rst @@ -5,11 +5,8 @@ Snowpark pandas API This page gives an overview of all public Snowpark pandas objects, functions and methods. For your convenience, here is all the :doc:`Supported APIs ` -Use this reference for Snowflake-specific behavior rather than assuming that -upstream pandas or Modin documentation describes every supported parameter. -The :doc:`support matrices ` include partial implementations -and unsupported parameters; individual method pages provide further details. -See :doc:`hybrid_execution` for how the execution backend is selected. +This reference describes Snowflake-specific API behavior and parameter support. +See :doc:`hybrid_execution` for execution backend selection. .. toctree:: diff --git a/docs/source/modin/interoperability.rst b/docs/source/modin/interoperability.rst index 3ca4c9c0ff..2143e3990f 100644 --- a/docs/source/modin/interoperability.rst +++ b/docs/source/modin/interoperability.rst @@ -9,21 +9,11 @@ works in Snowpark pandas as well. Snowpark pandas supports the `dataframe interchange protocol `_, which some libraries use to interoperate with Snowpark pandas to the same level of support as pandas. -Scope of this reference -======================= - -The tables below describe specific Plotly and scikit-learn operations, not a -complete compatibility certification for every version or API of a library. -Accepting a native pandas object does not by itself establish that an API accepts -a Snowpark pandas object or executes its work in Snowflake. - -For a library or operation not covered here, such as an Altair, Seaborn, -Matplotlib, XGBoost, NLTK, or Streamlit operation, check that library's accepted -input types. You can explicitly convert a Snowpark pandas DataFrame or Series -with ``to_pandas()`` and then use APIs that accept native pandas objects. -Conversion materializes the data on the client; ensure the result fits in -memory. This is a local-pandas path, not a claim of direct Snowpark pandas -interoperability. For NumPy-specific behavior, see :doc:`numpy`. +Compatibility varies by library, version, and operation. The tables below cover +Plotly and scikit-learn; see :doc:`numpy` for NumPy support. + +For APIs that require native pandas objects, use ``to_pandas()`` to convert the +data. Conversion loads the result into client memory. plotly.express ============== diff --git a/docs/source/modin/supported/index.rst b/docs/source/modin/supported/index.rst index 4ec5701070..e429056990 100644 --- a/docs/source/modin/supported/index.rst +++ b/docs/source/modin/supported/index.rst @@ -8,12 +8,8 @@ of the most recent release. To view the docs for the most recent release, check that you’re viewing the stable version of the docs. -Read each table's legend and the missing-parameter and notes columns, not just -the method name. An entry can describe a partial implementation or an -unsupported operation. These tables describe Snowpark pandas support; they are -not a promise that every upstream pandas or Modin API is available with identical -behavior. Also consult :doc:`../hybrid_execution` when interpreting where an -operation runs. +Snowpark pandas supports a subset of pandas and Modin APIs. Each table lists +supported operations, parameter limitations, and implementation notes. .. toctree:: :maxdepth: 2 From 9ba2d5493809c0c459606ea7b908343a499ad601 Mon Sep 17 00:00:00 2001 From: Qinyi Ding Date: Fri, 2 Oct 2026 15:53:21 -0700 Subject: [PATCH 4/4] [DOC-12399] Narrow batch to thirteen confident docstring fixes Defer build logic, new tests, broad reference audits and unresolved behavior requests. Attach to_snowflake documentation as a direct docstring so the remaining diff changes documentation only. Generated with [Snowflake CoCo](https://docs.snowflake.com/en/user-guide/cortex-code/cortex-code) Co-authored-by: Snowflake CoCo --- docs/source/conf.py | 25 ++-- docs/source/modin/index.rst | 5 +- docs/source/modin/interoperability.rst | 6 - docs/source/modin/supported/index.rst | 5 +- docs/source/snowpark/dataframe.rst | 3 - docs/source/snowpark/files.rst | 3 +- docs/source/snowpark/io.rst | 2 +- docs/source/snowpark/window.rst | 1 - src/snowflake/snowpark/dataframe.py | 28 +---- src/snowflake/snowpark/files.py | 70 +++-------- .../modin/plugin/extensions/pd_extensions.py | 53 ++++++--- src/snowflake/snowpark/session.py | 10 +- tests/unit/modin/test_docstrings.py | 8 -- tests/unit/test_documentation.py | 110 ------------------ 14 files changed, 78 insertions(+), 251 deletions(-) delete mode 100644 tests/unit/test_documentation.py diff --git a/docs/source/conf.py b/docs/source/conf.py index c51793fe7d..983d2136d8 100644 --- a/docs/source/conf.py +++ b/docs/source/conf.py @@ -12,7 +12,6 @@ # import os import sys -from pathlib import Path # -- Project information ----------------------------------------------------- @@ -22,8 +21,7 @@ author = "Snowflake Inc." # The full version, including alpha/beta/rc tags -REPO_ROOT = Path(__file__).resolve().parents[2] -SRC_DIR = str(REPO_ROOT / "src") +SRC_DIR = "../../src" sys.path.insert(0, os.path.abspath(SRC_DIR)) SNOWPARK_SRC_DIR = os.path.join(SRC_DIR, "snowflake", "snowpark") VERSION = (1, 1, 1, None) # Default, needed so code will compile @@ -355,21 +353,22 @@ def linkcode_resolve(domain, info): try: if isinstance(obj, property): - obj = obj.fget - obj = inspect.unwrap(obj) - fn = inspect.getsourcefile(obj) - if fn is None: - return None - relative_path = Path(fn).resolve().relative_to(REPO_ROOT).as_posix() - except (TypeError, ValueError, OSError): + fn = inspect.getsourcefile(inspect.unwrap(obj.fget)) + else: + fn = inspect.getsourcefile(inspect.unwrap(obj)) + except TypeError as e: return None try: - source, lineno = inspect.getsourcelines(obj) + if isinstance(obj, property): + source, lineno = inspect.getsourcelines(obj.fget) + else: + source, lineno = inspect.getsourcelines(obj) linespec = f"#L{lineno}-L{lineno + len(source) - 1}" - except (TypeError, OSError): + except TypeError: linespec = "" return ( f"https://github.com/snowflakedb/snowpark-python/blob/" - f"v{release}/{relative_path}{linespec}" + f"v{release}/{os.path.relpath(fn)}{linespec}" ) + diff --git a/docs/source/modin/index.rst b/docs/source/modin/index.rst index 10c4c4b777..ef076871ab 100644 --- a/docs/source/modin/index.rst +++ b/docs/source/modin/index.rst @@ -5,9 +5,6 @@ Snowpark pandas API This page gives an overview of all public Snowpark pandas objects, functions and methods. For your convenience, here is all the :doc:`Supported APIs ` -This reference describes Snowflake-specific API behavior and parameter support. -See :doc:`hybrid_execution` for execution backend selection. - .. toctree:: :maxdepth: 2 @@ -25,4 +22,4 @@ See :doc:`hybrid_execution` for execution backend selection. interoperability hybrid_execution numpy - performance + performance \ No newline at end of file diff --git a/docs/source/modin/interoperability.rst b/docs/source/modin/interoperability.rst index 2143e3990f..a155a8a21e 100644 --- a/docs/source/modin/interoperability.rst +++ b/docs/source/modin/interoperability.rst @@ -9,12 +9,6 @@ works in Snowpark pandas as well. Snowpark pandas supports the `dataframe interchange protocol `_, which some libraries use to interoperate with Snowpark pandas to the same level of support as pandas. -Compatibility varies by library, version, and operation. The tables below cover -Plotly and scikit-learn; see :doc:`numpy` for NumPy support. - -For APIs that require native pandas objects, use ``to_pandas()`` to convert the -data. Conversion loads the result into client memory. - plotly.express ============== diff --git a/docs/source/modin/supported/index.rst b/docs/source/modin/supported/index.rst index e429056990..2d7999c495 100644 --- a/docs/source/modin/supported/index.rst +++ b/docs/source/modin/supported/index.rst @@ -8,9 +8,6 @@ of the most recent release. To view the docs for the most recent release, check that you’re viewing the stable version of the docs. -Snowpark pandas supports a subset of pandas and Modin APIs. Each table lists -supported operations, parameter limitations, and implementation notes. - .. toctree:: :maxdepth: 2 @@ -24,4 +21,4 @@ supported operations, parameter limitations, and implementation notes. groupby_supported resampling_supported series_dt_supported - series_str_supported + series_str_supported \ No newline at end of file diff --git a/docs/source/snowpark/dataframe.rst b/docs/source/snowpark/dataframe.rst index e5dee2dfb4..f6083cff84 100644 --- a/docs/source/snowpark/dataframe.rst +++ b/docs/source/snowpark/dataframe.rst @@ -88,8 +88,6 @@ DataFrame DataFrame.toDF DataFrame.toLocalIterator DataFrame.toPandas - DataFrame.to_arrow - DataFrame.to_arrow_batches DataFrame.to_df DataFrame.to_local_iterator DataFrame.to_pandas @@ -153,7 +151,6 @@ DataFrame :toctree: api/ DataFrame.ai - DataFrame.analytics DataFrame.columns DataFrame.na DataFrame.queries diff --git a/docs/source/snowpark/files.rst b/docs/source/snowpark/files.rst index 0cf9e7ead0..45099a4dae 100644 --- a/docs/source/snowpark/files.rst +++ b/docs/source/snowpark/files.rst @@ -26,9 +26,7 @@ Files :toctree: api/ ~SnowflakeFile.close - ~SnowflakeFile.detach ~SnowflakeFile.fileno - ~SnowflakeFile.flush ~SnowflakeFile.isatty ~SnowflakeFile.open ~SnowflakeFile.open_new_result @@ -51,3 +49,4 @@ Files .. rubric:: Attributes None + diff --git a/docs/source/snowpark/io.rst b/docs/source/snowpark/io.rst index 8cb636784d..453aa6df87 100644 --- a/docs/source/snowpark/io.rst +++ b/docs/source/snowpark/io.rst @@ -46,7 +46,6 @@ Input/Output DataFrameWriter.option DataFrameWriter.options DataFrameWriter.parquet - DataFrameWriter.partition_by DataFrameWriter.save DataFrameWriter.saveAsTable DataFrameWriter.save_as_table @@ -86,3 +85,4 @@ Input/Output ListResult.md5 ListResult.sha1 ListResult.last_modified + diff --git a/docs/source/snowpark/window.rst b/docs/source/snowpark/window.rst index 5a3e136457..f265ba8dfc 100644 --- a/docs/source/snowpark/window.rst +++ b/docs/source/snowpark/window.rst @@ -10,7 +10,6 @@ Window :toctree: api/ Window - WindowSpec .. rubric:: Methods diff --git a/src/snowflake/snowpark/dataframe.py b/src/snowflake/snowpark/dataframe.py index 55b82ae8ec..e70420c8c4 100644 --- a/src/snowflake/snowpark/dataframe.py +++ b/src/snowflake/snowpark/dataframe.py @@ -750,11 +750,6 @@ def stat(self) -> DataFrameStatFunctions: @property def analytics(self) -> DataFrameAnalyticsFunctions: - """Returns the :class:`DataFrameAnalyticsFunctions` namespace for this - DataFrame. Access methods through ``df.analytics``, for example - :meth:`DataFrameAnalyticsFunctions.moving_agg` or - :meth:`DataFrameAnalyticsFunctions.compute_lag`. - """ return self._analytics @property @@ -1158,21 +1153,10 @@ def to_pandas( 2. If you use :func:`Session.sql` with this method, the input query of :func:`Session.sql` can only be a SELECT statement. - 3. For TIMESTAMP columns, TIMESTAMP_LTZ and TIMESTAMP_TZ are both - converted to ``datetime64[ns, tz]`` in pandas, as pandas cannot - distinguish between the two. TIMESTAMP_NTZ is converted to - ``datetime64[ns]`` (without timezone). - - 4. Snowflake SQL types and pandas dtypes are not interchangeable. - Conversion does not preserve every detail of the Snowflake schema, - such as NUMBER precision and scale. NULL values can also affect the - resulting pandas dtype. See the Python Connector's - `Snowflake to pandas data mapping - `_. - Inspect :attr:`schema` before conversion and the returned DataFrame's - ``dtypes`` afterwards. Cast columns in Snowpark before conversion if - you need a particular SQL type. A pandas ``astype`` conversion after - fetching cannot recover precision already lost during conversion. + 3. For TIMESTAMP columns: + - TIMESTAMP_LTZ and TIMESTAMP_TZ are both converted to `datetime64[ns, tz]` in pandas, + as pandas cannot distinguish between the two. + - TIMESTAMP_NTZ is converted to `datetime64[ns]` (without timezone). """ if _emit_ast: @@ -1335,11 +1319,11 @@ def to_arrow( ) -> Union["pyarrow.Table", AsyncJob]: """ Executes the query representing this DataFrame and returns the result as a - `pyarrow Table `_. + `pyarrow Table `. When the data is too large to fit into memory, you can use :meth:`to_arrow_batches`. - This function requires the optional dependency ``snowflake-snowpark-python[pandas]`` to be installed. + This function requires the optional dependenct snowflake-snowpark-python[pandas] be installed. Args: statement_params: Dictionary of statement level parameters to be set while executing this action. diff --git a/src/snowflake/snowpark/files.py b/src/snowflake/snowpark/files.py index a651cadea7..622329b0ba 100644 --- a/src/snowflake/snowpark/files.py +++ b/src/snowflake/snowpark/files.py @@ -40,10 +40,7 @@ class SnowflakeFile(RawIOBase): SnowflakeFile supports most operations supported by Python IOBase objects. A SnowflakeFile object can be used as a Python IOBase object. - Do not call the constructor directly. Call :meth:`open` to create a read-only - SnowflakeFile object, or :meth:`open_new_result` to create a write-only result - file. The constructor's ``from_result_api`` argument is for internal use. - The ``is_owner_file`` argument is deprecated; use ``require_scoped_url``. + The constructor of this class is not supposed to be called directly. Call :meth:`~snowflake.snowpark.file.SnowflakeFile.open` to create a read-only SnowflakeFile object, and call :meth:`~snowflake.snowpark.file.SnowflakeFile.open_new_result` to create a write-only SnowflakeFile object. This class is used to read and write files in UDFs and stored procedures. On Snowflake, it is used to read and write stage files. It also supports Python IOBase and BufferedBase methods such as :meth:`read`, :meth:`write`, :meth:`close`. @@ -55,8 +52,8 @@ class SnowflakeFile(RawIOBase): >>> from snowflake.snowpark.functions import udf >>> @udf ... def read_file(url: str) -> str: - ... with SnowflakeFile.open(url, "r") as file: - ... return file.read() + ... file = SnowflakeFile.open(url, "r") + ... return file.read() To write to a staged file first write to a result file via the following example. The result file will return as a scoped URL which can be copied to a permanent stage @@ -79,11 +76,6 @@ class SnowflakeFile(RawIOBase): (sessions in local testing mode that aren't connected to a real stage), and Snowflake stages. Scoped and Stage URLs (https://) are not yet supported. - Not every IOBase operation is supported. :meth:`fileno` raises an OSError, - :meth:`detach` is unsupported, :meth:`flush` does not write buffered data, - and :meth:`isatty` returns False. Consult each method's reference for - environment-specific limitations. - Note: 1. All of the implementation in this file is for local testing purposes. @@ -167,6 +159,17 @@ def open( Opens a file for reading and returns a :class:`SnowflakeFile` stream. Use a ``with`` statement to close the stream after reading. + For example, inside a handler with a caller-provided scoped URL:: + + from snowflake.snowpark.files import SnowflakeFile + + def read_text(url): + with SnowflakeFile.open(url, "r") as source: + return source.read() + + This example reads the whole file into memory. Use binary mode + (``"rb"``) when the consumer expects bytes rather than text. + In UDFs and Stored Procedures, the object works like a read-only Python IOBase object and as a wrapper for an IO stream of remote files. All files are accessed in the context of the UDF owner (with the exception of caller's rights stored procedures which use the caller's context). @@ -180,42 +183,7 @@ def open( file_location: scoped URL, file URL, or string path for files located in a stage mode: A string used to mark the type of an IO stream. Supported modes are "r" for text read and "rb" for binary read. is_owner_file: (Deprecated) A boolean value, if True, the API is intended to access owner's files and all URI/URL are allowed. If False, the API is intended to access files passed into the function by the caller and only scoped URL is allowed. - require_scoped_url: Defaults to True, the recommended setting for - caller-supplied file locations. In Snowflake, file_location must - then be a scoped URL. Do not disable this requirement for - untrusted input. Local testing does not validate scoped URLs - or reproduce Snowflake access-control checks. - - Note: - Reading an entire file with ``read()`` and then parsing it can hold - both the file and the parsed objects in memory. For large files, - read bounded chunks or use a streaming parser; avoid accumulating - all chunks in a list. Available memory depends on the execution - environment, so there is no universal safe file-size threshold. - - For UTF-8 files that might begin with a byte order mark (BOM), binary - mode lets you control decoding explicitly. Python's ``utf-8-sig`` - incremental decoder removes an initial UTF-8 BOM while preserving - multibyte characters split across chunks. Do not decode each chunk - independently or assume a JSON parser will ignore a BOM. - - Example (handler code; ``url`` is a caller-provided scoped URL):: - - import codecs - from snowflake.snowpark.files import SnowflakeFile - - def decoded_chunks(url): - decoder = codecs.getincrementaldecoder("utf-8-sig")() - with SnowflakeFile.open(url, "rb") as source: - while True: - chunk = source.read(64 * 1024) - if not chunk: - break - yield decoder.decode(chunk) - yield decoder.decode(b"", final=True) - - Consume these chunks incrementally. This helper decodes text; it does - not by itself parse a JSON document or bound the memory used by a caller. + require_scoped_url: A boolean value, if True, file_location must be a scoped URL. A scoped URL ensures that the caller cannot access the UDF owners files that the caller does not have access to. """ if mode not in _READ_MODES: raise ValueError( @@ -228,13 +196,7 @@ def decoded_chunks(url): @classmethod def open_new_result(cls, mode: str = "w") -> SnowflakeFile: """ - Used to create a :class:`SnowflakeFile` which can only be used for write-based IO operations. UDFs/Stored Procedures should return the file to materialize it, and it is then made accessible via a scoped URL returned in the query results. - - This creates a result file, not a permanent named stage file. Return the - file object from the handler after writing; returning only its contents - does not expose the file. Copy results that must be retained to a stage - using `COPY FILES `_. - See the write example in :class:`SnowflakeFile`. + Used to create a :class:`~snowflake.snowpark.file.SnowflakeFile` which can only be used for write-based IO operations. UDFs/Stored Procedures should return the file to materialize it, and it is then made accessible via a scoped URL returned in the query results. In UDFs and Stored Procedures, the object works like a write-only Python IOBase object and as a wrapper for an IO stream of remote files. diff --git a/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py b/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py index 3f8ae847fd..1f3674a48b 100644 --- a/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py +++ b/src/snowflake/snowpark/modin/plugin/extensions/pd_extensions.py @@ -417,28 +417,19 @@ def _check_obj_and_set_backend_to_snowflake( Save the Snowpark pandas DataFrame or Series as a Snowflake table. Args: - obj: The Snowpark pandas DataFrame or Series to write. This function - executes the write and returns None, not a row-count report. + obj: Either a Snowpark pandas DataFrame or Series name: - Destination table name or fully-qualified object identifier, such as - ``"MY_DB.MY_SCHEMA.MY_TABLE"`` or - ``["MY_DB", "MY_SCHEMA", "MY_TABLE"]``. Unqualified names use the - session's current database and schema. Use double-quoted identifier - components to preserve mixed case. + Name of the SQL table or fully-qualified object identifier if_exists: How to behave if table already exists. default 'fail' - fail: Raise ValueError. - replace: Drop the table before inserting new values. - append: Insert new values to the existing table. The order of insertion is not guaranteed. index: default True - If true, save the object's index levels as table columns in addition - to its data columns. Set False to omit the index. Row ordering is - not preserved by a Snowflake table. + If true, save DataFrame index columns as table columns. index_label: Column label for index column(s). If None is given (default) and index is True, - then the index names are used. For an unnamed index, supply a label - when saving it. For MultiIndex, provide one label per level. - Labels must not duplicate data-column labels. Ignored when index=False. + then the index names are used. A sequence should be given if the DataFrame uses MultiIndex. table_type: The table type of table to be created. The supported values are: ``temp``, ``temporary``, and ``transient``. An empty string means to create a permanent table. Learn more about table @@ -514,7 +505,6 @@ def _read_snowflake_ray_backend( return df.set_backend("Ray") -@doc(_TO_SNOWFLAKE_DOC) def to_snowflake( obj: Union[DataFrame, Series], name: Union[str, Iterable[str]], @@ -523,6 +513,41 @@ def to_snowflake( index_label: Optional[IndexLabel] = None, table_type: Literal["", "temp", "temporary", "transient"] = "", ) -> None: + """Save a Snowpark pandas DataFrame or Series as a Snowflake table. + + Args: + obj: The Snowpark pandas DataFrame or Series to write. + name: Destination table name or fully-qualified identifier, such as + ``"MY_DB.MY_SCHEMA.MY_TABLE"`` or + ``["MY_DB", "MY_SCHEMA", "MY_TABLE"]``. Unqualified names use the + session's current database and schema. Double-quote identifier + components to preserve mixed case. + if_exists: How to handle an existing table. Defaults to ``"fail"``: + + - ``"fail"``: Raise ValueError if the table exists. + - ``"replace"``: Drop the table and write the new values. + - ``"append"``: Add rows to the existing table. + + index: If True (the default), save index levels as table columns in + addition to the data columns. Set False to omit the index. + index_label: Column label or labels for the saved index. Defaults to + the index names. Supply a label for an unnamed index and one label + per level for a MultiIndex. Labels must not duplicate data-column + labels. Ignored when ``index=False``. + table_type: Type of table to create: ``"temp"`` or ``"temporary"``, + ``"transient"``, or ``""`` (the default) for a permanent table. + See `table types + `_. + + Returns: + None. The write is executed by this call. Row ordering is not preserved + by a Snowflake table. + + See also: + :func:`DataFrame.to_snowflake `, + :func:`Series.to_snowflake `, + :func:`read_snowflake `. + """ _snowpark_pandas_obj_check(obj) return obj.to_snowflake( name=name, diff --git a/src/snowflake/snowpark/session.py b/src/snowflake/snowpark/session.py index 4b44f8b503..13b84be1d8 100644 --- a/src/snowflake/snowpark/session.py +++ b/src/snowflake/snowpark/session.py @@ -3545,17 +3545,9 @@ def write_pandas( **kwargs: Dict[str, Any], ) -> Table: """Writes a pandas DataFrame to a table in Snowflake and returns a - Snowpark :class:`Table` object referring to the table where the + Snowpark :class:`DataFrame` object referring to the table where the pandas DataFrame was written to. - The return value is not an insertion report. Calling ``count()`` on it - counts all rows currently in the target table, including pre-existing - rows when appending. The number of input rows is not necessarily the - number loaded if ``on_error`` permits skipping errors. For per-load - row counts, the Python Connector's - `write_pandas `_ - returns a result tuple containing ``num_rows``; that is a different API. - Args: df: The pandas DataFrame or Snowpark pandas DataFrame or Series we'd like to write back. table_name: Name of the table we want to insert into, without the diff --git a/tests/unit/modin/test_docstrings.py b/tests/unit/modin/test_docstrings.py index 4a1b81a943..f9e858f9b3 100644 --- a/tests/unit/modin/test_docstrings.py +++ b/tests/unit/modin/test_docstrings.py @@ -2,20 +2,12 @@ # Copyright (c) 2012-2025 Snowflake Computing Inc. All rights reserved. # -import inspect - import modin.pandas as pd import pandas as native_pd import snowflake.snowpark.modin.plugin as plugin -def test_to_snowflake_has_parameter_documentation(): - documentation = inspect.getdoc(pd.to_snowflake) - for parameter in ("obj", "name", "if_exists", "index", "index_label", "table_type"): - assert f"{parameter}:" in documentation - - def test_base_property_snow_1305329(): assert "Snowpark pandas" in pd.base.BasePandasDataset.iloc.__doc__ diff --git a/tests/unit/test_documentation.py b/tests/unit/test_documentation.py deleted file mode 100644 index 90124a0a89..0000000000 --- a/tests/unit/test_documentation.py +++ /dev/null @@ -1,110 +0,0 @@ -# -# Copyright (c) 2012-2025 Snowflake Computing Inc. All rights reserved. -# - -import functools -import inspect -import runpy -import sys -from pathlib import Path -from unittest.mock import patch - -import pytest - -REPOSITORY = Path(__file__).resolve().parents[2] - - -def documented_function(): - return None - - -@functools.wraps(documented_function) -def decorated_function(): - return documented_function() - - -class DocumentedClass: - @property - def value(self): - return None - - -@pytest.fixture -def documentation_config(): - original_path = sys.path[:] - try: - return runpy.run_path(str(REPOSITORY / "docs/source/conf.py")) - finally: - sys.path[:] = original_path - - -@pytest.mark.parametrize("working_directory", ["repository", "docs", "elsewhere"]) -@pytest.mark.parametrize( - "name,target", - [ - ("documented_function", documented_function), - ("decorated_function", documented_function), - ("DocumentedClass.value", DocumentedClass.value.fget), - ], -) -def test_source_links_are_independent_of_working_directory( - monkeypatch, tmp_path, documentation_config, working_directory, name, target -): - directory = { - "repository": REPOSITORY, - "docs": REPOSITORY / "docs/source", - "elsewhere": tmp_path, - }[working_directory] - monkeypatch.chdir(directory) - source, line = inspect.getsourcelines(target) - expected = ( - "https://github.com/snowflakedb/snowpark-python/blob/" - f"v{documentation_config['release']}/tests/unit/test_documentation.py" - f"#L{line}-L{line + len(source) - 1}" - ) - assert ( - documentation_config["linkcode_resolve"]( - "py", {"module": __name__, "fullname": name} - ) - == expected - ) - - -@pytest.mark.parametrize( - "domain,module,name", - [ - ("js", __name__, "documented_function"), - ("py", "missing_module", "missing"), - ("py", __name__, "missing"), - ("py", "builtins", "len"), - ("py", "inspect", "getsourcefile"), - ], -) -def test_source_links_skip_unavailable_or_external_objects( - documentation_config, domain, module, name -): - assert ( - documentation_config["linkcode_resolve"]( - domain, {"module": module, "fullname": name} - ) - is None - ) - - -def test_source_links_allow_unavailable_line_numbers(documentation_config): - with patch("inspect.getsourcelines", side_effect=OSError): - result = documentation_config["linkcode_resolve"]( - "py", {"module": __name__, "fullname": "documented_function"} - ) - assert result.endswith("/tests/unit/test_documentation.py") - - -def test_configuration_loads_outside_documentation_directory(monkeypatch, tmp_path): - monkeypatch.chdir(tmp_path) - original_path = sys.path[:] - try: - config = runpy.run_path(str(REPOSITORY / "docs/source/conf.py")) - assert config["REPO_ROOT"] == REPOSITORY - assert Path(config["SRC_DIR"]) == REPOSITORY / "src" - finally: - sys.path[:] = original_path