From 42e30c2639bbfaf38f7ea3eddf94545b0d8b00e0 Mon Sep 17 00:00:00 2001 From: Hariharan Jayakumar Date: Thu, 1 Oct 2026 04:29:09 +0000 Subject: [PATCH 1/2] Read documents with AI_COMPLETE on claude-opus-5, unparsed Demo branch. Three default changes so that supplying a schema gets the document read as a file by AI_COMPLETE on claude-opus-5, rather than parsed to text and handed to AI_EXTRACT. Engine default ai_extract -> ai_complete. AI_EXTRACT flattens tabular and clause-structured content instead of extracting it, and its schema path cannot express the verbatim multi-paragraph answers clause-level review asks for. The AI_EXTRACT path stays reachable by explicit option, where its per-field confidence scores still have no AI_COMPLETE equivalent. Model default claude-4-sonnet -> claude-opus-5. Skip AI_PARSE_DOCUMENT when a schema is present and the engine is AI_COMPLETE, and pass the document as a FILE instead. Parsing flattens the document, and layout is information: table columns, and the printed section numbers clause extraction is asked to quote, survive in the page image and not in flattened text. It also removes a Cortex call per document along with its own failure and latency modes. Conditional on there being something to extract -- with no schema the parsed text is the read's only output, so skipping the parse there would return a DataFrame with nothing in it. An explicit parse_mode always wins, and the default is filled before validate() so validation sees the parse_mode the read will use. Send model_parameters, which were previously absent entirely. AI_COMPLETE then applied its own max_tokens default of 4096, low enough to truncate a document answering more than a handful of fields -- and truncation discards the whole response rather than the overflow, surfacing as a bare internal error that is indistinguishable from a server fault, or occasionally as success with a silently partial answer. The ceiling is per-model, so one shared constant is unsafe: claude-opus-5's ceiling sent to claude-4-sonnet fails every call. Only models whose limit has been confirmed are listed; anything else sends no max_tokens and gets the server default, which is always accepted. claude-opus-5 is capped at 65536 rather than its 128000 maximum, because a larger ceiling does not rescue a document whose answer cannot fit in one response -- it only lets the call run longer before failing. temperature is 0 for every model: the same document and schema should not yield different answers on consecutive reads. Known gap: a flat {field: question} map is an AI_EXTRACT-only shape and is passed through unchanged, so AI_COMPLETE rejects it with "invalid response format object". A JSON Schema is required until schema preparation can convert one. Co-Authored-By: Claude Opus 5 (1M context) --- .../snowpark/_internal/document_reader.py | 41 ++++++++++++++++++- .../_internal/document_reader_options.py | 25 ++++++++++- 2 files changed, 64 insertions(+), 2 deletions(-) diff --git a/src/snowflake/snowpark/_internal/document_reader.py b/src/snowflake/snowpark/_internal/document_reader.py index d3c535cd2b..9773e0d6f4 100644 --- a/src/snowflake/snowpark/_internal/document_reader.py +++ b/src/snowflake/snowpark/_internal/document_reader.py @@ -64,7 +64,44 @@ from snowflake.snowpark.session import Session -_DEFAULT_EXTRACTION_MODEL = "claude-4-sonnet" +# DEMO BRANCH: claude-opus-5 rather than claude-4-sonnet. +_DEFAULT_EXTRACTION_MODEL = "claude-opus-5" + +# AI_COMPLETE applies its own max_tokens default of 4096 when none is sent, which is low +# enough to truncate a document answering more than a handful of fields. Truncation stops +# generation mid-JSON, response_format then rejects the malformed reply, and it surfaces as a +# bare internal error -- indistinguishable from a server fault. It can also report success +# with a silently partial answer, so status alone is not a quality signal. +# +# The ceiling is PER-MODEL, and the server states each model's limit when exceeded +# ("max_tokens parameter exceeds the maximum possible value (N)"). One shared constant is +# therefore unsafe: claude-opus-5's ceiling sent to claude-4-sonnet fails every call. Only +# models whose ceiling has been confirmed are listed -- anything else sends no max_tokens and +# gets the server default, which is always accepted. +# +# claude-opus-5 accepts 128000, but 65536 is used deliberately. A larger ceiling does not +# rescue a document whose answer cannot fit in one response -- it only lets the call run +# longer before failing, so the practical effect of the maximum is a slower failure. +_MODEL_MAX_OUTPUT_TOKENS = { + "claude-opus-5": 65536, + "claude-4-sonnet": 32000, +} + + +def complete_model_parameters(model: str) -> dict: + """``model_parameters`` for AI_COMPLETE: deterministic, and uncapped where it is safe. + + ``temperature`` is 0 for every model -- the same document and schema should not yield + different answers on consecutive reads, so determinism is a correctness property here + rather than a preference. + """ + parameters: dict = {"temperature": 0} + max_tokens = _MODEL_MAX_OUTPUT_TOKENS.get(model) + if max_tokens is not None: + parameters["max_tokens"] = max_tokens + return parameters + + _DEFAULT_EXTRACTION_PROMPT = "Extract the requested fields from this document." _PDF_READER_FILE_PATH = os.path.join(os.path.dirname(__file__), "pdf_reader.py") @@ -410,6 +447,7 @@ def build_ai_complete_call( return ai_complete( model, prompt_text, + model_parameters=complete_model_parameters(model), response_format=response_format, return_error_details=True, ) @@ -417,6 +455,7 @@ def build_ai_complete_call( model, options.prompt or _DEFAULT_EXTRACTION_PROMPT, file=input_col, + model_parameters=complete_model_parameters(model), response_format=response_format, return_error_details=True, ) diff --git a/src/snowflake/snowpark/_internal/document_reader_options.py b/src/snowflake/snowpark/_internal/document_reader_options.py index f8a54fe233..6496523274 100644 --- a/src/snowflake/snowpark/_internal/document_reader_options.py +++ b/src/snowflake/snowpark/_internal/document_reader_options.py @@ -136,7 +136,11 @@ class DocumentReaderOptions: page_filter: Optional[list] = None mode: str = "PERMISSIVE" corrupt_record_column: str = _DEFAULT_CORRUPT_RECORD_COLUMN - extraction_engine: str = "ai_extract" + # DEMO BRANCH: AI_COMPLETE rather than AI_EXTRACT. AI_EXTRACT flattens tabular and + # clause-structured content instead of extracting it, and cannot express the verbatim + # multi-paragraph answers that clause-level review asks for. This leaves the AI_EXTRACT + # path reachable only by explicit option. + extraction_engine: str = "ai_complete" extraction: Optional[ExtractionSpec] = None model: Optional[str] = None prompt: Optional[str] = None @@ -193,6 +197,25 @@ def from_reader_options( model=cur_options.get("MODEL", defaults.model), prompt=cur_options.get("PROMPT", defaults.prompt), ) + # DEMO BRANCH: hand the document to AI_COMPLETE as a FILE rather than as text + # AI_PARSE_DOCUMENT produced. Parsing flattens the document, and layout is + # information: table columns, and the printed section numbers clause extraction is + # asked to quote, survive in the page image and not in flattened text. Skipping the + # parse also removes a whole Cortex call per document, along with its own failure and + # latency modes -- parse cost scales with page content in ways the file's size and + # page count do not predict. + # + # Conditional on there being something to extract. With no schema, AI_COMPLETE has + # nothing to do and the parsed text is the read's only output, so skipping the parse + # would hand back a DataFrame with nothing in it. An explicit parse_mode always wins; + # this only fills the default, and runs before validate() so validation sees the + # parse_mode the read will actually use. + if ( + "PARSE_MODE" not in cur_options + and options.extract_enabled + and options.extraction_engine == "ai_complete" + ): + options.parse_mode = "none" options.validate() return options From 53b0c6a00cb99b1e6eb86dedaad47478ef3a6374 Mon Sep 17 00:00:00 2001 From: Hariharan Jayakumar Date: Thu, 1 Oct 2026 05:48:50 +0000 Subject: [PATCH 2/2] Accept a flat question map on the AI_COMPLETE path A response_format like {"vendor": "Who issued this invoice?"} is AI_EXTRACT's own shape. AI_COMPLETE requires {"type": "json", "schema": {...}} and rejects the bare map with "invalid response format object", so passing it through unchanged was only safe while AI_EXTRACT was the default engine. Now that it is not, a question map -- the form a caller reaches for when they have questions rather than a type contract -- fails outright. Rewrite it for AI_COMPLETE instead: each field becomes a string property whose description is the question, which is how a model consumes it under either engine. AI_EXTRACT continues to receive the caller's shape exactly as given, so its behaviour is unchanged. The array form is handled too, and keeps its questions. Both ["name", "question"] pairs and "name: question" strings carry the question inside the item, and taking only the field name would discard the only instruction the model gets for that field. Every converted field is typed string, because a question implies no type. A field whose answer should be a list or a number still needs a real JSON Schema -- the conversion makes the call succeed, it cannot add a contract the input never expressed. field_types is therefore left unset rather than populated with StringType: a question is not grounds to start casting output columns, least of all on the AI_EXTRACT path, which shares field_types. Co-Authored-By: Claude Opus 5 (1M context) --- .../_internal/document_reader_options.py | 53 +++++++++++++++++-- 1 file changed, 50 insertions(+), 3 deletions(-) diff --git a/src/snowflake/snowpark/_internal/document_reader_options.py b/src/snowflake/snowpark/_internal/document_reader_options.py index 6496523274..72527ce5c8 100644 --- a/src/snowflake/snowpark/_internal/document_reader_options.py +++ b/src/snowflake/snowpark/_internal/document_reader_options.py @@ -58,8 +58,9 @@ class ExtractionSpec: needs its own envelope for each engine -- AI_EXTRACT wants {"schema": {...}}, AI_COMPLETE wants {"type": "json", "schema": {...}} (see its docstring's "Structured output with response format" example) -- confirmed live against both engines, since neither's - own docstring documents this shape. Every other shape (AI_EXTRACT's own flat Q&A - dict/array forms) is passed to both engines exactly as given.""" + own docstring documents this shape. AI_EXTRACT's own flat Q&A dict/array forms reach + AI_EXTRACT exactly as given, but AI_COMPLETE rejects them, so for that engine they are + rewritten into a JSON Schema -- see _questions_as_json_schema.""" fields: List[str] field_columns: List[str] @@ -82,6 +83,43 @@ def _scalar_type(property_schema: Any) -> Optional[DataType]: return None return _JSON_SCHEMA_SCALAR_TYPES.get(json_type) + @staticmethod + def _questions_as_json_schema(response_format: Any, fields: List[str]) -> dict: + """Express AI_EXTRACT's flat question form as a JSON Schema, for AI_COMPLETE. + + ``{"vendor": "Who issued this invoice?"}`` is AI_EXTRACT's own shape and AI_COMPLETE + rejects it outright ("invalid response format object"), so sending it unchanged is + only safe while AI_EXTRACT is the engine. Each question becomes its field's + ``description``, which is how a model reads it under either engine. + + Every field is typed ``string``: a question carries no type information, so there is + nothing to infer one from. A field whose answer should be a list or a number needs a + real JSON Schema -- this conversion makes the call succeed, it cannot add a contract + the input never expressed. + """ + questions: Dict[str, Any] = {} + if isinstance(response_format, dict): + questions = dict(response_format) + elif isinstance(response_format, list): + # The array form carries its question inside the item: either a ["name", + # "question"] pair or a "name: question" string. Dropping it would discard the + # only instruction the model gets for that field. + for item in response_format: + if isinstance(item, (list, tuple)) and len(item) >= 2: + questions[str(item[0])] = item[1] + elif isinstance(item, str) and ":" in item: + name, _, question = item.partition(":") + questions[name.strip()] = question.strip() + + properties: Dict[str, Any] = {} + for field in fields: + node: Dict[str, Any] = {"type": "string"} + question = questions.get(field) + if isinstance(question, str) and question.strip(): + node["description"] = question + properties[field] = node + return {"type": "object", "properties": properties} + @classmethod def from_response_format(cls, response_format: Any) -> Optional["ExtractionSpec"]: if not response_format: @@ -108,7 +146,6 @@ def field_name(item: Any) -> str: fields = [] ai_extract_format = response_format - ai_complete_format = response_format if is_json_schema: ai_extract_format = {"schema": response_format} ai_complete_format = {"type": "json", "schema": response_format} @@ -117,6 +154,16 @@ def field_name(item: Any) -> str: field: cls._scalar_type(properties[field]) for field in fields } else: + # AI_EXTRACT keeps its own shape; AI_COMPLETE only accepts a JSON Schema, so the + # questions are rewritten into one rather than passed through to be rejected. + ai_complete_format = { + "type": "json", + "schema": cls._questions_as_json_schema(response_format, fields), + } + # field_types stays unset. The conversion types every field `string` purely + # because a question implies no type, and that is not grounds to start casting + # output columns -- least of all on the AI_EXTRACT path, which shares this and + # whose behaviour is unchanged here. field_types = {field: None for field in fields} return cls(