From ee45872d6dd5a6df874d588c61d545a96da2685a Mon Sep 17 00:00:00 2001 From: "spec-sync[bot]" Date: Fri, 18 Sep 2026 21:01:53 +0000 Subject: [PATCH 1/2] chore(spec-sync): update V2 spec snapshot + regenerated reference models --- specs/_generated/v2_models.py | 35 ++++++++++++++----- specs/v2-aide.json | 63 +++++++++++++++++++++++++++++------ 2 files changed, 80 insertions(+), 18 deletions(-) diff --git a/specs/_generated/v2_models.py b/specs/_generated/v2_models.py index 81c21cf..f242040 100644 --- a/specs/_generated/v2_models.py +++ b/specs/_generated/v2_models.py @@ -151,6 +151,15 @@ class Status(Enum): failed = 'failed' +class Type1(Enum): + """ + The node type. `page` is a page of a parsed document. `sheet` is one sheet of a parsed spreadsheet. + """ + + page = 'page' + sheet = 'sheet' + + class ServiceTier(Enum): """ The service tier the request ran in: `standard` or `priority`. A sync request reports `priority` (same lane, same price). @@ -2614,18 +2623,23 @@ class Grounding(BaseModel): can be lifted out of the tree and still locates its content. """ - box: Box = Field( + address: Optional[str] = Field( + None, + description='Spreadsheet only. Excel-style reference of the content this grounding covers, with the sheet name: `Sales!C5` for a cell, `Sales!C5:F20` for a table, the anchor cell (`Sales!B2`) for content parsed out of an embedded image. Stable across re-parses of the same file. Omitted for page-based documents.', + title='Address', + ) + box: Optional[Box] = Field( ..., - description="Bounding box in normalized page coordinates (`0`–`1` fractions of page width/height, at most 5 decimal places). A page node's box is always the full page `{0, 0, 1, 1}`.", + description="Bounding box in normalized page coordinates (`0`–`1` fractions of page width/height, at most 5 decimal places). A page node's box is always the full page `{0, 0, 1, 1}`. `null` (omitted from the response) when the source is a workbook, which has no visual position. For content parsed out of an image embedded in a spreadsheet, this is the fraction of that image, not of a page.", ) confidence: Optional[float] = Field( None, description='How sure the model is of the text in this segment, in `[0, 1]`. Present only on word-granularity `atomic_grounding` entries (`dpt-3-verity`), where it is the lowest per-character OCR confidence in the word — so a word is only as trustworthy as its weakest character. Omitted on node-level grounding and on models that ground at line granularity.', title='Confidence', ) - page: int = Field( + page: Optional[int] = Field( ..., - description="1-indexed page number this grounding is on. On a page node, the page's own number.", + description="1-indexed page number this grounding is on. On a page node, the page's own number. `null` (omitted from the response) when the source is a workbook, which has no page.", title='Page', ) range: Range = Field( @@ -2983,7 +2997,7 @@ class Element(BaseModel): ) id: str = Field( ..., - description='Semantic element id, unique within the document. Format `-`, where `` is a per-type 0-based counter assigned in reading order — `text-0` is the first text element in the document, `figure-0` the first figure, `table_cell-0` the first cell of the first table. Stable within a response but not across re-parses of the same document.', + description='Element id, unique within the document. An opaque string — do not parse it or assume a format. Stable within a response but not across re-parses of the same document; for spreadsheet content, use `grounding.address` instead, which IS stable across re-parses.', title='Id', ) markdown: Optional[str] = Field( @@ -3016,7 +3030,12 @@ class Page(BaseModel): ) grounding: Grounding = Field( ..., - description="The page's spatial data: `page` is the 1-indexed page number in the source document (not contiguous when `options.pages` filters out some pages); `range` covers this page's content in the top-level `markdown` string (zero-length `start == end` for failed pages); `box` is always the full page `{0, 0, 1, 1}`.", + description="The node's spatial data. On a `page` node: `page` is the 1-indexed page number in the source document (not contiguous when `options.pages` filters out some pages); `range` covers this page's content in the top-level `markdown` string (zero-length `start == end` for failed pages); `box` is always the full page `{0, 0, 1, 1}`. On a `sheet` node: `page` and `box` are both `null` (a spreadsheet sheet has no page number or visual position); `range` covers the sheet's content in the top-level `markdown` string.", + ) + id: Optional[str] = Field( + None, + description='The sheet name, present only on a `sheet` node. `null` (omitted from the response) on a `page` node.', + title='Id', ) markdown: Optional[str] = Field( None, @@ -3033,9 +3052,9 @@ class Page(BaseModel): description='Whether this page was parsed successfully (`ok`) or failed (`failed`).', title='Status', ) - type: Literal['page'] = Field( + type: Optional[Type1] = Field( 'page', - description='The node type. Identifies this node as a page in the structure tree.', + description='The node type. `page` is a page of a parsed document. `sheet` is one sheet of a parsed spreadsheet.', title='Type', ) diff --git a/specs/v2-aide.json b/specs/v2-aide.json index 1bfa2c5..ac81b2f 100644 --- a/specs/v2-aide.json +++ b/specs/v2-aide.json @@ -268,7 +268,7 @@ "description": "The element's spatial data: the page it appears on, its `[start, end)` range in the top-level `markdown` string, and its bounding box in normalized page coordinates." }, "id": { - "description": "Semantic element id, unique within the document. Format `-`, where `` is a per-type 0-based counter assigned in reading order — `text-0` is the first text element in the document, `figure-0` the first figure, `table_cell-0` the first cell of the first table. Stable within a response but not across re-parses of the same document.", + "description": "Element id, unique within the document. An opaque string — do not parse it or assume a format. Stable within a response but not across re-parses of the same document; for spreadsheet content, use `grounding.address` instead, which IS stable across re-parses.", "title": "Id", "type": "string" }, @@ -369,9 +369,29 @@ "Grounding": { "description": "Where a node lives: its page, its slice of `markdown`, and its box.\n\nThe same shape is used for page nodes, element nodes, and each\n`atomic_grounding` entry, so any grounding object is self-contained — it\ncan be lifted out of the tree and still locates its content.", "properties": { + "address": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "description": "Spreadsheet only. Excel-style reference of the content this grounding covers, with the sheet name: `Sales!C5` for a cell, `Sales!C5:F20` for a table, the anchor cell (`Sales!B2`) for content parsed out of an embedded image. Stable across re-parses of the same file. Omitted for page-based documents.", + "title": "Address" + }, "box": { - "$ref": "#/components/schemas/Box", - "description": "Bounding box in normalized page coordinates (`0`–`1` fractions of page width/height, at most 5 decimal places). A page node's box is always the full page `{0, 0, 1, 1}`." + "anyOf": [ + { + "$ref": "#/components/schemas/Box" + }, + { + "type": "null" + } + ], + "description": "Bounding box in normalized page coordinates (`0`–`1` fractions of page width/height, at most 5 decimal places). A page node's box is always the full page `{0, 0, 1, 1}`. `null` (omitted from the response) when the source is a workbook, which has no visual position. For content parsed out of an image embedded in a spreadsheet, this is the fraction of that image, not of a page." }, "confidence": { "anyOf": [ @@ -387,9 +407,16 @@ "title": "Confidence" }, "page": { - "description": "1-indexed page number this grounding is on. On a page node, the page's own number.", - "title": "Page", - "type": "integer" + "anyOf": [ + { + "type": "integer" + }, + { + "type": "null" + } + ], + "description": "1-indexed page number this grounding is on. On a page node, the page's own number. `null` (omitted from the response) when the source is a workbook, which has no page.", + "title": "Page" }, "range": { "$ref": "#/components/schemas/Range", @@ -416,7 +443,20 @@ }, "grounding": { "$ref": "#/components/schemas/Grounding", - "description": "The page's spatial data: `page` is the 1-indexed page number in the source document (not contiguous when `options.pages` filters out some pages); `range` covers this page's content in the top-level `markdown` string (zero-length `start == end` for failed pages); `box` is always the full page `{0, 0, 1, 1}`." + "description": "The node's spatial data. On a `page` node: `page` is the 1-indexed page number in the source document (not contiguous when `options.pages` filters out some pages); `range` covers this page's content in the top-level `markdown` string (zero-length `start == end` for failed pages); `box` is always the full page `{0, 0, 1, 1}`. On a `sheet` node: `page` and `box` are both `null` (a spreadsheet sheet has no page number or visual position); `range` covers the sheet's content in the top-level `markdown` string." + }, + "id": { + "anyOf": [ + { + "type": "string" + }, + { + "type": "null" + } + ], + "default": null, + "description": "The sheet name, present only on a `sheet` node. `null` (omitted from the response) on a `page` node.", + "title": "Id" }, "markdown": { "anyOf": [ @@ -455,9 +495,12 @@ "type": "string" }, "type": { - "const": "page", "default": "page", - "description": "The node type. Identifies this node as a page in the structure tree.", + "description": "The node type. `page` is a page of a parsed document. `sheet` is one sheet of a parsed spreadsheet.", + "enum": [ + "page", + "sheet" + ], "title": "Type", "type": "string" } @@ -1471,7 +1514,7 @@ "info": { "description": "AIDE Gateway — proxy surface + Temporal job surface.", "title": "AIDE Gateway", - "version": "1.0.0" + "version": "1.1.0" }, "openapi": "3.1.0", "paths": { From 8f06bc0a9517787994390bbf3279e7966b17a151 Mon Sep 17 00:00:00 2001 From: "spec-sync[bot]" Date: Fri, 18 Sep 2026 21:06:27 +0000 Subject: [PATCH 2/2] feat(spec-sync): wire client.v2 to spec diff (AI) --- api.md | 2 +- docs/v2-testing.md | 46 +++++++++- src/landingai_ade/types/v2/parse_response.py | 32 ++++++- tests/api_resources/v2/test_parse.py | 89 ++++++++++++++++++++ tests/contract/test_v2_smoke.py | 46 ++++++++++ tests/test_v2_types.py | 68 +++++++++++++++ 6 files changed, 277 insertions(+), 6 deletions(-) diff --git a/api.md b/api.md index 3974a1a..9cf4cc4 100644 --- a/api.md +++ b/api.md @@ -100,7 +100,7 @@ from landingai_ade.types.v2 import ( ``` - Job -- unified job shape: `job_id`, `status` (JobStatus: `pending` / `processing` / `completed` / `failed` / `cancelled`), `created_at`, `completed_at`, `progress`, `result` (a `V2ParseResponse` for parse jobs, a `V2ExtractResult` for extract jobs, a `V2BuildSchemaResponse` for build-schema jobs, or `None` until completion), `error` (JobError), `metadata` (the result's metadata receipt as a `dict`, populated top-level only when `output_save_url` was set and the result was delivered to `output_url` instead of inline; `None` otherwise, since inline jobs carry it on `result.metadata`), `raw` (the full original envelope as a `dict`), and the `.is_terminal` property. -- V2ParseResponse -- `markdown`, `structure`, `grounding`, `metadata` (V2ParseMetadata, which nests V2ParseBilling and carries `output_markdown_chars`, `range_units`, and `openapi_spec`). `structure` is a typed V2ParseStructure tree (`document` → V2ParsePage → V2ParseElement); each node below the root carries its spatial data inline in a V2ParseNodeGrounding (`page`, V2ParseRange, V2ParseBox, normalized page coordinates, and an optional `confidence` in `[0, 1]` that is present only on word-granularity `atomic_grounding` segments (`dpt-3-verity`), where it is the lowest per-character OCR confidence in the word, and that is `None` on node-level grounding and on line-granularity models (`dpt-3-pro`)), and leaf elements additionally carry an `atomic_grounding` list. With `options.inline_markdown`, each node also carries its `markdown` slice. The legacy top-level `grounding` tree (V2ParseGrounding → `V2ParseGroundingPage` → `V2ParseGroundingElement` → `V2ParseGroundingEntry`) is retained for older gateway responses. Element `type`/page `status` are permissive strings and unknown keys are retained. +- V2ParseResponse -- `markdown`, `structure`, `grounding`, `metadata` (V2ParseMetadata, which nests V2ParseBilling and carries `output_markdown_chars`, `range_units`, and `openapi_spec`). `structure` is a typed V2ParseStructure tree (`document` → V2ParsePage → V2ParseElement); each node below the root carries its spatial data inline in a V2ParseNodeGrounding (`page`, V2ParseRange, V2ParseBox, normalized page coordinates, and an optional `confidence` in `[0, 1]` that is present only on word-granularity `atomic_grounding` segments (`dpt-3-verity`), where it is the lowest per-character OCR confidence in the word, and that is `None` on node-level grounding and on line-granularity models (`dpt-3-pro`)), and leaf elements additionally carry an `atomic_grounding` list. A `V2ParsePage` node is a `page` of a parsed document or a `sheet` of a parsed spreadsheet (`type`); a `sheet` node names itself in `id` (`None` on a `page` node). For spreadsheet content, `V2ParseNodeGrounding.page` and `.box` are `None` -- a workbook has no page number and no visual position -- and the Excel-style `address` (`Sales!C5`, `Sales!C5:F20`) locates the content instead; `address` is `None` for page-based documents. `V2ParseElement.id` is an opaque string: do not parse it or assume a format, and prefer `grounding.address`, which is stable across re-parses of the same spreadsheet. With `options.inline_markdown`, each node also carries its `markdown` slice. The legacy top-level `grounding` tree (V2ParseGrounding → `V2ParseGroundingPage` → `V2ParseGroundingElement` → `V2ParseGroundingEntry`) is retained for older gateway responses. Element `type`/page `status` are permissive strings and unknown keys are retained. - V2ExtractResult -- `extraction`, `extraction_metadata`, `markdown`, `output_ref`, `schema_violation_error` (set when `strict=False` and the schema had unextractable fields), `warnings`, and `metadata` (V2ExtractMetadata, which carries `model_version`, `input_markdown_chars`, `output_extraction_chars`, `range_units`, `openapi_spec`, and nests V2ExtractBilling). - V2BuildSchemaResponse -- `extraction_schema` (the generated JSON Schema serialized as a string) and `metadata` (V2BuildSchemaMetadata: `job_id`, `duration_ms`, `openapi_spec`, `filename`/`org_id`/`version` (retained for compatibility), a `warnings` list of V2BuildSchemaWarning (`code`, `msg`), and nested V2BuildSchemaBilling). - V2GroundResult -- `grounding` (a tree mirroring the input `extraction_metadata`, each `{value, ranges}` leaf replaced by the list of `structure` blocks its ranges overlap) and `metadata` (V2GroundMetadata: `job_id`, `duration_ms`, `openapi_spec`, and nested V2GroundBilling). diff --git a/docs/v2-testing.md b/docs/v2-testing.md index 6648919..30a7329 100644 --- a/docs/v2-testing.md +++ b/docs/v2-testing.md @@ -28,15 +28,18 @@ LANDINGAI_ADE_STAGING_APIKEY=... rye run pytest tests/contract/test_v2_smoke.py `V2ParseResponse` with: - `markdown` -- the full document as one Markdown string. -- `structure` (`V2ParseStructure`) -- the `document → page → element` tree. +- `structure` (`V2ParseStructure`) -- the `document → page | sheet → element` tree. **Every node below the root carries its spatial data inline** in a `V2ParseNodeGrounding` object (`grounding`): - - `page` -- 1-indexed page number. + - `page` -- 1-indexed page number. `None` for spreadsheet content (see below). - `range` (`V2ParseRange`) -- `{start, end}` code-point offsets into `markdown` (`metadata.range_units` names the unit, always `"unicode_codepoints"`). - `box` (`V2ParseBox`) -- `{xmin, ymin, xmax, ymax}` as `[0, 1]` fractions of the page width/height (a page node's box is the full page `{0, 0, 1, 1}`). + `None` for spreadsheet content. + - `address` -- spreadsheet-only Excel-style reference; `None` for page-based + documents (see below). - `confidence` -- an optional `[0, 1]` probability. It is present **only** on word-granularity `atomic_grounding` segments (`dpt-3-verity`), where it is the lowest per-character OCR confidence in the word -- so a word is only as @@ -59,6 +62,45 @@ The legacy top-level `grounding` tree (`V2ParseGrounding` and friends) is retain on the model for backward compatibility with older gateway responses; current responses omit it in favor of the inline `grounding` above. +### Spreadsheets: `sheet` nodes and `grounding.address` + +The tree under `structure` now covers workbooks as well as page-based documents. +Nothing was renamed — a spreadsheet reuses `V2ParsePage` and +`V2ParseNodeGrounding`, with a different set of fields populated: + +- **`V2ParsePage.type` is `"page"` or `"sheet"`.** It was a `const: "page"` in the + previous snapshot and is an enum now. The SDK already typed it as a permissive + `str` (not a `Literal`), so `"sheet"` deserializes without a model change — + but code that *compared* against `"page"` to find the page nodes will silently + skip every sheet. +- **`V2ParsePage.id`** (new) is the sheet name, e.g. `"Sales"`. It is present only + on a `sheet` node and `None` on a `page` node. +- **`V2ParseNodeGrounding.address`** (new) is the Excel-style reference of the + content, sheet name included: `Sales!C5` for a cell, `Sales!C5:F20` for a table, + the anchor cell (`Sales!B2`) for content parsed out of an embedded image. It is + `None` for page-based documents, and unlike `V2ParseElement.id` it is **stable + across re-parses of the same file** — it is the right key to join a re-parse on. +- **`grounding.page` and `grounding.box` are now `Optional`** on the wire, not just + in the model: a workbook has no page number and no visual position, so both come + back `null`/omitted for spreadsheet content. (The model already declared them + `Optional[...] = None`, so this needed no code change; treat missing and `None` + alike.) The exception is content parsed out of an image embedded in a + spreadsheet, where `box` is the fraction *of that image*, not of a page. +- **`V2ParseElement.id` is documented as opaque.** The `-` format + (`text-0`, `table_cell-0`) is gone from the spec — do not parse it or assume a + shape. It is still unique within a response and still unstable across re-parses. + +Testing it: the live suite parses a PDF, so only the page-based half is assertable +there, and only as absence — `test_parse_node_locators_match_the_source_kind` in +`tests/contract/test_v2_smoke.py` walks the tree and asserts `address` and the node +`id` are `None` while `page`/`box` are populated. That much is a spec guarantee for +a page-based document. The spreadsheet half needs a workbook fixture staging is not +guaranteed to accept, so it is pinned deterministically against a mocked body +instead: `test_parse_sync_spreadsheet_sheet_nodes_and_addresses` in +`tests/api_resources/v2/test_parse.py` (plus `test_parse_response_spreadsheet_sheet_nodes` +in `tests/test_v2_types.py`) asserts the `sheet` node, its `id`, the per-node +`address`, and the `None` `page`/`box`. + ### Encrypted PDFs (`password`) `options.password` is a **supported** parse option — earlier snapshots documented diff --git a/src/landingai_ade/types/v2/parse_response.py b/src/landingai_ade/types/v2/parse_response.py index ed6fb2d..3c8c763 100644 --- a/src/landingai_ade/types/v2/parse_response.py +++ b/src/landingai_ade/types/v2/parse_response.py @@ -71,11 +71,23 @@ class V2ParseRange(BaseModel): class V2ParseNodeGrounding(BaseModel): - """Where a node lives: its 1-indexed `page`, its `range` in `markdown`, and its `box`.""" + """Where a node lives: its 1-indexed `page`, its `range` in `markdown`, and its `box`. + Spreadsheet sources have no pages and no visual layout, so `page` and `box` are + both omitted there and the Excel-style `address` locates the content instead.""" + + # `null`/omitted when the source is a workbook, which has no page. page: Optional[int] = None range: Optional[V2ParseRange] = None + # `null`/omitted when the source is a workbook, which has no visual position. For + # content parsed out of an image embedded in a spreadsheet, this is the fraction + # of that image, not of a page. box: Optional[V2ParseBox] = None + # Spreadsheet only. Excel-style reference of the content this grounding covers, + # with the sheet name: `Sales!C5` for a cell, `Sales!C5:F20` for a table, the + # anchor cell (`Sales!B2`) for content parsed out of an embedded image. Stable + # across re-parses of the same file. Omitted for page-based documents. + address: Optional[str] = None # How sure the model is of the text in this segment, in `[0, 1]`. # Present only on word-granularity `atomic_grounding` entries # (`dpt-3-verity`), where it is the lowest per-character OCR confidence in the @@ -96,6 +108,10 @@ class V2ParseElement(BaseModel): optional fields are only present for the relevant element types.""" type: str + # Element id, unique within the document. An opaque string -- do not parse it or + # assume a format. Stable within a response but not across re-parses of the same + # document; for spreadsheet content, `grounding.address` IS stable across + # re-parses. id: str # Deprecated: replaced by `grounding.range` upstream; populated only by older # gateway responses. @@ -120,7 +136,14 @@ class V2ParseElement(BaseModel): class V2ParsePage(BaseModel): + """A node of the document tree: a `page` of a parsed document, or a `sheet` of a + parsed spreadsheet.""" + + # `page` for a page of a parsed document, `sheet` for one sheet of a parsed + # spreadsheet. type: str = "page" + # The sheet name, present only on a `sheet` node; omitted on a `page` node. + id: Optional[str] = None # Deprecated: page number is now carried on `grounding.page` (1-indexed); # populated only by older responses. page: Optional[int] = None @@ -131,7 +154,10 @@ class V2ParsePage(BaseModel): width: Optional[int] = None height: Optional[int] = None dpi: Optional[int] = None - # The page's spatial data (`{page, range, box}`); `box` is the full page. + # The node's spatial data. On a `page` node `box` is the full page + # `{0, 0, 1, 1}` and `page` is the 1-indexed page number; on a `sheet` node both + # are omitted (a sheet has no page number or visual position) and only `range` + # is meaningful. grounding: Optional[V2ParseNodeGrounding] = None # This page's slice of the top-level `markdown`; only when # `options.inline_markdown` is true. @@ -142,7 +168,7 @@ class V2ParsePage(BaseModel): class V2ParseStructure(BaseModel): - """Root of the `structure` tree (`document -> page -> element`).""" + """Root of the `structure` tree (`document -> page | sheet -> element`).""" type: str = "document" # The full document markdown; only when `options.inline_markdown` is true. diff --git a/tests/api_resources/v2/test_parse.py b/tests/api_resources/v2/test_parse.py index 5877760..6ffa5d6 100644 --- a/tests/api_resources/v2/test_parse.py +++ b/tests/api_resources/v2/test_parse.py @@ -165,6 +165,50 @@ def multipart_field(body: bytes, name: str) -> Optional[str]: } +# A ParseResponse for a parsed spreadsheet: the `structure` tree holds `sheet` nodes +# (named by `id`) rather than `page` nodes, and each grounding locates its content by +# Excel-style `address` instead of `page` + `box`, both of which a workbook lacks. +SHEET_PARSE_BODY: Dict[str, Any] = { + "markdown": "| Q1 |\n| --- |\n| 1250000 |", + "metadata": { + "req_id": "r1", + "job_id": "parse-2", + "model_version": "dpt-3", + "page_count": 1, + "failed_pages": [], + "output_markdown_chars": 26, + "range_units": "unicode_codepoints", + }, + "structure": { + "type": "document", + "children": [ + { + "type": "sheet", + "id": "Sales", + "status": "ok", + "grounding": {"range": {"start": 0, "end": 26}, "address": "Sales"}, + "children": [ + { + "type": "table", + "id": "9f2c1b", + "grounding": {"range": {"start": 0, "end": 26}, "address": "Sales!C5:F20"}, + "children": [ + { + "type": "table_cell", + "id": "4ae07d", + "row": 0, + "col": 0, + "grounding": {"range": {"start": 0, "end": 6}, "address": "Sales!C5"}, + } + ], + } + ], + } + ], + }, +} + + @respx.mock def test_parse_sync_ok_routes_to_v2_and_sends_options_json() -> None: client = LandingAIADE(apikey=APIKEY, environment="production") @@ -514,6 +558,51 @@ def test_parse_sync_inline_grounding_structure() -> None: assert result.metadata.openapi_spec is not None +@respx.mock +def test_parse_sync_spreadsheet_sheet_nodes_and_addresses() -> None: + # A parsed workbook uses the same typed tree: `sheet` nodes carrying the sheet + # name as `id`, and grounding located by the Excel-style `address` instead of + # `page` + `box`, which a workbook has neither of. + from landingai_ade.types.v2 import V2ParseStructure, V2ParseNodeGrounding + + client = LandingAIADE(apikey=APIKEY) + respx.post("https://api.ade.landing.ai/v2/parse").mock(return_value=httpx.Response(200, json=SHEET_PARSE_BODY)) + result = client.v2.parse(document=b"xlsx") + + assert isinstance(result.structure, V2ParseStructure) + sheet = result.structure.children[0] + assert sheet.type == "sheet" and sheet.id == "Sales" + assert isinstance(sheet.grounding, V2ParseNodeGrounding) + assert sheet.grounding.address == "Sales" + # Absent on a workbook, not merely empty. + assert sheet.grounding.page is None and sheet.grounding.box is None + assert sheet.grounding.range is not None and sheet.grounding.range.end == 26 + + table = sheet.children[0] + assert table.grounding is not None and table.grounding.address == "Sales!C5:F20" + assert table.children is not None + cell = table.children[0] + assert cell.grounding is not None and cell.grounding.address == "Sales!C5" + # Element ids are opaque -- `4ae07d` has no `-` shape to parse. + assert cell.id == "4ae07d" + + +@respx.mock +def test_parse_sync_page_nodes_omit_the_spreadsheet_locators() -> None: + # The page-based half of the same shape: `id` on the node and `address` on the + # grounding are spreadsheet-only, so a PDF response leaves both `None`. + client = LandingAIADE(apikey=APIKEY) + respx.post("https://api.ade.landing.ai/v2/parse").mock(return_value=httpx.Response(200, json=INLINE_PARSE_BODY)) + result = client.v2.parse(document=b"pdf") + + assert result.structure is not None + page = result.structure.children[0] + assert page.type == "page" and page.id is None + assert page.grounding is not None and page.grounding.address is None + el = page.children[0] + assert el.grounding is not None and el.grounding.address is None + + @respx.mock def test_parse_job_get_result_envelope_and_error() -> None: # Current parse-job GET carries the response under `result` (not legacy `data`) diff --git a/tests/contract/test_v2_smoke.py b/tests/contract/test_v2_smoke.py index bec3aba..332ee4c 100644 --- a/tests/contract/test_v2_smoke.py +++ b/tests/contract/test_v2_smoke.py @@ -154,6 +154,52 @@ def _walk(elements: List[V2ParseElement]) -> None: _walk(page.children) +def test_parse_node_locators_match_the_source_kind(staging_client: LandingAIADE) -> None: + # The structure tree now covers spreadsheets as well as page-based documents: a + # node is a `page` or a `sheet`, a `sheet` node names itself in `id`, and a + # grounding locates spreadsheet content by the Excel-style `grounding.address` + # instead of `page` + `box`, which a workbook has neither of. + # + # The fixture here is a PDF, so only the page-based half is assertable, and it is + # a spec guarantee rather than an environment property: `id` and `address` are + # documented as spreadsheet-only, so they are absent on every node of this + # response, while `page` and `box` are present. The spreadsheet half needs a + # workbook fixture staging is not guaranteed to accept; it is pinned + # deterministically instead by `test_parse_sync_spreadsheet_sheet_nodes_and_addresses` + # in tests/api_resources/v2/test_parse.py. + pdf = Path(__file__).parent / "sample.pdf" + resp = staging_client.v2.parse(document=pdf) + assert isinstance(resp, V2ParseResponse) + assert resp.structure is not None and resp.structure.children + + def _check_grounding(grounding: Optional[V2ParseNodeGrounding]) -> None: + if grounding is None: + return + # Spreadsheet-only locator: absent for a page-based document. + assert grounding.address is None + # Absent-or-valid: a page-based node that carries a page carries a 1-indexed one. + if grounding.page is not None: + assert grounding.page >= 1 + + def _walk(elements: List[V2ParseElement]) -> None: + for el in elements: + _check_grounding(el.grounding) + for seg in el.atomic_grounding or []: + _check_grounding(seg) + _walk(el.children or []) + + for page in resp.structure.children: + # A novel node type must not fail the client, but this PDF yields `page` nodes. + assert page.type in ("page", "sheet") + if page.type == "page": + # The sheet name is `sheet`-only, and a page node is grounded on a page. + assert page.id is None + assert page.grounding is not None and page.grounding.page is not None + assert page.grounding.box is not None + _check_grounding(page.grounding) + _walk(page.children) + + def test_parse_sync_password_requires_pdf(staging_client: LandingAIADE) -> None: # `options.password` is now a supported parse option (an earlier snapshot # documented it as unimplemented). The one half of its contract staging is diff --git a/tests/test_v2_types.py b/tests/test_v2_types.py index d29b081..675b74e 100644 --- a/tests/test_v2_types.py +++ b/tests/test_v2_types.py @@ -233,6 +233,74 @@ def test_parse_response_inline_grounding_and_metadata() -> None: assert r.metadata.openapi_spec is not None +def test_parse_response_spreadsheet_sheet_nodes() -> None: + # A parsed spreadsheet comes back as the same tree with `sheet` nodes instead of + # `page` nodes: the node carries the sheet name as `id`, and its grounding has no + # `page` and no `box` (a sheet has no page number or visual position) -- the + # Excel-style `grounding.address` locates the content instead. + r = V2ParseResponse( + markdown="| Q1 |\n| --- |\n| 1250000 |", + structure={ # type: ignore[arg-type] + "type": "document", + "children": [ + { + "type": "sheet", + "id": "Sales", + "status": "ok", + "grounding": {"range": {"start": 0, "end": 26}, "address": "Sales"}, + "children": [ + { + "type": "table", + "id": "9f2c1b", + "grounding": { + "range": {"start": 0, "end": 26}, + "address": "Sales!C5:F20", + }, + } + ], + } + ], + }, + ) + assert r.structure is not None + sheet = r.structure.children[0] + assert sheet.type == "sheet" and sheet.id == "Sales" + assert sheet.grounding is not None + # Both absent on a workbook, not merely falsy. + assert sheet.grounding.page is None and sheet.grounding.box is None + assert sheet.grounding.address == "Sales" + table = sheet.children[0] + assert table.grounding is not None and table.grounding.address == "Sales!C5:F20" + # `id` is an opaque string now -- no `-` format to key off. + assert table.id == "9f2c1b" + + +def test_parse_response_page_nodes_omit_the_spreadsheet_locators() -> None: + # The page-based half of the same shape: a `page` node has no sheet name and its + # grounding has no `address`, so both read as `None` rather than raising. + r = V2ParseResponse( + markdown="# hi", + structure={ # type: ignore[arg-type] + "type": "document", + "children": [ + { + "type": "page", + "grounding": { + "page": 1, + "range": {"start": 0, "end": 4}, + "box": {"xmin": 0, "ymin": 0, "xmax": 1, "ymax": 1}, + }, + } + ], + }, + ) + assert r.structure is not None + page = r.structure.children[0] + assert page.type == "page" and page.id is None + assert page.grounding is not None and page.grounding.address is None + assert page.grounding.page == 1 and page.grounding.box is not None + + def test_extract_result_metadata_char_counts_warnings_and_schema_violation() -> None: # The char counters moved onto `metadata` (from `billing`) upstream, and # `schema_violation_error` / `warnings` were added to the result.