diff --git a/api.md b/api.md index 67c1625..daa31df 100644 --- a/api.md +++ b/api.md @@ -99,7 +99,7 @@ from landingai_ade.types.v2 import ( ) ``` -- Job -- unified job shape: `job_id`, `status` (JobStatus: `pending` / `processing` / `completed` / `failed` / `cancelled`), `created_at`, `completed_at`, `progress`, `result` (a `V2ParseResponse` for parse jobs, a `V2ExtractResult` for extract jobs, a `V2BuildSchemaResponse` for build-schema jobs, or `None` until completion), `error` (JobError), `raw` (the full original envelope as a `dict`), and the `.is_terminal` property. +- Job -- unified job shape: `job_id`, `status` (JobStatus: `pending` / `processing` / `completed` / `failed` / `cancelled`), `created_at`, `completed_at`, `progress`, `result` (a `V2ParseResponse` for parse jobs, a `V2ExtractResult` for extract jobs, a `V2BuildSchemaResponse` for build-schema jobs, or `None` until completion), `error` (JobError), `metadata` (the result's metadata receipt as a `dict`, populated top-level only when `output_save_url` was set and the result was delivered to `output_url` instead of inline; `None` otherwise, since inline jobs carry it on `result.metadata`), `raw` (the full original envelope as a `dict`), and the `.is_terminal` property. - V2ParseResponse -- `markdown`, `structure`, `grounding`, `metadata` (V2ParseMetadata, which nests V2ParseBilling and carries `output_markdown_chars`, `range_units`, and `openapi_spec`). `structure` is a typed V2ParseStructure tree (`document` → V2ParsePageV2ParseElement); each node below the root carries its spatial data inline in a V2ParseNodeGrounding (`page`, V2ParseRange, V2ParseBox, normalized page coordinates), and leaf elements additionally carry an `atomic_grounding` list. With `options.inline_markdown`, each node also carries its `markdown` slice. The legacy top-level `grounding` tree (V2ParseGrounding → `V2ParseGroundingPage` → `V2ParseGroundingElement` → `V2ParseGroundingEntry`) is retained for older gateway responses. Element `type`/page `status` are permissive strings and unknown keys are retained. - V2ExtractResult -- `extraction`, `extraction_metadata`, `markdown`, `output_ref`, `schema_violation_error` (set when `strict=False` and the schema had unextractable fields), `warnings`, and `metadata` (V2ExtractMetadata, which carries `model_version`, `input_markdown_chars`, `output_extraction_chars`, `range_units`, `openapi_spec`, and nests V2ExtractBilling). - V2BuildSchemaResponse -- `extraction_schema` (the generated JSON Schema serialized as a string) and `metadata` (V2BuildSchemaMetadata: `job_id`, `duration_ms`, `openapi_spec`, `filename`/`org_id`/`version` (retained for compatibility), a `warnings` list of V2BuildSchemaWarning (`code`, `msg`), and nested V2BuildSchemaBilling). diff --git a/docs/v2-testing.md b/docs/v2-testing.md index f669a5b..ae51ba8 100644 --- a/docs/v2-testing.md +++ b/docs/v2-testing.md @@ -66,7 +66,11 @@ upstream; both are retained on `V2ExtractBilling` for backward compatibility. The async `extract_jobs.create` also accepts `output_save_url` (async jobs only): when set, the finished result is delivered to that URL and the completed job -reports `output_url` (on `Job.raw`) instead of an inline `result`. +reports `output_url` (on `Job.raw`) instead of an inline `result`. The metadata +receipt (billing included) still rides back on the job and is surfaced as a +`dict` on `Job.metadata` — the delivery moves the content, not the receipt. For +inline jobs `Job.metadata` is `None` and the metadata lives on +`result.metadata` instead. `parse_jobs` behaves the same way. ## Current build-schema-response shape @@ -109,6 +113,8 @@ field-name drift: - Failures arrive as a structured `error` object (`{code, message}`); older parse envelopes used a flat `failure_reason` string. Both map to `Job.error`. - `created_at` / `completed_at` accept ISO-8601 strings or epoch seconds. +- A top-level `metadata` object (present on `output_save_url` deliveries) is + passed through to `Job.metadata`; inline jobs leave it `None`. - Unknown / renamed `status` values fall back to `pending` rather than raising; the raw envelope is always preserved on `Job.raw`. diff --git a/specs/_generated/v2_models.py b/specs/_generated/v2_models.py index 2d60b3d..e74ed21 100644 --- a/specs/_generated/v2_models.py +++ b/specs/_generated/v2_models.py @@ -546,82 +546,7 @@ class V2ExtractPostResponse(BaseModel): ) -class V2ExtractBuildSchemaPostRequest(BaseModel): - """ - Input to V2BuildSchemaOperationWorkflow — the ``/v2/extract/build-schema`` - request body. - - Mirrors VTRA's ``BuildSchemaRequest``: generate a JSON Schema from one or - more source markdown documents and/or a natural-language ``prompt``, and/or - iterate on an existing ``schema``. At least one of ``markdowns`` / - ``markdown_urls`` / ``prompt`` / ``schema`` must be provided. - """ - - markdown_urls: Optional[list[str]] = Field( - None, - description='URLs to Markdown files to analyze for schema generation.', - title='Markdown Urls', - ) - markdowns: Optional[list[str]] = Field( - None, - description='Markdown files or inline content strings to analyze for schema generation. Multiple documents can be provided for better schema coverage.', - title='Markdowns', - ) - prompt: Optional[str] = Field( - None, - description='Instructions for how to generate or modify the schema.', - title='Prompt', - ) - schema_: Optional[str] = Field( - None, - alias='schema', - description='Existing JSON schema to iterate on or refine.', - title='Schema', - ) - - -class V2ExtractBuildSchemaPostRequest1(BaseModel): - markdown_urls: Optional[list[str]] = Field( - None, - description='URLs to Markdown files to analyze for schema generation. JSON-serialized string in form data.', - title='Markdown Urls', - ) - markdowns: Optional[list[Union[str, bytes]]] = Field( - None, description='Repeat the field for each file upload.' - ) - prompt: Optional[str] = Field( - None, - description='Instructions for how to generate or modify the schema. JSON-serialized string in form data.', - title='Prompt', - ) - schema_: Optional[str] = Field( - None, - alias='schema', - description='Existing JSON schema to iterate on or refine. JSON-serialized string in form data.', - title='Schema', - ) - - -class V2ExtractBuildSchemaPostResponse(BaseModel): - """ - Result returned by V2BuildSchemaOperationWorkflow — the - ``/v2/extract/build-schema`` response body (VTRA ``BuildSchemaResponse``). - - ``extraction_schema`` is the generated JSON Schema serialized as a STRING - (VTRA parity — the v1 field is a string, not an object). - """ - - extraction_schema: str = Field( - ..., - description='The generated JSON schema as a string.', - title='Extraction Schema', - ) - metadata: V2BuildSchemaMetadata = Field( - ..., description='The metadata for the schema generation process.' - ) - - -class V2ExtractBuildSchemaJobsGetParametersQuery(BaseModel): +class V2ExtractJobsGetParametersQuery(BaseModel): page: Optional[int] = Field( 0, description='Page number (0-indexed).', ge=0, title='Page' ) @@ -646,13 +571,13 @@ class Job(BaseModel): failure_reason: Optional[str] = None job_id: Optional[str] = Field( None, - description='The unique identifier for this v2-build-schema job. Format: ``extract-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', + description='The unique identifier for this v2-extract job. Format: ``extract-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', ) model_version: Optional[str] = None status: Optional[Status1] = None -class V2ExtractBuildSchemaJobsGetResponse(BaseModel): +class V2ExtractJobsGetResponse(BaseModel): has_more: Optional[bool] = None jobs: Optional[list[Job]] = None page: Optional[int] = None @@ -668,164 +593,6 @@ class ServiceTier2(Enum): priority = 'priority' -class V2ExtractBuildSchemaJobsPostRequest(BaseModel): - """ - Input to V2BuildSchemaOperationWorkflow — the ``/v2/extract/build-schema`` - request body. - - Mirrors VTRA's ``BuildSchemaRequest``: generate a JSON Schema from one or - more source markdown documents and/or a natural-language ``prompt``, and/or - iterate on an existing ``schema``. At least one of ``markdowns`` / - ``markdown_urls`` / ``prompt`` / ``schema`` must be provided. - """ - - markdown_urls: Optional[list[str]] = Field( - None, - description='URLs to Markdown files to analyze for schema generation.', - title='Markdown Urls', - ) - markdowns: Optional[list[str]] = Field( - None, - description='Markdown files or inline content strings to analyze for schema generation. Multiple documents can be provided for better schema coverage.', - title='Markdowns', - ) - prompt: Optional[str] = Field( - None, - description='Instructions for how to generate or modify the schema.', - title='Prompt', - ) - schema_: Optional[str] = Field( - None, - alias='schema', - description='Existing JSON schema to iterate on or refine.', - title='Schema', - ) - service_tier: Optional[ServiceTier2] = Field( - None, - description='Async service tier. ``priority`` runs in the fast lane at the sync billing rate; absent → ``standard``.', - ) - - -class V2ExtractBuildSchemaJobsPostRequest1(BaseModel): - markdown_urls: Optional[list[str]] = Field( - None, - description='URLs to Markdown files to analyze for schema generation. JSON-serialized string in form data.', - title='Markdown Urls', - ) - markdowns: Optional[list[Union[str, bytes]]] = Field( - None, description='Repeat the field for each file upload.' - ) - prompt: Optional[str] = Field( - None, - description='Instructions for how to generate or modify the schema. JSON-serialized string in form data.', - title='Prompt', - ) - schema_: Optional[str] = Field( - None, - alias='schema', - description='Existing JSON schema to iterate on or refine. JSON-serialized string in form data.', - title='Schema', - ) - service_tier: Optional[ServiceTier2] = Field( - None, - description='Async service tier. ``priority`` runs in the fast lane at the sync billing rate; absent → ``standard``.', - ) - - -class V2ExtractBuildSchemaJobsPostResponse(BaseModel): - created_at: Optional[str] = None - job_id: Optional[str] = Field( - None, - description='The unique identifier for this v2-build-schema job. Format: ``extract-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', - ) - status: Optional[Status1] = None - - -class Error(BaseModel): - """ - Present once status is ``failed``. - """ - - code: Optional[str] = Field( - None, description='Stable error code (``internal_error`` when unmapped).' - ) - message: Optional[str] = None - - -class Result(BaseModel): - """ - Result returned by V2BuildSchemaOperationWorkflow — the - ``/v2/extract/build-schema`` response body (VTRA ``BuildSchemaResponse``). - - ``extraction_schema`` is the generated JSON Schema serialized as a STRING - (VTRA parity — the v1 field is a string, not an object). - """ - - extraction_schema: str = Field( - ..., - description='The generated JSON schema as a string.', - title='Extraction Schema', - ) - metadata: V2BuildSchemaMetadata = Field( - ..., description='The metadata for the schema generation process.' - ) - - -class V2ExtractBuildSchemaJobsJobIdGetResponse(BaseModel): - completed_at: Optional[str] = Field( - None, description='Present once the job is terminal.' - ) - created_at: Optional[str] = None - error: Optional[Error] = Field( - None, description='Present once status is ``failed``.' - ) - job_id: Optional[str] = Field( - None, - description='The unique identifier for this v2-build-schema job. Format: ``extract-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', - ) - progress: Optional[float] = Field( - None, - description='Job completion as a decimal from 0 (not started) to 1 (complete). Present while ``processing``.', - ge=0.0, - le=1.0, - ) - result: Optional[Result] = Field( - None, description='Present once status is ``completed``.' - ) - status: Optional[Status1] = None - - -class V2ExtractJobsGetParametersQuery(BaseModel): - page: Optional[int] = Field( - 0, description='Page number (0-indexed).', ge=0, title='Page' - ) - page_size: Optional[int] = Field( - 10, description='Number of items per page.', ge=1, le=100, title='Page Size' - ) - status: Optional[str] = Field( - None, description='Filter by job status.', title='Status' - ) - - -class Job1(BaseModel): - completed_at: Optional[str] = None - created_at: Optional[str] = None - failure_reason: Optional[str] = None - job_id: Optional[str] = Field( - None, - description='The unique identifier for this v2-extract job. Format: ``extract-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', - ) - model_version: Optional[str] = None - status: Optional[Status1] = None - - -class V2ExtractJobsGetResponse(BaseModel): - has_more: Optional[bool] = None - jobs: Optional[list[Job1]] = None - page: Optional[int] = None - page_size: Optional[int] = None - - class V2ExtractJobsPostRequest(BaseModel): """ Input to V2ExtractOperationWorkflow. @@ -931,7 +698,18 @@ class V2ExtractJobsPostResponse(BaseModel): status: Optional[Status1] = None -class Result1(BaseModel): +class Error(BaseModel): + """ + Present once status is ``failed``. + """ + + code: Optional[str] = Field( + None, description='Stable error code (``internal_error`` when unmapped).' + ) + message: Optional[str] = None + + +class Result(BaseModel): """ Result returned by V2ExtractOperationWorkflow — the ``/v2/extract`` response body (``docs/extract-v2-proposal.md`` → Response). @@ -980,17 +758,21 @@ class V2ExtractJobsJobIdGetResponse(BaseModel): None, description='The unique identifier for this v2-extract job. Format: ``extract-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', ) + metadata: Optional[dict[str, Any]] = Field( + None, + description="The result's metadata block (billing included), present alongside ``output_url`` once a job with ``output_save_url`` has ``completed`` — the delivery moves the content, not the receipt. Same shape as the inline ``result``'s ``metadata``; inline jobs carry it there instead.", + ) output_url: Optional[str] = Field( None, description='The URL the result was delivered to. Present once the job has ``completed`` and ``output_save_url`` was set, instead of inline ``result``.', ) progress: Optional[float] = Field( None, - description='Job completion as a decimal from 0 (not started) to 1 (complete). Present while ``processing``.', + description='Estimated completion as a decimal from 0 to 1 — an estimate, not a measurement: it typically advances between polls while the job is ``processing``, may jump forward when the service reports a real milestone (e.g. parsed pages), and approaches but never reaches 1 (long-running jobs plateau near 0.98 — completion is signaled by ``status``, and a job may complete from any progress value). Present while ``processing``.', ge=0.0, le=1.0, ) - result: Optional[Result1] = Field( + result: Optional[Result] = Field( None, description='Present once status is ``completed`` and ``output_save_url`` was not set. When ``output_save_url`` was set, the result is delivered there and ``output_url`` is returned instead.', ) @@ -1106,7 +888,7 @@ class V2ParseJobsGetParametersQuery(BaseModel): ) -class Status7(Enum): +class Status4(Enum): """ The job's current status: ``pending``, ``processing``, ``completed``, or ``failed``. """ @@ -1117,7 +899,7 @@ class Status7(Enum): failed = 'failed' -class Job2(BaseModel): +class Job1(BaseModel): completed_at: Optional[str] = Field( None, description='ISO-8601 timestamp for when the job finished, if terminal.' ) @@ -1135,7 +917,7 @@ class Job2(BaseModel): model_version: Optional[str] = Field( None, description='The model snapshot used to parse the document.' ) - status: Optional[Status7] = Field( + status: Optional[Status4] = Field( None, description="The job's current status: ``pending``, ``processing``, ``completed``, or ``failed``.", ) @@ -1146,14 +928,14 @@ class V2ParseJobsGetResponse(BaseModel): None, description='Whether more jobs exist beyond this page; request the next ``page`` to fetch them.', ) - jobs: Optional[list[Job2]] = Field( + jobs: Optional[list[Job1]] = Field( None, description="The caller's parse jobs for this page, newest first." ) page: Optional[int] = Field(None, description='The 0-indexed page number.') page_size: Optional[int] = Field(None, description='Items per page.') -class ServiceTier6(Enum): +class ServiceTier4(Enum): """ Async service tier (``POST /jobs`` only). ``priority`` runs in the fast lane at the sync billing rate; absent → ``standard``. """ @@ -1162,7 +944,7 @@ class ServiceTier6(Enum): priority = 'priority' -class Status8(Enum): +class Status5(Enum): """ The job's status at creation — normally ``pending`` (a just-created job that is still running is reported as ``pending``), but may already be a terminal ``completed`` / ``failed`` if the job finished before the create response was rendered. """ @@ -1181,13 +963,13 @@ class V2ParseJobsPostResponse(BaseModel): ..., description='The unique identifier for the created parse job. Poll ``GET /v2/parse/jobs/{job_id}`` for its status and result. Format: ``-<26-character Crockford base32 ULID>`` matching ``^(parse|extract)-[0-9a-hjkmnp-tv-z]{26}$``. Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', ) - status: Status8 = Field( + status: Status5 = Field( ..., description="The job's status at creation — normally ``pending`` (a just-created job that is still running is reported as ``pending``), but may already be a terminal ``completed`` / ``failed`` if the job finished before the create response was rendered.", ) -class Error2(BaseModel): +class Error1(BaseModel): """ Present once the job has ``failed`` — the failure code + message. """ @@ -1196,7 +978,7 @@ class Error2(BaseModel): message: Optional[str] = None -class Status9(Enum): +class Status6(Enum): """ The job's current status: ``pending``, ``processing``, ``completed``, or ``failed``. """ @@ -1255,14 +1037,14 @@ class V2WorkflowJobsGetParametersQuery(BaseModel): ) -class Status10(Enum): +class Status7(Enum): pending = 'pending' processing = 'processing' completed = 'completed' failed = 'failed' -class Job3(BaseModel): +class Job2(BaseModel): completed_at: Optional[str] = None created_at: Optional[str] = None failure_reason: Optional[str] = None @@ -1271,17 +1053,17 @@ class Job3(BaseModel): description='The unique identifier for this v2-workflow job. Format: ``v2-workflow-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', ) model_version: Optional[str] = None - status: Optional[Status10] = None + status: Optional[Status7] = None class V2WorkflowJobsGetResponse(BaseModel): has_more: Optional[bool] = None - jobs: Optional[list[Job3]] = None + jobs: Optional[list[Job2]] = None page: Optional[int] = None page_size: Optional[int] = None -class ServiceTier7(Enum): +class ServiceTier5(Enum): """ Async service tier. ``priority`` runs in the fast lane at the sync billing rate; absent → ``standard``. """ @@ -1296,10 +1078,10 @@ class V2WorkflowJobsPostResponse(BaseModel): None, description='The unique identifier for this v2-workflow job. Format: ``v2-workflow-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', ) - status: Optional[Status10] = None + status: Optional[Status7] = None -class Error3(BaseModel): +class Error2(BaseModel): """ Present once status is ``failed``. """ @@ -1310,7 +1092,7 @@ class Error3(BaseModel): message: Optional[str] = None -class Result2(BaseModel): +class Result1(BaseModel): """ Result returned by V2WorkflowOperationWorkflow. @@ -1351,7 +1133,7 @@ class V2WorkflowJobsJobIdGetResponse(BaseModel): None, description='Present once the job is terminal.' ) created_at: Optional[str] = None - error: Optional[Error3] = Field( + error: Optional[Error2] = Field( None, description='Present once status is ``failed``.' ) job_id: Optional[str] = Field( @@ -1360,14 +1142,14 @@ class V2WorkflowJobsJobIdGetResponse(BaseModel): ) progress: Optional[float] = Field( None, - description='Job completion as a decimal from 0 (not started) to 1 (complete). Present while ``processing``.', + description='Estimated completion as a decimal from 0 to 1 — an estimate, not a measurement: it typically advances between polls while the job is ``processing``, may jump forward when the service reports a real milestone (e.g. parsed pages), and approaches but never reaches 1 (long-running jobs plateau near 0.98 — completion is signaled by ``status``, and a job may complete from any progress value). Present while ``processing``.', ge=0.0, le=1.0, ) - result: Optional[Result2] = Field( + result: Optional[Result1] = Field( None, description='Present once status is ``completed``.' ) - status: Optional[Status10] = None + status: Optional[Status7] = None class BlocksOptions(BaseModel): @@ -1509,7 +1291,7 @@ class V2ParseJobsPostRequest(BaseModel): None, description='Public URL the full response is delivered to; the API response then carries ``output_url`` instead of inline data.', ) - service_tier: Optional[ServiceTier6] = Field( + service_tier: Optional[ServiceTier4] = Field( None, description='Async service tier (``POST /jobs`` only). ``priority`` runs in the fast lane at the sync billing rate; absent → ``standard``.', ) @@ -1659,7 +1441,7 @@ class V2WorkflowJobsPostRequest(BaseModel): ], title='Output', ) - service_tier: Optional[ServiceTier7] = Field( + service_tier: Optional[ServiceTier5] = Field( None, description='Async service tier. ``priority`` runs in the fast lane at the sync billing rate; absent → ``standard``.', ) @@ -1700,7 +1482,7 @@ class V2WorkflowJobsPostRequest1(BaseModel): ], title='Output', ) - service_tier: Optional[ServiceTier7] = Field( + service_tier: Optional[ServiceTier5] = Field( None, description='Async service tier. ``priority`` runs in the fast lane at the sync billing rate; absent → ``standard``.', ) @@ -1861,7 +1643,7 @@ class V2ParseJobsJobIdGetResponse(BaseModel): created_at: Optional[str] = Field( None, description='ISO-8601 timestamp for when the job was created.' ) - error: Optional[Error2] = Field( + error: Optional[Error1] = Field( None, description='Present once the job has ``failed`` — the failure code + message.', ) @@ -1869,6 +1651,10 @@ class V2ParseJobsJobIdGetResponse(BaseModel): None, description='The unique identifier for this parse job. Format: ``-<26-character Crockford base32 ULID>`` matching ``^(parse|extract)-[0-9a-hjkmnp-tv-z]{26}$``. Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.', ) + metadata: Optional[ParseMetadata] = Field( + None, + description='The parse metadata (billing included), present alongside ``output_url`` once a job with ``output_save_url`` has ``completed`` — the delivery moves the content, not the receipt. Inline jobs carry it inside ``result`` instead.', + ) output_url: Optional[str] = Field( None, description='The URL the result was delivered to. Present once the job has ``completed`` and ``output_save_url`` was set, instead of inline ``result``.', @@ -1881,7 +1667,7 @@ class V2ParseJobsJobIdGetResponse(BaseModel): None, description='The parse response, present once the job has ``completed`` and ``output_save_url`` was not set. When ``output_save_url`` was set, the result is delivered there and ``output_url`` is returned instead.', ) - status: Optional[Status9] = Field( + status: Optional[Status6] = Field( None, description="The job's current status: ``pending``, ``processing``, ``completed``, or ``failed``.", ) diff --git a/specs/v2-aide.json b/specs/v2-aide.json index da25f28..ac712a7 100644 --- a/specs/v2-aide.json +++ b/specs/v2-aide.json @@ -1724,6 +1724,13 @@ "description": "The unique identifier for this v2-extract job. Format: ``extract-<26-character Crockford base32 ULID>`` (``[0-9a-hjkmnp-tv-z]{26}`` tail). Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.", "type": "string" }, + "metadata": { + "description": "The result's metadata block (billing included), present alongside ``output_url`` once a job with ``output_save_url`` has ``completed`` — the delivery moves the content, not the receipt. Same shape as the inline ``result``'s ``metadata``; inline jobs carry it there instead.", + "type": [ + "object", + "null" + ] + }, "output_url": { "description": "The URL the result was delivered to. Present once the job has ``completed`` and ``output_save_url`` was set, instead of inline ``result``.", "type": [ @@ -1732,7 +1739,7 @@ ] }, "progress": { - "description": "Job completion as a decimal from 0 (not started) to 1 (complete). Present while ``processing``.", + "description": "Estimated completion as a decimal from 0 to 1 — an estimate, not a measurement: it typically advances between polls while the job is ``processing``, may jump forward when the service reports a real milestone (e.g. parsed pages), and approaches but never reaches 1 (long-running jobs plateau near 0.98 — completion is signaled by ``status``, and a job may complete from any progress value). Present while ``processing``.", "maximum": 1, "minimum": 0, "type": "number" @@ -2449,6 +2456,17 @@ "description": "The unique identifier for this parse job. Format: ``-<26-character Crockford base32 ULID>`` matching ``^(parse|extract)-[0-9a-hjkmnp-tv-z]{26}$``. Opaque, server-minted, and stable for the life of the job — the same id is returned on the sync response, the async 202, and every poll. Treat it as opaque; older id formats remain accepted indefinitely and are never re-issued.", "type": "string" }, + "metadata": { + "anyOf": [ + { + "$ref": "#/components/schemas/ParseMetadata" + }, + { + "type": "null" + } + ], + "description": "The parse metadata (billing included), present alongside ``output_url`` once a job with ``output_save_url`` has ``completed`` — the delivery moves the content, not the receipt. Inline jobs carry it inside ``result`` instead." + }, "output_url": { "description": "The URL the result was delivered to. Present once the job has ``completed`` and ``output_save_url`` was set, instead of inline ``result``.", "type": [ @@ -3108,7 +3126,7 @@ "type": "string" }, "progress": { - "description": "Job completion as a decimal from 0 (not started) to 1 (complete). Present while ``processing``.", + "description": "Estimated completion as a decimal from 0 to 1 — an estimate, not a measurement: it typically advances between polls while the job is ``processing``, may jump forward when the service reports a real milestone (e.g. parsed pages), and approaches but never reaches 1 (long-running jobs plateau near 0.98 — completion is signaled by ``status``, and a job may complete from any progress value). Present while ``processing``.", "maximum": 1, "minimum": 0, "type": "number" diff --git a/src/landingai_ade/resources/v2/_normalize.py b/src/landingai_ade/resources/v2/_normalize.py index 4f9bd0e..942f7dd 100644 --- a/src/landingai_ade/resources/v2/_normalize.py +++ b/src/landingai_ade/resources/v2/_normalize.py @@ -37,6 +37,15 @@ def _progress(value: object) -> Optional[float]: return None +def _metadata(value: object) -> Optional[Dict[str, object]]: + # The top-level `metadata` receipt rides back on a completed job only when + # `output_save_url` was set (the result was delivered out-of-band). Pass it + # through as a plain dict; the inline-result case leaves this None. + if isinstance(value, Mapping): + return dict(cast(Dict[str, object], value)) + return None + + def _status(raw: Mapping[str, Any]) -> JobStatus: value = raw.get("status") if value is None: @@ -79,6 +88,7 @@ def normalize_parse_job(raw: Mapping[str, Any]) -> Job: progress=_progress(raw.get("progress")), result=result, error=error, + metadata=_metadata(raw.get("metadata")), raw=dict(raw), ) @@ -106,6 +116,7 @@ def normalize_extract_job(raw: Mapping[str, Any]) -> Job: progress=_progress(raw.get("progress")), result=result, error=error, + metadata=_metadata(raw.get("metadata")), raw=dict(raw), ) diff --git a/src/landingai_ade/types/v2/job.py b/src/landingai_ade/types/v2/job.py index 1f98631..864de98 100644 --- a/src/landingai_ade/types/v2/job.py +++ b/src/landingai_ade/types/v2/job.py @@ -35,6 +35,11 @@ class Job(BaseModel): # Populated on completion: V2ParseResponse for parse jobs, V2ExtractResult for extract jobs. result: Optional[object] = None error: Optional[JobError] = None + # The result's metadata block (billing included), surfaced top-level only when + # `output_save_url` was set: the content was delivered to `output_url` instead of + # inline, but the metadata receipt still rides back on the job. Same shape as the + # inline result's `metadata`; for inline jobs it lives on `result.metadata` instead. + metadata: Optional[Dict[str, object]] = None # Full original envelope for fields not surfaced above (org_id, output_url, version, ...). raw: Dict[str, object] = Field(default_factory=dict) diff --git a/tests/contract/test_v2_smoke.py b/tests/contract/test_v2_smoke.py index 595e37b..3aaa58e 100644 --- a/tests/contract/test_v2_smoke.py +++ b/tests/contract/test_v2_smoke.py @@ -54,6 +54,10 @@ def test_extract_jobs(staging_client: LandingAIADE) -> None: done = staging_client.v2.extract_jobs.wait(job.job_id, timeout=300) assert done.status is JobStatus.COMPLETED assert isinstance(done.result, V2ExtractResult) + # This inline job carries its metadata on `result.metadata`; the top-level + # `Job.metadata` receipt is only populated for `output_save_url` deliveries. + assert done.metadata is None + assert done.result.metadata.model_version def test_parse_sync(staging_client: LandingAIADE) -> None: @@ -105,3 +109,7 @@ def test_parse_jobs(staging_client: LandingAIADE) -> None: assert isinstance(done.result, V2ParseResponse) assert isinstance(done.result.markdown, str) assert done.result.markdown + # Inline delivery: the metadata rides on `result.metadata`, so the top-level + # `Job.metadata` receipt (set only for `output_save_url` deliveries) is absent. + assert done.metadata is None + assert done.result.metadata is not None diff --git a/tests/test_v2_normalize.py b/tests/test_v2_normalize.py index aed1ba6..55b06b3 100644 --- a/tests/test_v2_normalize.py +++ b/tests/test_v2_normalize.py @@ -115,6 +115,52 @@ def test_normalize_extract_job_error_object() -> None: assert job.error is not None and job.error.code == "internal_error" +def test_normalize_parse_job_output_save_url_surfaces_metadata() -> None: + # When `output_save_url` was set, the completed job delivers the result to + # `output_url` (no inline `result`) but still rides the metadata receipt back + # top-level, where it's surfaced on `Job.metadata`. + raw: Dict[str, Any] = { + "job_id": "parse-osu", + "status": "completed", + "output_url": "https://example.com/out.json", + "metadata": {"job_id": "parse-osu", "page_count": 3, "billing": {"total_credits": 2.0}}, + } + job = normalize_parse_job(raw) + assert job.status is JobStatus.COMPLETED + assert job.result is None # delivered out-of-band, not inline + assert job.metadata is not None + assert job.metadata["page_count"] == 3 + assert job.raw["output_url"] == "https://example.com/out.json" + + +def test_normalize_extract_job_output_save_url_surfaces_metadata() -> None: + raw: Dict[str, Any] = { + "job_id": "extract-osu", + "status": "completed", + "output_url": "https://example.com/out.json", + "metadata": {"job_id": "extract-osu", "duration_ms": 42, "billing": {"total_credits": 1.0}}, + } + job = normalize_extract_job(raw) + assert job.status is JobStatus.COMPLETED + assert job.result is None + assert job.metadata is not None + assert job.metadata["duration_ms"] == 42 + + +def test_normalize_job_inline_result_leaves_metadata_none() -> None: + # Inline jobs carry metadata inside `result.metadata`; the top-level + # `Job.metadata` receipt is only populated for `output_save_url` deliveries. + raw: Dict[str, Any] = { + "job_id": "parse-inline", + "status": "completed", + "result": {"markdown": "# hi", "metadata": {"job_id": "parse-inline", "page_count": 1}}, + } + job = normalize_parse_job(raw) + assert job.metadata is None + assert isinstance(job.result, V2ParseResponse) + assert job.result.metadata is not None and job.result.metadata.page_count == 1 + + def test_normalize_parse_job_minimal_create_envelope_defaults_to_pending() -> None: # Live /v2/parse/jobs create (202) response is minimal: only job_id, no status. raw = {"job_id": "parse-api-x"} diff --git a/tests/test_v2_types.py b/tests/test_v2_types.py index 3656c53..76c3534 100644 --- a/tests/test_v2_types.py +++ b/tests/test_v2_types.py @@ -46,6 +46,20 @@ def test_job_holds_typed_result_and_error() -> None: assert job.raw["org_id"] == "o1" +def test_job_metadata_defaults_none_and_holds_receipt() -> None: + # `metadata` defaults to None (inline jobs carry it on `result.metadata`) and, + # for `output_save_url` deliveries, holds the top-level metadata receipt. + inline = Job(job_id="j1", status=JobStatus.COMPLETED) + assert inline.metadata is None + delivered = Job( + job_id="j2", + status=JobStatus.COMPLETED, + metadata={"job_id": "j2", "page_count": 2, "billing": {"total_credits": 1.5}}, + raw={"output_url": "https://example.com/out.json"}, + ) + assert delivered.metadata is not None and delivered.metadata["page_count"] == 2 + + def test_job_raw_default_is_independent_per_instance() -> None: job_a = Job(job_id="j1", status=JobStatus.PENDING) job_b = Job(job_id="j2", status=JobStatus.PENDING)