From 37413345b47ef7c5e230617796a4b4e98b6fbce8 Mon Sep 17 00:00:00 2001 From: Chris Bartholomew Date: Wed, 23 Sep 2026 08:55:05 -0400 Subject: [PATCH] docs: metadata schemas need a top-level "document" wrapper MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The metadata examples in README.md and nodejs-api/README.md show a bare field-to-type map as the `schema` value. That shape can never work: the backend reads a top-level `document` key off the schema before looking at its contents, and throws java.util.concurrent.CompletionException: java.lang.IllegalArgumentException: Document schema is missing if it is absent. Anyone copying the published example hits this immediately, which is how it was reported. Verified against the production endpoint (api.vectorize.io), same request shape the SDK builds, inferSchema=false: bare formal JSON Schema -> FAILS, "Document schema is missing" bare field-to-type map (old doc) -> FAILS, "Document schema is missing" {"document": formal JSON Schema} -> SUCCEEDS, metadata extracted {"document": field-to-type map} -> SUCCEEDS, metadata extracted So the wrapper is the only thing that was missing — both schema representations work once wrapped. The examples now use a real JSON Schema anyway, because declared types are honored end to end: the invoice example in this commit returns total_amount as the number 1899.5 rather than a string. The exact Python block added here was executed against production before committing, and returned: {"date":"2026-02-04","invoice_number":"INV-77120", "total_amount":1899.5,"vendor_name":"ACME Supplies Ltd."} Also drops the "OpenAPI spec format" wording, which described neither the old example nor the actual requirement, and documents infer_metadata_schema / inferMetadataSchema plus the optional `sections` key. --- README.md | 41 +++++++++++++++++++++++++++++++++++------ nodejs-api/README.md | 18 ++++++++++++++---- 2 files changed, 49 insertions(+), 10 deletions(-) diff --git a/README.md b/README.md index 9491125..72a3cc1 100644 --- a/README.md +++ b/README.md @@ -117,22 +117,51 @@ Split documents into semantic chunks perfect for RAG pipelines: - Preserves context across chunks ### Metadata Extraction -Extract structured data using JSON schemas (OpenAPI spec format recommended): +Extract structured data using JSON Schema. + +Each `schema` must be an object with a top-level `document` key wrapping the JSON +Schema. Without that wrapper the extraction fails with +`IllegalArgumentException: Document schema is missing`. + ```python result = extract_text_from_file('invoice.pdf', options=ExtractionOptions( metadata_schemas=[{ 'id': 'invoice-data', 'schema': { - 'invoice_number': 'string', - 'date': 'string', - 'total_amount': 'number', - 'vendor_name': 'string' + 'document': { + 'type': 'object', + 'properties': { + 'invoice_number': {'type': 'string', 'description': 'Invoice reference number'}, + 'date': {'type': 'string', 'description': 'Invoice date as printed'}, + 'total_amount': {'type': 'number', 'description': 'Total amount due'}, + 'vendor_name': {'type': 'string', 'description': 'Name of the vendor'} + }, + 'required': [] + } } - }] + }], + infer_metadata_schema=False )) # Returns structured JSON metadata ``` +Give every property a `type` and a `description` — both materially improve +extraction accuracy. + +Set `infer_metadata_schema=False` when you supply your own schemas, so no schema +is generated and yours are used as-is. Leave it `True` (the default) to have a +schema inferred from the document instead. + +You may optionally add a sibling `sections` key for per-chunk metadata; omit it +for document-level extraction only: + +```python +'schema': { + 'document': { ... }, + 'sections': {'line-items': { ... }} +} +``` + ### Parsing Instructions Guide the extraction with custom instructions: ```python diff --git a/nodejs-api/README.md b/nodejs-api/README.md index 6afd6c4..d9b0822 100644 --- a/nodejs-api/README.md +++ b/nodejs-api/README.md @@ -305,18 +305,28 @@ import type { MetadataExtractionStrategySchema } from '@vectorize-io/iris'; -// Type-safe options with structured schema (OpenAPI spec format) +// Type-safe options with a JSON Schema. +// NOTE: `schema` must wrap the JSON Schema in a top-level `document` key. +// Without it the extraction fails with +// "IllegalArgumentException: Document schema is missing". const options: ExtractionOptions = { chunkSize: 512, parsingInstructions: 'Extract code blocks', metadataSchemas: [{ id: 'doc-meta', schema: { - title: 'string', - author: 'string', - date: 'string' + document: { + type: 'object', + properties: { + title: { type: 'string', description: 'Document title' }, + author: { type: 'string', description: 'Author name' }, + date: { type: 'string', description: 'Publication date as printed' } + }, + required: [] + } } }], + inferMetadataSchema: false, pollInterval: 2000, timeout: 300000 };