Compare commits

..

22 Commits

Author SHA1 Message Date
Neeraj Pradhan 6f66304985 Force github reparse of the workflow 2026-01-23 11:35:42 -08:00
Neeraj Pradhan b174fa8fab Run hourly extract tests to catch SDK schema drifts (#1089)
* Run hourly extract tests to catch SDK schema drifts

* fix url

* fix prod/staging env
2026-01-22 18:18:45 -08:00
github-actions[bot] b12ffef916 chore: version packages (#1087)
Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
2026-01-21 12:44:43 -08:00
Neeraj Pradhan 07ec282257 Bump up patch version for python packages (#1086) 2026-01-21 12:30:23 -08:00
Neeraj Pradhan 013b689812 Bump up minor version for python packages (#1085) 2026-01-21 12:13:13 -08:00
Adrian Lyjak 3040951cb8 Use error description in invalid extraction error (#1081)
* fix: display extraction job error in InvalidExtractionData exception

Refactored InvalidExtractionData to read the `error` field from
ExtractRun and prominently display it in the exception message.
The job-level error is now stored in the `extraction_error` attribute
and included in the invalid_item's metadata as `job_error`.

* Create three-yaks-beg.md

---------

Co-authored-by: Claude <noreply@anthropic.com>
2026-01-18 17:43:21 -05:00
github-actions[bot] 9239498945 chore: version packages (#1076)
Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
2026-01-14 19:15:05 +01:00
Pierre-Loic Doulcet 19cbb25631 remove extension filter (#1075)
* remove extension filter

* changeset

* Update ninety-goats-look.md

Make it a patch version

* Update package.json

back out of version bump

* Update pyproject.toml

back out of version bump

* Update package.json

back out of version bump

* Update pyproject.toml

back out of version bump

---------

Co-authored-by: Adrian Lyjak <adrianlyjak@gmail.com>
2026-01-14 19:13:39 +01:00
github-actions[bot] 812e2f7d72 chore: version packages (#1073)
Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
2026-01-12 19:03:13 +01:00
Clelia (Astra) Bertelli d7864afe3f fix: bug fix retry logic in Classify and Extract (#1066)
* fix: bug fix retry logic in Classify and Extract

* chore: apply suggestion

* chore: add PARTIAL_SUCCESS to classify
2026-01-12 18:57:40 +01:00
github-actions[bot] ade8d027a5 chore: version packages (#1071)
Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
2026-01-09 20:29:00 -05:00
Adrian Lyjak 997bcc8531 forgot ts changeset (#1070) 2026-01-09 20:23:29 -05:00
github-actions[bot] 8be554c234 chore: version packages (#1068)
Co-authored-by: github-actions[bot] <41898282+github-actions[bot]@users.noreply.github.com>
2026-01-09 18:56:51 -05:00
Adrian Lyjak f777cab0c5 Add bounding box type support to TS too (#1069)
ts too
2026-01-09 18:55:16 -05:00
Adrian Lyjak b9b83c953d Parse bounding boxes from extract jobs results in agent data (#1067) 2026-01-09 18:47:57 -05:00
github-actions[bot] 3ec7024626 chore: version packages (#1058) 2025-12-10 11:53:30 -06:00
Logan d5b18a03fa Remove generate from build path to fix publishing (#1057) 2025-12-10 11:52:43 -06:00
Clelia (Astra) Bertelli 18dd04b6de docs: correct links in readme (#1056) 2025-12-10 17:08:58 +01:00
github-actions[bot] 685a5e6ccc chore: version packages (#1054) 2025-12-09 15:30:13 -06:00
Jim Geurts 576c3d9076 feat: support zod v4 & v3 (#1052) 2025-12-09 15:29:23 -06:00
Logan c8321d2bc5 improve parse ts polling (#1053) 2025-12-09 15:21:19 -06:00
Tuana Çelik 131bbed7aa batch parse sctript with asyncio (#1051)
* batch parse sctript with asyncio

* lint

---------

Co-authored-by: Logan Markewich <logan.markewich@live.com>
2025-12-08 18:50:11 +01:00
34 changed files with 761 additions and 3442 deletions
+160
View File
@@ -0,0 +1,160 @@
name: Hourly Extract E2E Tests
on:
schedule:
- cron: "0 * * * *" # Runs every hour at the top of the hour
workflow_dispatch:
# Allows manual triggering
inputs:
environment:
description: "Environment to run the tests in"
required: false
default: staging
type: choice
options:
- staging
- production
notify_slack:
description: "Notify Slack"
required: false
default: false
type: boolean
workflow_call:
env:
UV_VERSION: "0.7.20"
PYTHON_VERSION: "3.12"
SLACK_CHANNEL_ID: C078PHNTF44 # Extract channel ID
API_E2E_LOG_PATH: ${{ github.workspace }}/extract-e2e.log
jobs:
extract-e2e:
name: "Hourly Extract E2E Tests (${{ matrix.environment }})"
runs-on: ubuntu-latest
timeout-minutes: 30
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}-${{ matrix.environment }}
cancel-in-progress: true
strategy:
fail-fast: false
matrix:
environment: ${{ github.event_name == 'schedule' && fromJson('["staging", "production"]') || fromJson(format('["{0}"]', github.event.inputs.environment || 'staging')) }}
steps:
- name: Set runtime inputs
id: runtime
run: |
environment=${{ matrix.environment }}
notify_slack=${{ github.event.inputs.notify_slack || github.event_name == 'schedule' }}
echo "environment=${environment}" >> $GITHUB_OUTPUT
echo "notify_slack=${notify_slack}" >> $GITHUB_OUTPUT
if [ "${environment}" = "production" ]; then
echo "LLAMA_CLOUD_BASE_URL=https://api.cloud.llamaindex.ai" >> $GITHUB_ENV
api_key_secret="${{ secrets.LLAMA_CLOUD_API_KEY }}"
project_id_secret="${{ secrets.LLAMA_CLOUD_PROJECT_ID }}"
else
echo "LLAMA_CLOUD_BASE_URL=https://api.staging.llamaindex.ai" >> $GITHUB_ENV
api_key_secret="${{ secrets.LLAMA_CLOUD_API_KEY_STAGING }}"
project_id_secret="${{ secrets.LLAMA_CLOUD_PROJECT_ID_STAGING }}"
fi
if [ -n "$api_key_secret" ]; then
echo "LLAMA_CLOUD_API_KEY=$api_key_secret" >> $GITHUB_ENV
fi
if [ -n "$project_id_secret" ]; then
echo "LLAMA_CLOUD_PROJECT_ID=$project_id_secret" >> $GITHUB_ENV
fi
- uses: actions/checkout@v5
with:
fetch-depth: 0
- name: Install uv
uses: astral-sh/setup-uv@v7
with:
version: ${{ env.UV_VERSION }}
- name: Set up Python
run: uv python install ${{ env.PYTHON_VERSION }} && uv python pin ${{ env.PYTHON_VERSION }}
- name: Run Extract E2E tests
id: extract-tests
continue-on-error: true
working-directory: py
run: |
set -o pipefail
rm -f "$API_E2E_LOG_PATH"
uv run pytest -v -n 8 --timeout=300 --session-timeout=1740 tests/extract/ 2>&1 | tee "$API_E2E_LOG_PATH"
- name: Extract pytest failure summary
id: failed-tests
if: steps.extract-tests.outcome == 'failure' || cancelled()
run: |
summary="$(python3 - <<'PY'
import os
from pathlib import Path
log_path = Path(os.environ["API_E2E_LOG_PATH"])
if not log_path.exists():
print("Test log not found.")
raise SystemExit(0)
lines = log_path.read_text(errors="ignore").splitlines()
def find_section(keyword):
for i, line in enumerate(lines):
if line.startswith("=") and keyword in line:
return i
return None
start = (
find_section("FAILURES")
or find_section("ERRORS")
or find_section("short test summary info")
)
if start is not None:
snippet = lines[start : start + 200]
else:
snippet = lines[-200:]
print("\n".join(snippet).strip())
PY
)"
if [ -z "$summary" ]; then
summary="Failed test summary not available. Review the full run logs."
fi
{
printf 'summary<<EOF\n%s\nEOF\n' "$summary"
} >> "$GITHUB_OUTPUT"
- name: Check test results
if: always()
run: |
if [ "${{ steps.extract-tests.outcome }}" == "failure" ]; then
echo "Extract E2E tests failed"
exit 1
fi
- name: Post to Extract Slack channel
id: slack
if: (failure() || cancelled()) && steps.runtime.outputs.notify_slack == 'true'
uses: slackapi/slack-github-action@v1.27.0
with:
channel-id: ${{ env.SLACK_CHANNEL_ID }}
slack-message: |
ALERT: *Hourly Extract E2E Tests Failure*
*Environment*: ${{ steps.runtime.outputs.environment }}
*Repository*: ${{ github.repository }}
*Branch*: ${{ github.ref_name }}
<${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }}|View Details>
Failed tests:
```
${{ steps.failed-tests.outputs.summary }}
```
env:
SLACK_BOT_TOKEN: ${{ secrets.SLACK_BOT_TOKEN }}
+1 -1
View File
@@ -398,6 +398,6 @@ Another option (orthogonal to the above) is to break the document into smaller s
## Additional Resources
- [Extract Documentation](https://docs.cloud.llamaindex.ai/llamaextract/getting_started) - Details on Extract features, API and examples.
- [Example Notebook](docs/examples-py/extract/resume_screening.ipynb) - Detailed walkthrough of resume parsing
- [Example Notebook](examples/extract/resume_screening.ipynb) - Detailed walkthrough of resume parsing
- [Example Application with TypeScript](./examples-ts/extract/) - End-to-end examples using LlamaExtract TypeScript client.
- [Discord Community](https://discord.com/invite/eN6D2HQ4aX) - Get help and share feedback
+5 -5
View File
@@ -97,7 +97,7 @@ for page in result.pages:
print(page.structuredData)
```
See more details about the result object in the [example notebook](./docs/examples-py/parse/demo_json_tour.ipynb).
See more details about the result object in the [example notebook](./examples/parse/demo_json_tour.ipynb).
### Using with file object / bytes
@@ -153,10 +153,10 @@ Full documentation for `SimpleDirectoryReader` can be found on the [LlamaIndex D
Several end-to-end indexing examples can be found in the examples folder
- [Getting Started](docs/examples-py/parse/demo_basic.ipynb)
- [Advanced RAG Example](docs/examples-py/parse/demo_advanced.ipynb)
- [Raw API Usage](docs/examples-py/parse/demo_api.ipynb)
- [Result Object Tour](docs/examples-py/parse/demo_json_tour.ipynb)
- [Getting Started](examples/parse/demo_basic.ipynb)
- [Advanced RAG Example](examples/parse/demo_advanced.ipynb)
- [Raw API Usage](examples/parse/demo_api.ipynb)
- [Result Object Tour](examples/parse/demo_json_tour.ipynb)
## Documentation
+48 -3260
View File
File diff suppressed because it is too large Load Diff
+19
View File
@@ -1,5 +1,24 @@
# llama-cloud-services-py
## 0.6.91
### Patch Changes
- 07ec282: Bump up patch versions for python packages
- 3040951: Use error description in ExtractedData invalid extraction error
## 0.6.90
### Patch Changes
- 19cbb25: Remove extension filter
## 0.6.89
### Patch Changes
- b9b83c9: Parse bounding boxes from extract jobs results in agent data
## 0.6.88
### Patch Changes
@@ -11,6 +11,9 @@ from .schema import (
InvalidExtractionData,
ExtractedFieldMetadata,
ExtractedFieldMetaDataDict,
FieldCitation,
BoundingBox,
PageDimensions,
)
from .client import AsyncAgentDataClient
@@ -28,4 +31,7 @@ __all__ = [
"InvalidExtractionData",
"ExtractedFieldMetadata",
"ExtractedFieldMetaDataDict",
"FieldCitation",
"BoundingBox",
"PageDimensions",
]
@@ -174,6 +174,22 @@ class TypedAgentDataItems(BaseModel, Generic[AgentDataT]):
)
class BoundingBox(BaseModel):
"""Bounding box coordinates for a citation location on a page."""
x: float = Field(description="X coordinate of the bounding box origin")
y: float = Field(description="Y coordinate of the bounding box origin")
w: float = Field(description="Width of the bounding box")
h: float = Field(description="Height of the bounding box")
class PageDimensions(BaseModel):
"""Dimensions of a page in the source document."""
width: float = Field(description="Width of the page")
height: float = Field(description="Height of the page")
class FieldCitation(BaseModel):
page: Optional[int] = Field(
None, description="The page number that the field occurred on"
@@ -182,6 +198,14 @@ class FieldCitation(BaseModel):
None,
description="The original text this field's value was derived from",
)
bounding_boxes: Optional[List[BoundingBox]] = Field(
None,
description="Bounding boxes indicating where the citation appears on the page",
)
page_dimensions: Optional[PageDimensions] = Field(
None,
description="Dimensions of the page containing the citation",
)
class ExtractedFieldMetadata(BaseModel):
@@ -201,6 +225,10 @@ class ExtractedFieldMetadata(BaseModel):
None,
description="The confidence score for the field based on the extracted text only",
)
parsing_confidence: Optional[float] = Field(
None,
description="The confidence score for the field based on the parsing/OCR quality",
)
citation: Optional[List[FieldCitation]] = Field(
None,
description="The citation for the field, including page number and matching text",
@@ -447,26 +475,49 @@ class ExtractedData(BaseModel, Generic[ExtractedT]):
},
)
except ValidationError as e:
# Capture the job-level error from the extraction run if available
job_error = result.error
invalid_item = ExtractedData[Dict[str, Any]].create(
data=result.data or {},
status="error",
field_metadata=field_metadata,
metadata={"extraction_error": str(e), **(metadata or {})},
metadata={
"extraction_error": str(e),
**({"job_error": job_error} if job_error else {}),
**(metadata or {}),
},
file_id=file_id,
file_name=file_name,
file_hash=file_hash,
)
raise InvalidExtractionData(invalid_item) from e
raise InvalidExtractionData(invalid_item, extraction_error=job_error) from e
class InvalidExtractionData(Exception):
"""
Exception raised when the extracted data does not conform to the schema.
Attributes:
invalid_item: The ExtractedData instance containing the invalid data and metadata
extraction_error: The error message from the extraction job, if available
"""
def __init__(self, invalid_item: ExtractedData[Dict[str, Any]]):
def __init__(
self,
invalid_item: ExtractedData[Dict[str, Any]],
extraction_error: Optional[str] = None,
):
self.invalid_item = invalid_item
super().__init__("Not able to parse the extracted data, parsed invalid format")
self.extraction_error = extraction_error
# Build an informative error message
if extraction_error:
message = f"Extraction error: {extraction_error}"
else:
message = "Not able to parse the extracted data, parsed invalid format"
super().__init__(message)
def calculate_overall_confidence(
+4 -6
View File
@@ -3,7 +3,7 @@ from typing import BinaryIO
import os
from pathlib import Path
from llama_cloud.client import AsyncLlamaCloud
from llama_cloud.types import File, FileCreate
from llama_cloud.types import File
from typing import Optional
from llama_cloud_services.utils import SourceText, FileInput
@@ -73,11 +73,9 @@ class FileClient:
presigned_url = await self.client.files.generate_presigned_url(
project_id=self.project_id,
organization_id=self.organization_id,
request=FileCreate(
name=name,
external_file_id=external_file_id,
file_size=file_size,
),
name=name,
external_file_id=external_file_id,
file_size=file_size,
)
httpx_client = self.client._client_wrapper.httpx_client
upload_response = await httpx_client.put(
+19 -10
View File
@@ -21,7 +21,6 @@ from llama_cloud import (
PipelineCreateTransformConfig,
PipelineFileCreateCustomMetadataValue,
PipelineType,
ProjectCreate,
ManagedIngestionStatus,
CloudDocumentCreate,
CloudDocument,
@@ -507,14 +506,19 @@ class LlamaCloudIndex(BaseManagedIndex):
client = get_client(api_key, base_url, app_url, timeout)
if project_id is None:
# create project if it doesn't exist
project = client.projects.upsert_project(
# get project by name
projects = client.projects.list_projects(
organization_id=organization_id,
request=ProjectCreate(name=project_name),
project_name=project_name,
)
if not projects:
raise ValueError(
f"Project '{project_name}' not found. Please create it first in the LlamaCloud UI."
)
project = projects[0]
project_id = project.id
if verbose:
print(f"Created project {project_id} with name {project_name}")
print(f"Found project {project_id} with name {project_name}")
# create pipeline
pipeline_create = PipelineCreate(
@@ -563,15 +567,20 @@ class LlamaCloudIndex(BaseManagedIndex):
app_url = app_url or os.environ.get("LLAMA_CLOUD_APP_URL", DEFAULT_APP_URL)
aclient = get_aclient(api_key, base_url, app_url, timeout)
# create project if it doesn't exist
project = await aclient.projects.upsert_project(
organization_id=organization_id, request=ProjectCreate(name=project_name)
# get project by name
projects = await aclient.projects.list_projects(
organization_id=organization_id, project_name=project_name
)
if not projects:
raise ValueError(
f"Project '{project_name}' not found. Please create it first in the LlamaCloud UI."
)
project = projects[0]
if project.id is None:
raise ValueError(f"Failed to create/get project {project_name}")
raise ValueError(f"Failed to get project {project_name}")
if verbose:
print(f"Created project {project.id} with name {project.name}")
print(f"Found project {project.id} with name {project.name}")
# create pipeline
pipeline_create = PipelineCreate(
+3 -5
View File
@@ -751,11 +751,9 @@ class LlamaParse(BasePydanticReader):
file_path = str(file_input)
file_ext = os.path.splitext(file_path)[1].lower()
if file_ext not in SUPPORTED_FILE_TYPES:
raise Exception(
f"Currently, only the following file types are supported: {SUPPORTED_FILE_TYPES}\n"
f"Current file type: {file_ext}"
)
mime_type = mimetypes.guess_type(file_path)[0]
mime_type = "application/octet-stream"
else:
mime_type = mimetypes.guess_type(file_path)[0]
# Open the file here for the duration of the async context
# load data, set the mime type
fs = fs or get_default_fs()
+24
View File
@@ -1,5 +1,29 @@
# llama_parse
## 0.6.91
### Patch Changes
- 07ec282: Bump up patch versions for python packages
- Updated dependencies [07ec282]
- Updated dependencies [3040951]
- llama-cloud-services-py@0.6.91
## 0.6.90
### Patch Changes
- 19cbb25: Remove extension filter
- Updated dependencies [19cbb25]
- llama-cloud-services-py@0.6.90
## 0.6.89
### Patch Changes
- Updated dependencies [b9b83c9]
- llama-cloud-services-py@0.6.89
## 0.6.88
### Patch Changes
+3 -3
View File
@@ -146,9 +146,9 @@ Full documentation for `SimpleDirectoryReader` can be found on the [LlamaIndex D
Several end-to-end indexing examples can be found in the examples folder
- [Getting Started](/docs/examples-py/parse/demo_basic.ipynb)
- [Advanced RAG Example](/docs/examples-py/parse/demo_advanced.ipynb)
- [Raw API Usage](/docs/examples-py/parse/demo_api.ipynb)
- [Getting Started](../../examples/parse/demo_basic.ipynb)
- [Advanced RAG Example](../../examples/parse/demo_advanced.ipynb)
- [Raw API Usage](../../examples/parse/demo_api.ipynb)
## Documentation
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "llama_parse",
"version": "0.6.88",
"version": "0.6.91",
"description": "",
"main": "index.js",
"private": false,
+2 -2
View File
@@ -11,13 +11,13 @@ dev = [
[project]
name = "llama-parse"
version = "0.6.88"
version = "0.6.91"
description = "Parse files into RAG-Optimized formats."
authors = [{name = "Logan Markewich", email = "logan@llamaindex.ai"}]
requires-python = ">=3.9,<4.0"
readme = "README.md"
license = "MIT"
dependencies = ["llama-cloud-services>=0.6.88"]
dependencies = ["llama-cloud-services>=0.6.91"]
[project.scripts]
llama-parse = "llama_parse.cli.main:parse"
+1 -1
View File
@@ -1,6 +1,6 @@
{
"name": "llama-cloud-services-py",
"version": "0.6.88",
"version": "0.6.91",
"private": false,
"license": "MIT",
"scripts": {},
+2 -2
View File
@@ -23,7 +23,7 @@ dev = [
[project]
name = "llama-cloud-services"
version = "0.6.88"
version = "0.6.91"
description = "Tailored SDK clients for LlamaCloud services."
authors = [{name = "Logan Markewich", email = "logan@runllama.ai"}]
requires-python = ">=3.9,<4.0"
@@ -31,7 +31,7 @@ readme = "README.md"
license = "MIT"
dependencies = [
"llama-index-core>=0.12.0",
"llama-cloud==0.1.45",
"llama-cloud==0.1.46",
"pydantic>=2.8,!=2.10",
"click>=8.1.7,<9",
"python-dotenv>=1.0.1,<2",
+34 -2
View File
@@ -1,11 +1,43 @@
import os
from typing import List
from llama_cloud_services.extract import LlamaExtract
from typing import Any, Dict, List, Optional, Union
from llama_cloud.core.api_error import ApiError
from llama_cloud.types import ExtractConfig
from pydantic import BaseModel
from tenacity import (
retry,
retry_if_exception,
stop_after_attempt,
wait_exponential,
)
from llama_cloud_services.extract import ExtractionAgent, LlamaExtract
# Global storage for agents to cleanup
_TEST_AGENTS_TO_CLEANUP: List[str] = []
def _is_rate_limit_error(exception: BaseException) -> bool:
"""Check if the exception is a rate limit error (429)."""
return isinstance(exception, ApiError) and exception.status_code == 429
@retry(
retry=retry_if_exception(_is_rate_limit_error),
wait=wait_exponential(multiplier=1, min=1, max=30),
stop=stop_after_attempt(5),
reraise=True,
)
def create_agent_with_retry(
extractor: LlamaExtract,
name: str,
data_schema: Union[Dict[str, Any], type[BaseModel]],
config: Optional[ExtractConfig] = None,
) -> ExtractionAgent:
"""Create an extraction agent with retry logic for rate limiting."""
return extractor.create_agent(name=name, data_schema=data_schema, config=config)
def pytest_configure(config):
"""Register custom markers for extract tests."""
config.addinivalue_line("markers", "agent_name: custom agent name for test")
+2 -2
View File
@@ -6,7 +6,7 @@ from pydantic import BaseModel
from llama_cloud_services.extract import LlamaExtract, ExtractionAgent, SourceText
from llama_cloud.types import ExtractConfig, ExtractMode, ExtractRun
from tests.extract.util import load_test_dotenv
from .conftest import register_agent_for_cleanup
from .conftest import register_agent_for_cleanup, create_agent_with_retry
load_test_dotenv()
@@ -87,7 +87,7 @@ def test_agent(llama_extract, test_agent_name, test_schema_dict, request):
except Exception as e:
print(f"Warning: Failed to cleanup existing agent: {e}")
agent = llama_extract.create_agent(name=name, data_schema=schema)
agent = create_agent_with_retry(llama_extract, name=name, data_schema=schema)
# Add agent to cleanup list via conftest helper
register_agent_for_cleanup(agent.id)
+5 -3
View File
@@ -8,7 +8,7 @@ import uuid
from llama_cloud.types import ExtractConfig, ExtractMode
from deepdiff import DeepDiff
from tests.extract.util import json_subset_match_score, load_test_dotenv
from .conftest import register_agent_for_cleanup
from .conftest import register_agent_for_cleanup, create_agent_with_retry
load_test_dotenv()
@@ -117,8 +117,10 @@ def extraction_agent(test_case: ExtractionTestCase, extractor: LlamaExtract):
except Exception as e:
print(f"Warning: Failed to cleanup existing agent: {str(e)}")
# Create new agent
agent = extractor.create_agent(agent_name, schema, config=test_case.config)
# Create new agent with retry logic for rate limiting
agent = create_agent_with_retry(
extractor, name=agent_name, data_schema=schema, config=test_case.config
)
# Register agent for cleanup at the end of the test session
register_agent_for_cleanup(agent.id)
+8 -5
View File
@@ -8,7 +8,6 @@ from llama_cloud import (
AutoTransformConfig,
PipelineCreate,
PipelineFileCreate,
ProjectCreate,
CompositeRetrievalMode,
LlamaParseParameters,
ReRankConfig,
@@ -60,11 +59,15 @@ def local_figures_file() -> str:
def _setup_index_with_file(
client: LlamaCloud, index_name: str, remote_file: Tuple[str, str]
) -> LlamaCloudIndex:
# create project if it doesn't exist
project_create = ProjectCreate(name=project_name)
project = client.projects.upsert_project(
organization_id=organization_id, request=project_create
# get project by name
projects = client.projects.list_projects(
organization_id=organization_id, project_name=project_name
)
if not projects:
raise ValueError(
f"Project '{project_name}' not found. Please create it first in the LlamaCloud UI."
)
project = projects[0]
# create pipeline
pipeline_create = PipelineCreate(
@@ -11,10 +11,12 @@ from llama_cloud.types.aggregate_group import AggregateGroup
from pydantic import BaseModel, Field, ValidationError
from llama_cloud_services.beta.agent_data.schema import (
BoundingBox,
ExtractedData,
ExtractedFieldMetadata,
FieldCitation,
InvalidExtractionData,
PageDimensions,
TypedAgentData,
TypedAggregateGroup,
calculate_overall_confidence,
@@ -421,6 +423,7 @@ def create_extract_run(
},
data_schema: Dict[str, Any] = {},
file: File = create_file(),
error: Optional[str] = None,
) -> ExtractRun:
return ExtractRun.parse_obj(
{
@@ -437,6 +440,7 @@ def create_extract_run(
"status": "SUCCESS",
"project_id": str(uuid.uuid4()),
"from_ui": False,
"error": error,
}
)
@@ -542,6 +546,46 @@ def test_extracted_data_from_extraction_result_invalid_data():
assert invalid_data.field_metadata["name"].confidence == 0.9
assert invalid_data.overall_confidence == 0.9
# Verify default error message when no job error present
assert exc_info.value.extraction_error is None
assert "Not able to parse the extracted data" in str(exc_info.value)
def test_extracted_data_from_extraction_result_with_job_error():
"""Test ExtractedData.from_extraction_result with job-level error prominently displayed."""
job_error_message = "Failed to process document: unsupported file format"
# Create ExtractRun with both invalid data AND a job-level error
extract_run = create_extract_run(
data={
"missing_name": "Valid Name",
"age": "not_a_number",
}, # Invalid age, missing name
extraction_metadata={
"name": {"confidence": 0.9},
},
data_schema={},
file=create_file(id="error-file", name="bad_data.pdf"),
error=job_error_message,
)
# Should raise InvalidExtractionData with the job error prominently displayed
with pytest.raises(InvalidExtractionData) as exc_info:
ExtractedData.from_extraction_result(
extract_run, Person, metadata={"test": "metadata"}
)
# Verify the exception message prominently shows the job error
exception = exc_info.value
assert exception.extraction_error == job_error_message
assert f"Extraction error: {job_error_message}" == str(exception)
# Verify the invalid_item contains both errors in metadata
invalid_data = exception.invalid_item
assert invalid_data.metadata.get("job_error") == job_error_message
assert "extraction_error" in invalid_data.metadata # Validation error still present
assert "test" in invalid_data.metadata # Original metadata preserved
class Dimensions(BaseModel):
length: Optional[str] = Field(
@@ -663,3 +707,69 @@ def test_field_conflict_in_schema():
assert isinstance(
extracted["majority_opinion"]["reasoning"], ExtractedFieldMetadata
)
def test_parse_extracted_field_metadata_with_bounding_boxes():
"""Test parse_extracted_field_metadata with bounding boxes and page dimensions."""
raw_metadata = {
"document_type": {
"citation": [
{
"page": 1,
"matching_text": "FACTURE ORIGINALE",
"bounding_boxes": [{"x": 77.28, "y": 615.12, "w": 70.6, "h": 7.2}],
"page_dimensions": {"width": 222.24, "height": 736.56},
}
],
"parsing_confidence": 1.0,
"extraction_confidence": 0.7252506422636493,
"confidence": 0.7252506422636493,
},
"summary": {
"citation": [
{
"page": 1,
"matching_text": "FACTURE ORIGINALE",
"bounding_boxes": [{"x": 77.28, "y": 615.12, "w": 70.6, "h": 7.2}],
"page_dimensions": {"width": 222.24, "height": 736.56},
},
{
"page": 1,
"matching_text": "Café filtre assiette — $1.90",
"bounding_boxes": [
{"x": 10.56, "y": 172.83, "w": 171.85, "h": 497.01}
],
"page_dimensions": {"width": 222.24, "height": 736.56},
},
],
"parsing_confidence": 1.0,
"extraction_confidence": 0.5700013128334419,
"confidence": 0.5700013128334419,
},
}
result = parse_extracted_field_metadata(raw_metadata)
# Verify document_type citation with bounding boxes
assert isinstance(result["document_type"], ExtractedFieldMetadata)
assert result["document_type"].parsing_confidence == 1.0
assert result["document_type"].extraction_confidence == 0.7252506422636493
assert result["document_type"].confidence == 0.7252506422636493
assert len(result["document_type"].citation) == 1
citation = result["document_type"].citation[0]
assert citation.page == 1
assert citation.matching_text == "FACTURE ORIGINALE"
assert len(citation.bounding_boxes) == 1
assert citation.bounding_boxes[0] == BoundingBox(x=77.28, y=615.12, w=70.6, h=7.2)
assert citation.page_dimensions == PageDimensions(width=222.24, height=736.56)
# Verify summary citation with multiple bounding boxes
assert isinstance(result["summary"], ExtractedFieldMetadata)
assert len(result["summary"].citation) == 2
assert result["summary"].citation[0].bounding_boxes[0].x == 77.28
assert result["summary"].citation[1].bounding_boxes[0].x == 10.56
# Verify round-trip serialization
result2 = parse_extracted_field_metadata(result)
assert result2 == result
+17 -16
View File
@@ -34,9 +34,10 @@ TEST_PIPELINE = Pipeline(
def mock_client() -> MagicMock:
"""Mock client with sensible defaults."""
client = MagicMock()
client.projects.upsert_project.return_value = Project(
default_project = Project(
id="default-proj", name=DEFAULT_PROJECT_NAME, organization_id="default-org"
)
client.projects.list_projects.return_value = [default_project]
client.pipelines.upsert_pipeline.return_value = Pipeline(
id="default-pipe",
name="default",
@@ -100,8 +101,8 @@ def test_from_documents_uses_provided_project_id(mock_client: MagicMock) -> None
project_id=provided_project_id,
)
# Assert - project upsert not called; pipeline uses provided project_id
mock_client.projects.upsert_project.assert_not_called()
# Assert - project list not called (project_id provided); pipeline uses provided project_id
mock_client.projects.list_projects.assert_not_called()
assert mock_client.pipelines.upsert_pipeline.call_count == 1
assert (
mock_client.pipelines.upsert_pipeline.call_args.kwargs["project_id"]
@@ -110,29 +111,29 @@ def test_from_documents_uses_provided_project_id(mock_client: MagicMock) -> None
assert index.project.id == provided_project_id
def test_from_documents_upserts_project_when_project_id_missing(
def test_from_documents_lists_project_when_project_id_missing(
mock_client: MagicMock,
) -> None:
organization_id = "org-xyz"
index_name = "my_new_index"
# Project is created when project_id is not provided
upserted_project = Project(
# Project is found by name when project_id is not provided
found_project = Project(
id="proj-999", name=DEFAULT_PROJECT_NAME, organization_id=organization_id
)
mock_client.projects.upsert_project.return_value = upserted_project
mock_client.projects.list_projects.return_value = [found_project]
test_pipeline = Pipeline(
id="pipe-xyz",
name=index_name,
project_id=upserted_project.id,
project_id=found_project.id,
embedding_config=EMBEDDING_CONFIG,
)
with patch.object(
base,
"resolve_project_and_pipeline",
return_value=(upserted_project, test_pipeline),
return_value=(found_project, test_pipeline),
):
docs = [Document(text="world")]
index = LlamaCloudIndex.from_documents(
@@ -141,15 +142,15 @@ def test_from_documents_upserts_project_when_project_id_missing(
organization_id=organization_id,
)
# Assert - project was upserted with org id and default project name
mock_client.projects.upsert_project.assert_called_once()
kwargs = mock_client.projects.upsert_project.call_args.kwargs
# Assert - project was listed with org id and default project name
mock_client.projects.list_projects.assert_called_once()
kwargs = mock_client.projects.list_projects.call_args.kwargs
assert kwargs["organization_id"] == organization_id
assert kwargs["request"].name == DEFAULT_PROJECT_NAME
assert kwargs["project_name"] == DEFAULT_PROJECT_NAME
# Pipeline created under the upserted project id
# Pipeline created under the found project id
assert (
mock_client.pipelines.upsert_pipeline.call_args.kwargs["project_id"]
== upserted_project.id
== found_project.id
)
assert index.project.id == upserted_project.id
assert index.project.id == found_project.id
Generated
+6 -6
View File
@@ -1,5 +1,5 @@
version = 1
revision = 2
revision = 3
requires-python = ">=3.9, <4.0"
resolution-markers = [
"python_full_version >= '3.14'",
@@ -1595,21 +1595,21 @@ wheels = [
[[package]]
name = "llama-cloud"
version = "0.1.45"
version = "0.1.46"
source = { registry = "https://pypi.org/simple" }
dependencies = [
{ name = "certifi" },
{ name = "httpx" },
{ name = "pydantic" },
]
sdist = { url = "https://files.pythonhosted.org/packages/e0/b7/3a2a209f1c3fa516de172cb13e03f5a897adea5523f2ee0f544d035e3704/llama_cloud-0.1.45.tar.gz", hash = "sha256:140244008cc5710e31ae97c6043973a3a9969a51b0f38155fa33a8434078e8aa", size = 140968, upload-time = "2025-12-03T02:22:49.484Z" }
sdist = { url = "https://files.pythonhosted.org/packages/40/f3/f4d6520f8d546e6c5a02f6ebeed5c09774a074b8d2c24ad559ace97a56a6/llama_cloud-0.1.46.tar.gz", hash = "sha256:e86f8791c053590d70cc59e0fc13ce72f9b681a8e658bc61df86d0285288d8ee", size = 127752, upload-time = "2026-01-21T18:40:57.103Z" }
wheels = [
{ url = "https://files.pythonhosted.org/packages/62/1d/466b0df69b81ce9410ad6ec7229a1e6601ff69640f02f246e06cfcc7428c/llama_cloud-0.1.45-py3-none-any.whl", hash = "sha256:500299a6d3f25f97bcf6755d6338523023564fa8f376955c2cf299bbc9561cc2", size = 397184, upload-time = "2025-12-03T02:22:48.335Z" },
{ url = "https://files.pythonhosted.org/packages/c4/3a/6caaea28c8c804add33c91d356ed7d5a5412d6c9598e1450af95a15e0bcd/llama_cloud-0.1.46-py3-none-any.whl", hash = "sha256:6c6546c09c04a038c86d84d42f00eae8fd3bff49991ad3aab844bd866ecdf352", size = 361989, upload-time = "2026-01-21T18:40:54.863Z" },
]
[[package]]
name = "llama-cloud-services"
version = "0.6.85"
version = "0.6.90"
source = { editable = "." }
dependencies = [
{ name = "click", version = "8.1.8", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.10'" },
@@ -1649,7 +1649,7 @@ dev = [
requires-dist = [
{ name = "click", specifier = ">=8.1.7,<9" },
{ name = "eval-type-backport", marker = "python_full_version < '3.10'", specifier = ">=0.2.0,<0.3" },
{ name = "llama-cloud", specifier = "==0.1.45" },
{ name = "llama-cloud", specifier = "==0.1.46" },
{ name = "llama-index-core", specifier = ">=0.12.0" },
{ name = "packaging", specifier = ">=23.0" },
{ name = "platformdirs", specifier = ">=4.3.7,<5" },
+33
View File
@@ -1,5 +1,38 @@
# llama-cloud-services
## 0.5.3
### Patch Changes
- d7864af: bugfixes in retry logic for LlamaExtract and LlamaClassify
## 0.5.2
### Patch Changes
- 997bcc8: Add types for bounding boxes
## 0.5.1
### Patch Changes
- d5b18a0: Fix publishing
## 0.5.0
### Minor Changes
- 576c3d9: feat: support zod v4 & v3
Adds support for zod v4 while maintaining backward compatibility with v3.
- Updated zod peer dependency to accept both v3 and v4: `^3.25.76 || ^4.0.0`
- Migrated all import statements to use `zod/v4` import path for compatibility
### Patch Changes
- c8321d2: Improve parse results polling
- 576c3d9: Support zod v3 an v4
## 0.4.3
### Patch Changes
+9 -8
View File
@@ -1,12 +1,12 @@
{
"name": "llama-cloud-services",
"version": "0.4.3",
"version": "0.5.3",
"type": "module",
"license": "MIT",
"scripts": {
"get-openapi": "node ./scripts/get-openapi.js",
"generate": "./node_modules/.bin/openapi-ts",
"build": "pnpm run generate && bunchee",
"build": "bunchee",
"dev": "bunchee --watch",
"lint": "eslint src/ --ignore-pattern client/*.ts --no-warn-ignored",
"format": "prettier --write ./src/ tests/",
@@ -116,9 +116,9 @@
"@eslint/js": "^9.32.0",
"@hey-api/client-fetch": "^0.10.1",
"@hey-api/openapi-ts": "^0.67.5",
"@llamaindex/core": "^0.6.19",
"@llamaindex/core": "^0.6.22",
"@llamaindex/env": "^0.1.30",
"@llamaindex/workflow-core": "^0.4.1",
"@llamaindex/workflow-core": "^1.3.3",
"@types/node": "^20.19.9",
"@typescript-eslint/eslint-plugin": "^8.38.0",
"@typescript-eslint/parser": "^8.38.0",
@@ -131,18 +131,19 @@
"turbo": "^2.5.5",
"typescript": "^5.8.3",
"typescript-eslint": "^8.38.0",
"vitest": "^2.0.0"
"vitest": "^2.0.0",
"zod": "^4.1.13"
},
"peerDependencies": {
"@llamaindex/core": "^0.6.19",
"@llamaindex/env": "^0.1.30",
"@llamaindex/workflow-core": "^0.4.1"
"@llamaindex/workflow-core": "^1.3.3",
"zod": "^3.25.0 || ^4.0.0"
},
"dependencies": {
"ajv": "^8.17.1",
"file-type": "^21.0.0",
"p-retry": "^6.2.1",
"zod": "^3.25.76"
"p-retry": "^6.2.1"
},
"packageManager": "pnpm@10.8.1"
}
@@ -2,11 +2,14 @@ export { AgentClient, createAgentDataClient } from "./client";
export type {
AggregateAgentDataOptions,
BoundingBox,
ComparisonOperator,
ExtractedData,
ExtractedFieldMetadata,
ExtractedFieldMetadataDict,
FieldCitation,
FilterOperation,
PageDimensions,
SearchAgentDataOptions,
StatusType,
TypedAgentData,
@@ -28,6 +28,44 @@ export type ComparisonOperator =
*/
export type FilterOperation = RawFilterOperation;
/**
* Bounding box coordinates for a citation location on a page
*/
export interface BoundingBox {
/** X coordinate of the bounding box origin */
x: number;
/** Y coordinate of the bounding box origin */
y: number;
/** Width of the bounding box */
w: number;
/** Height of the bounding box */
h: number;
}
/**
* Dimensions of a page in the source document
*/
export interface PageDimensions {
/** Width of the page */
width: number;
/** Height of the page */
height: number;
}
/**
* Citation information for an extracted field
*/
export interface FieldCitation {
/** The page number that the field occurred on */
page?: number;
/** The original text this field's value was derived from */
matching_text?: string;
/** Bounding boxes indicating where the citation appears on the page */
bounding_boxes?: BoundingBox[];
/** Dimensions of the page containing the citation */
page_dimensions?: PageDimensions;
}
/**
* Metadata for an extracted field, including confidence and citation information
*/
@@ -38,16 +76,11 @@ export interface ExtractedFieldMetadata {
confidence?: number;
/** The confidence score for the field based on the extracted text only */
extraction_confidence?: number;
/** The confidence score for the field based on the parsing/OCR quality */
parsing_confidence?: number;
citation?: FieldCitation[];
}
export interface FieldCitation {
/** The page number that the field occurred on */
page?: number;
/** The original text this field's value was derived from */
matching_text?: string;
}
/**
* Dictionary mapping field names to their metadata
* Values can be ExtractedFieldMetadata objects, nested dictionaries, or arrays
+8 -9
View File
@@ -108,20 +108,19 @@ async function pollForJobCompletion({
}
const response =
await getClassifyJobApiV1ClassifierJobsClassifyJobIdGet(jobOptions);
if (!response.response.ok) {
numIterations++;
}
if (typeof response.data != "undefined") {
status = response.data.status as StatusEnum;
if (status == StatusEnum.CANCELLED || status == StatusEnum.ERROR) {
throw new Error("There was an error during the classification job.");
} else if (status == StatusEnum.SUCCESS) {
throw new Error("There was an error extracting data from your file.");
} else if (
status == StatusEnum.SUCCESS ||
status == StatusEnum.PARTIAL_SUCCESS
) {
return true;
} else {
numIterations++;
await sleep(interval * 1000);
}
}
numIterations++;
await sleep(interval * 1000);
}
}
@@ -169,7 +168,7 @@ async function getJobResult({
retries++;
await sleep(retryInterval * 1000);
}
if (typeof response.data != "undefined") {
if (response.response.ok && typeof response.data != "undefined") {
return response.data as ClassifyJobResults;
} else {
throw new Error(
@@ -1,6 +1,6 @@
// This file is auto-generated by @hey-api/openapi-ts
import { z } from "zod";
import { z } from "zod/v4";
export const zNoneSegmentationConfig = z.object({
mode: z.literal("none").optional().default("none"),
@@ -770,7 +770,7 @@ export const zMetadataFilter = z.object({
export const zFilterCondition: z.ZodTypeAny = z.enum(["and", "or", "not"]);
export const zMetadataFilters: z.AnyZodObject = z.object({
export const zMetadataFilters: z.ZodObject<z.ZodRawShape> = z.object({
filters: z.array(z.unknown()),
condition: z.union([zFilterCondition, z.null()]).optional(),
});
@@ -782,7 +782,7 @@ export const zRetrievalMode: z.ZodTypeAny = z.enum([
"auto_routed",
]);
export const zPresetRetrievalParams: z.AnyZodObject = z.object({
export const zPresetRetrievalParams: z.ZodObject<z.ZodRawShape> = z.object({
dense_similarity_top_k: z
.union([z.number().int().gte(1).lte(100), z.null()])
.optional(),
@@ -825,7 +825,7 @@ export const zSupportedLlmModelNames: z.ZodTypeAny = z.enum([
"VERTEX_AI_CLAUDE_3_5_SONNET_V2",
]);
export const zLlmParameters: z.AnyZodObject = z.object({
export const zLlmParameters: z.ZodObject<z.ZodRawShape> = z.object({
model_name: zSupportedLlmModelNames.optional(),
system_prompt: z.union([z.string().max(3000), z.null()]).optional(),
temperature: z.union([z.number(), z.null()]).optional(),
@@ -834,7 +834,7 @@ export const zLlmParameters: z.AnyZodObject = z.object({
class_name: z.string().optional().default("base_component"),
});
export const zChatData: z.AnyZodObject = z.object({
export const zChatData: z.ZodObject<z.ZodRawShape> = z.object({
retrieval_parameters: zPresetRetrievalParams.optional(),
llm_parameters: z.union([zLlmParameters, z.null()]).optional(),
class_name: z.string().optional().default("base_component"),
@@ -2141,7 +2141,7 @@ export const zTextItem = z.object({
value: z.string(),
});
export const zListItem: z.AnyZodObject = z.object({
export const zListItem: z.ZodObject<z.ZodRawShape> = z.object({
type: z.literal("list").optional().default("list"),
bBox: z.union([z.unknown(), z.null()]).optional(),
items: z.array(z.unknown()),
+1 -1
View File
@@ -1,6 +1,6 @@
import { workflowEvent } from "@llamaindex/workflow-core";
import { zodEvent } from "@llamaindex/workflow-core/util/zod";
import { z } from "zod";
import { z } from "zod/v4";
import { parseFormSchema } from "./schema";
export const uploadEvent = zodEvent(
+3 -7
View File
@@ -296,20 +296,16 @@ async function pollForJobCompletion(
return false;
}
const response = await getJobApiV1ExtractionJobsJobIdGet(jobOptions);
if (!response.response.ok) {
numIterations++;
}
if (typeof response.data != "undefined") {
status = response.data.status as StatusEnum;
if (status == StatusEnum.CANCELLED || status == StatusEnum.ERROR) {
throw new Error("There was an error extracting data from your file.");
} else if (status == StatusEnum.SUCCESS) {
return true;
} else {
numIterations++;
await sleep(interval * 1000);
}
}
numIterations++;
await sleep(interval * 1000);
}
}
@@ -350,7 +346,7 @@ async function getJobResult(
retries++;
await sleep(retryInterval * 1000);
}
if (typeof response.data != "undefined") {
if (response.response.ok && typeof response.data != "undefined") {
return {
data: response.data.data,
extractionMetadata: response.data.extraction_metadata,
+2 -4
View File
@@ -2,7 +2,7 @@ import type { JSONValue } from "@llamaindex/core/global";
import type { ToolMetadata } from "@llamaindex/core/llms";
import type { BaseQueryEngine } from "@llamaindex/core/query-engine";
import { tool } from "@llamaindex/core/tools";
import { z } from "zod";
import { z } from "zod/v4";
const DEFAULT_NAME = "llama_cloud_index_tool";
const DEFAULT_DESCRIPTION =
@@ -21,9 +21,7 @@ export function createQueryEngineTool(
name: metadata?.name ?? DEFAULT_NAME,
description: metadata?.description ?? DEFAULT_DESCRIPTION,
parameters: z.object({
query: z.string({
description: "The query to search for",
}),
query: z.string().describe("The query to search for"),
}),
execute: async ({ query }) => {
const response = await queryEngine.query({ query });
+113 -59
View File
@@ -1,9 +1,10 @@
/* eslint-disable @typescript-eslint/no-explicit-any */
import { type Client, createClient, createConfig } from "@hey-api/client-fetch";
import { type FailedAttemptError } from "p-retry";
import { Document, FileReader } from "@llamaindex/core/schema";
import { fs, getEnv, path } from "@llamaindex/env";
import {
type BodyUploadFileApiParsingUploadPost,
type BodyUploadFileApiV1ParsingUploadPost,
type FailPageMode,
type ParserLanguages,
type ParsingMode,
@@ -32,6 +33,33 @@ type WriteStream = {
// eslint-disable-next-line no-var
var process: any;
function handleFailedAttempt(
error: FailedAttemptError,
jobId: string,
verbose: boolean,
) {
// Retry only on 5XX or socket errors.
const status = (error.cause as any)?.response?.status;
if (
!(
(status && status >= 500 && status < 600) ||
((error.cause as any)?.code &&
((error.cause as any).code === "ECONNRESET" ||
(error.cause as any).code === "ETIMEDOUT" ||
(error.cause as any).code === "ECONNREFUSED")) ||
(status && status === 404)
)
) {
throw error;
}
if (verbose) {
console.warn(
`Attempting to get job ${jobId} result (attempt ${error.attemptNumber}) failed. Retrying...`,
);
}
}
/**
* Represents a reader for parsing files using the LlamaParse API.
* See https://github.com/run-llama/llama_parse
@@ -188,6 +216,16 @@ export class LlamaParseReader extends FileReader {
extract_printed_page_number?: boolean | undefined;
tier?: string | undefined;
version?: string | undefined;
layout_aware?: boolean | undefined;
line_level_bounding_box?: boolean | undefined;
specialized_image_parsing?: boolean | undefined;
aggressive_table_extraction?: boolean | undefined;
preserve_very_small_text?: boolean | undefined;
spreadsheet_force_formula_computation?: boolean | undefined;
inline_images_in_markdown?: boolean | undefined;
keep_page_separator_when_merging_tables?: boolean | undefined;
remove_hidden_text?: boolean | undefined;
presentation_out_of_bounds_content?: boolean | undefined;
constructor(
params: Partial<Omit<LlamaParseReader, "language" | "apiKey">> & {
@@ -387,11 +425,25 @@ export class LlamaParseReader extends FileReader {
extract_printed_page_number: this.extract_printed_page_number,
tier: this.tier,
version: this.version,
layout_aware: this.layout_aware,
line_level_bounding_box: this.line_level_bounding_box,
specialized_image_parsing: this.specialized_image_parsing,
aggressive_table_extraction: this.aggressive_table_extraction,
preserve_very_small_text: this.preserve_very_small_text,
spreadsheet_force_formula_computation:
this.spreadsheet_force_formula_computation,
inline_images_in_markdown: this.inline_images_in_markdown,
webhook_configurations: undefined,
keep_page_separator_when_merging_tables:
this.keep_page_separator_when_merging_tables,
remove_hidden_text: this.remove_hidden_text,
presentation_out_of_bounds_content:
this.presentation_out_of_bounds_content,
} satisfies {
[Key in keyof BodyUploadFileApiParsingUploadPost]-?:
| BodyUploadFileApiParsingUploadPost[Key]
[Key in keyof BodyUploadFileApiV1ParsingUploadPost]-?:
| BodyUploadFileApiV1ParsingUploadPost[Key]
| undefined;
} as unknown as BodyUploadFileApiParsingUploadPost;
} as unknown as BodyUploadFileApiV1ParsingUploadPost;
const response = await uploadFileApiV1ParsingUploadPost({
client: this.#client,
@@ -443,26 +495,8 @@ export class LlamaParseReader extends FileReader {
}),
{
retries: this.maxErrorCount,
onFailedAttempt: (error) => {
// Retry only on 5XX or socket errors.
const status = (error.cause as any)?.response?.status;
if (
!(
(status && status >= 500 && status < 600) ||
((error.cause as any)?.code &&
((error.cause as any).code === "ECONNRESET" ||
(error.cause as any).code === "ETIMEDOUT" ||
(error.cause as any).code === "ECONNREFUSED"))
)
) {
throw error;
}
if (this.verbose) {
console.warn(
`Attempting to get job ${jobId} result (attempt ${error.attemptNumber}) failed. Retrying...`,
);
}
},
onFailedAttempt: (error) =>
handleFailedAttempt(error, jobId, this.verbose),
},
);
} catch (e: any) {
@@ -475,49 +509,69 @@ export class LlamaParseReader extends FileReader {
const status = (data as Record<string, unknown>)["status"];
if (status === "SUCCESS") {
let resultData;
switch (resultType) {
case "json": {
resultData =
await getJobJsonResultApiV1ParsingJobJobIdResultJsonGet({
client: this.#client,
throwOnError: true,
path: { job_id: jobId },
query: {
organization_id: this.organization_id ?? null,
},
signal: AbortSignal.timeout(this.maxTimeout * 1000),
});
break;
const resultData = await pRetry(
() =>
getJobJsonResultApiV1ParsingJobJobIdResultJsonGet({
client: this.#client,
throwOnError: true,
path: { job_id: jobId },
query: {
organization_id: this.organization_id ?? null,
},
signal: AbortSignal.timeout(this.maxTimeout * 1000),
}),
{
retries: this.maxErrorCount,
onFailedAttempt: (error) =>
handleFailedAttempt(error, jobId, this.verbose),
},
);
return resultData.data;
}
case "markdown": {
resultData =
await getJobResultApiV1ParsingJobJobIdResultMarkdownGet({
client: this.#client,
throwOnError: true,
path: { job_id: jobId },
query: {
organization_id: this.organization_id ?? null,
},
signal: AbortSignal.timeout(this.maxTimeout * 1000),
});
break;
const resultData = await pRetry(
() =>
getJobResultApiV1ParsingJobJobIdResultMarkdownGet({
client: this.#client,
throwOnError: true,
path: { job_id: jobId },
query: {
organization_id: this.organization_id ?? null,
},
signal: AbortSignal.timeout(this.maxTimeout * 1000),
}),
{
retries: this.maxErrorCount,
onFailedAttempt: (error) =>
handleFailedAttempt(error, jobId, this.verbose),
},
);
return resultData.data;
}
case "text": {
resultData =
await getJobTextResultApiV1ParsingJobJobIdResultTextGet({
client: this.#client,
throwOnError: true,
path: { job_id: jobId },
query: {
organization_id: this.organization_id ?? null,
},
signal: AbortSignal.timeout(this.maxTimeout * 1000),
});
break;
const resultData = await pRetry(
() =>
getJobTextResultApiV1ParsingJobJobIdResultTextGet({
client: this.#client,
throwOnError: true,
path: { job_id: jobId },
query: {
organization_id: this.organization_id ?? null,
},
signal: AbortSignal.timeout(this.maxTimeout * 1000),
}),
{
retries: this.maxErrorCount,
onFailedAttempt: (error) =>
handleFailedAttempt(error, jobId, this.verbose),
},
);
return resultData.data;
}
}
return resultData.data;
} else if (status === "PENDING") {
if (this.verbose && tries % 10 === 0) {
this.stdout?.write(".");
+8 -7
View File
@@ -1,6 +1,6 @@
import { FailPageMode, ParserLanguages, ParsingMode } from "./client";
import { z } from "zod";
import { z } from "zod/v4";
type Language = ParserLanguages;
const VALUES: [Language, ...Language[]] = [
@@ -52,9 +52,10 @@ export const parseFormSchema = z.object({
html_remove_navigation_elements: z.boolean().optional(),
http_proxy: z
.string()
.url(
'Set a valid URL for the HTTP proxy, e.g., "http://proxy.example.com:8080"',
)
.url({
error:
'Set a valid URL for the HTTP proxy, e.g., "http://proxy.example.com:8080"',
})
.refine(
(url) => {
try {
@@ -67,7 +68,7 @@ export const parseFormSchema = z.object({
}
},
{
message: "Invalid HTTP proxy URL",
error: "Invalid HTTP proxy URL",
},
)
.optional(),
@@ -100,7 +101,7 @@ export const parseFormSchema = z.object({
vendor_multimodal_model_name: z.string().optional(),
model: z.string().optional(),
webhook_url: z.string().url().optional(),
parse_mode: z.nativeEnum(ParsingMode).nullable().optional(),
parse_mode: z.enum(ParsingMode).nullable().optional(),
system_prompt: z.string().optional(),
system_prompt_append: z.string().optional(),
user_prompt: z.string().optional(),
@@ -129,7 +130,7 @@ export const parseFormSchema = z.object({
compact_markdown_table: z.boolean().optional(),
markdown_table_multiline_header_separator: z.string().optional(),
page_error_tolerance: z.number().min(0).max(1).optional(),
replace_failed_page_mode: z.nativeEnum(FailPageMode).nullable().optional(),
replace_failed_page_mode: z.enum(FailPageMode).nullable().optional(),
replace_failed_page_with_error_message_prefix: z.string().optional(),
replace_failed_page_with_error_message_suffix: z.string().optional(),
});