mirror of
https://github.com/run-llama/llama_cloud_services.git
synced 2026-07-19 16:43:32 -04:00
Compare commits
35 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| d0649ece6e | |||
| 5d4cabd843 | |||
| 9070a6ac16 | |||
| 4f24f537f6 | |||
| 8859a203e2 | |||
| b091364054 | |||
| 43b1a013ca | |||
| f81532e7f2 | |||
| 986d3987d3 | |||
| 1bf522311f | |||
| 24166dcfc8 | |||
| bfb7f3973f | |||
| 979f643c77 | |||
| aefd89cf1b | |||
| 8ea2b2c64e | |||
| 4a9a2a21d8 | |||
| e6a7939206 | |||
| 104a03e829 | |||
| 6e0f2f4ca0 | |||
| 0708d11f8a | |||
| be19185503 | |||
| 7571b0d6c4 | |||
| ad6734bf80 | |||
| 9ec2a8322e | |||
| 51011b9f30 | |||
| 09805f9e15 | |||
| 8ced6f6eab | |||
| 081ddeca34 | |||
| 2460908789 | |||
| c226d6a54c | |||
| 5d4c682eb2 | |||
| f72d3535c8 | |||
| 1ea09a366e | |||
| d4bbeb6389 | |||
| d028397603 |
@@ -0,0 +1,8 @@
|
||||
# Changesets
|
||||
|
||||
Hello and welcome! This folder has been automatically generated by `@changesets/cli`, a build tool that works
|
||||
with multi-package repos, or single-package repos to help you version and publish your code. You can
|
||||
find the full documentation for it [in our repository](https://github.com/changesets/changesets)
|
||||
|
||||
We have a quick list of common questions to get you started engaging with this project in
|
||||
[our documentation](https://github.com/changesets/changesets/blob/main/docs/common-questions.md)
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"$schema": "https://unpkg.com/@changesets/config@3.1.1/schema.json",
|
||||
"changelog": "@changesets/cli/changelog",
|
||||
"commit": false,
|
||||
"fixed": [],
|
||||
"linked": [],
|
||||
"access": "restricted",
|
||||
"baseBranch": "main",
|
||||
"updateInternalDependencies": "patch",
|
||||
"ignore": []
|
||||
}
|
||||
@@ -27,7 +27,7 @@ jobs:
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v6
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: ${{ env.UV_VERSION }}
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ jobs:
|
||||
- uses: pnpm/action-setup@v4
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
uses: actions/setup-node@v5
|
||||
with:
|
||||
node-version-file: "ts/llama_cloud_services/.nvmrc"
|
||||
|
||||
|
||||
@@ -30,12 +30,12 @@ jobs:
|
||||
|
||||
# Initializes the CodeQL tools for scanning.
|
||||
- name: Initialize CodeQL
|
||||
uses: github/codeql-action/init@v3
|
||||
uses: github/codeql-action/init@v4
|
||||
with:
|
||||
languages: python
|
||||
dependency-caching: true
|
||||
|
||||
- name: Perform CodeQL Analysis
|
||||
uses: github/codeql-action/analyze@v3
|
||||
uses: github/codeql-action/analyze@v4
|
||||
with:
|
||||
category: "/language:python"
|
||||
|
||||
@@ -22,7 +22,7 @@ jobs:
|
||||
with:
|
||||
fetch-depth: ${{ github.event_name == 'pull_request' && 2 || 0 }}
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v6
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: ${{ env.UV_VERSION }}
|
||||
|
||||
@@ -31,7 +31,7 @@ jobs:
|
||||
|
||||
- uses: pnpm/action-setup@v4
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
uses: actions/setup-node@v5
|
||||
with:
|
||||
node-version-file: "ts/llama_cloud_services/.nvmrc"
|
||||
- name: Install dependencies
|
||||
|
||||
@@ -1,66 +0,0 @@
|
||||
name: Publish Release - Python
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "v*"
|
||||
|
||||
workflow_dispatch:
|
||||
|
||||
env:
|
||||
UV_VERSION: "0.7.20"
|
||||
|
||||
jobs:
|
||||
build-n-publish:
|
||||
name: Build and publish to PyPI
|
||||
if: github.repository == 'run-llama/llama_cloud_services'
|
||||
runs-on: ubuntu-latest
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v5
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v6
|
||||
with:
|
||||
version: ${{ env.UV_VERSION }}
|
||||
|
||||
- name: Set up Python
|
||||
run: uv python install
|
||||
|
||||
- name: Display Python version
|
||||
run: python --version
|
||||
|
||||
- name: Build
|
||||
working-directory: py
|
||||
run: uv build
|
||||
|
||||
- name: Test installing built package
|
||||
shell: bash
|
||||
working-directory: py
|
||||
run: |
|
||||
uv venv
|
||||
uv pip install dist/*.whl
|
||||
|
||||
- name: Publish package
|
||||
shell: bash
|
||||
working-directory: py
|
||||
run: uv publish --token ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
|
||||
|
||||
- name: Build and publish llama-parse
|
||||
working-directory: py/llama_parse/
|
||||
run: |
|
||||
uv build
|
||||
uv publish --token ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
|
||||
|
||||
- name: Create GitHub Release
|
||||
id: create_release
|
||||
uses: actions/create-release@v1
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} # This token is provided by Actions, you do not need to create your own token
|
||||
with:
|
||||
tag_name: ${{ github.ref }}
|
||||
release_name: ${{ github.ref }} - LlamaCloud Services PY
|
||||
artifacts: "py/**/dist/*"
|
||||
generateReleaseNotes: true
|
||||
draft: false
|
||||
prerelease: false
|
||||
@@ -1,52 +0,0 @@
|
||||
name: Publish Release - TypeScript
|
||||
|
||||
on:
|
||||
push:
|
||||
tags:
|
||||
- "llama-cloud-services@*"
|
||||
|
||||
jobs:
|
||||
build-and-publish:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout Repo
|
||||
uses: actions/checkout@v5
|
||||
|
||||
- uses: pnpm/action-setup@v4
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
with:
|
||||
node-version-file: "ts/llama_cloud_services/.nvmrc"
|
||||
|
||||
- name: Install dependencies
|
||||
run: pnpm install --no-frozen-lockfile
|
||||
|
||||
- name: Run Build
|
||||
working-directory: ts/llama_cloud_services/
|
||||
run: pnpm build
|
||||
|
||||
- name: Build tarball
|
||||
run: |
|
||||
pnpm pack
|
||||
working-directory: ts/llama_cloud_services
|
||||
|
||||
- name: Setup npm authentication
|
||||
run: echo "//registry.npmjs.org/:_authToken=${NPM_TOKEN}" > ~/.npmrc
|
||||
env:
|
||||
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Release
|
||||
working-directory: ts/llama_cloud_services
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
run: pnpm publish --access public --no-git-checks
|
||||
|
||||
- name: Create release
|
||||
uses: ncipollo/release-action@v1
|
||||
with:
|
||||
artifacts: "ts/llama_cloud_services/llama-cloud-services*.tgz"
|
||||
name: Release ${{ github.ref_name }} - LlamaCloud Services TS
|
||||
generateReleaseNotes: true
|
||||
token: ${{ secrets.GITHUB_TOKEN }}
|
||||
@@ -22,7 +22,7 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v6
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: ${{ env.UV_VERSION }}
|
||||
|
||||
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
with:
|
||||
fetch-depth: 0
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v6
|
||||
uses: astral-sh/setup-uv@v7
|
||||
with:
|
||||
version: ${{ env.UV_VERSION }}
|
||||
|
||||
|
||||
@@ -24,7 +24,7 @@ jobs:
|
||||
- uses: actions/checkout@v5
|
||||
- uses: pnpm/action-setup@v4
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v4
|
||||
uses: actions/setup-node@v5
|
||||
with:
|
||||
node-version-file: "ts/llama_cloud_services/.nvmrc"
|
||||
- name: Install dependencies
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
name: Version Bump and Release
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- main
|
||||
|
||||
concurrency: ${{ github.workflow }}-${{ github.ref }}
|
||||
|
||||
jobs:
|
||||
release:
|
||||
name: Release
|
||||
runs-on: ubuntu-latest
|
||||
# Only run on main branch pushes
|
||||
if: github.ref == 'refs/heads/main'
|
||||
steps:
|
||||
- name: Checkout Repo
|
||||
uses: actions/checkout@v5
|
||||
|
||||
- uses: pnpm/action-setup@v4
|
||||
|
||||
- name: Setup Node.js
|
||||
uses: actions/setup-node@v5
|
||||
with:
|
||||
node-version: "22"
|
||||
cache: "pnpm"
|
||||
|
||||
- name: Setup Python
|
||||
uses: actions/setup-python@v6
|
||||
with:
|
||||
python-version: "3.11"
|
||||
|
||||
- name: Install uv
|
||||
uses: astral-sh/setup-uv@v7
|
||||
|
||||
- name: Install dependencies
|
||||
run: pnpm install
|
||||
|
||||
- name: Add auth token to .npmrc file
|
||||
run: |
|
||||
cat << EOF >> ".npmrc"
|
||||
//registry.npmjs.org/:_authToken=$NPM_TOKEN
|
||||
EOF
|
||||
env:
|
||||
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Create Release Pull Request or Publish packages
|
||||
id: changesets
|
||||
uses: changesets/action@v1
|
||||
with:
|
||||
commit: "chore: version packages"
|
||||
title: "chore: version packages"
|
||||
# Custom version script
|
||||
version: pnpm -w run version
|
||||
# Custom publish script
|
||||
publish: pnpm -w run publish
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
UV_PUBLISH_TOKEN: ${{ secrets.PYPI_TOKEN }}
|
||||
LLAMA_PARSE_PYPI_TOKEN: ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
|
||||
@@ -9,3 +9,4 @@ __pycache__/
|
||||
node_modules/
|
||||
.turbo/
|
||||
dist/
|
||||
.npmrc
|
||||
|
||||
+8
-1
@@ -5,9 +5,16 @@
|
||||
"private": true,
|
||||
"keywords": [],
|
||||
"author": "",
|
||||
"scripts": {
|
||||
"pre-commit-version": "pnpm changeset",
|
||||
"version": "./scripts/changeset-version.py version",
|
||||
"publish": "./scripts/changeset-version.py publish --tag"
|
||||
},
|
||||
"devDependencies": {
|
||||
"prettier": "^3.6.2",
|
||||
"lint-staged": "^15.4.2"
|
||||
"lint-staged": "^15.4.2",
|
||||
"@changesets/cli": "^2.29.5",
|
||||
"changesets": "^1.0.2"
|
||||
},
|
||||
"lint-staged": {
|
||||
"ts/llama_cloud_services/src/**/*.{ts,tsx,js,jsx}": [
|
||||
|
||||
Generated
+589
-10
File diff suppressed because it is too large
Load Diff
+3
-1
@@ -1,2 +1,4 @@
|
||||
packages:
|
||||
- "ts/**"
|
||||
- "ts/*"
|
||||
- "py"
|
||||
- "py/*"
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
# llama-cloud-services-py
|
||||
|
||||
## 0.6.76
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 4f24f53: Add aggressive_table_extraction flag in python sdk
|
||||
|
||||
## 0.6.75
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- f81532e: Safest types possible for parse
|
||||
|
||||
## 0.6.74
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 1bf5223: Fix default bbox values
|
||||
- 24166dc: Now only escape single dollar signs - preserve double for latex equations
|
||||
|
||||
## 0.6.73
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- e6a7939: Loosen packaging dep requirement
|
||||
|
||||
## 0.6.72
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- ad6734b: Fixup and test versioning
|
||||
|
||||
## 0.6.71
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 51011b9: Escape dollar signs in jupyter notebooks
|
||||
|
||||
## 0.6.70
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- d028397: Update llama-cloud api version, and integrate with agent data deletion
|
||||
@@ -194,6 +194,21 @@ class AsyncAgentDataClient(Generic[AgentDataT]):
|
||||
async def delete_item(self, item_id: str) -> None:
|
||||
await self.client.beta.delete_agent_data(item_id=item_id)
|
||||
|
||||
@agent_data_retry
|
||||
async def delete(
|
||||
self, filter: Optional[Dict[str, Dict[ComparisonOperator, Any]]] = None
|
||||
) -> int:
|
||||
"""
|
||||
Delete agent data by query, similar to search.
|
||||
Returns the number of deleted items.
|
||||
"""
|
||||
response = await self.client.beta.delete_agent_data_by_query_api_v_1_beta_agent_data_delete_post(
|
||||
deployment_name=self.deployment_name,
|
||||
collection=self.collection,
|
||||
filter=filter,
|
||||
)
|
||||
return response.deleted_count
|
||||
|
||||
@agent_data_retry
|
||||
async def search(
|
||||
self,
|
||||
|
||||
@@ -19,14 +19,12 @@ from llama_cloud import (
|
||||
ExtractAgent as CloudExtractAgent,
|
||||
ExtractConfig,
|
||||
ExtractJob,
|
||||
ExtractJobCreate,
|
||||
ExtractRun,
|
||||
File,
|
||||
FileData,
|
||||
ExtractMode,
|
||||
StatusEnum,
|
||||
ExtractTarget,
|
||||
LlamaExtractSettings,
|
||||
PaginatedExtractRunsResponse,
|
||||
)
|
||||
from llama_cloud.client import AsyncLlamaCloud
|
||||
@@ -463,56 +461,6 @@ class ExtractionAgent:
|
||||
)
|
||||
)
|
||||
|
||||
async def _run_extraction_test(
|
||||
self,
|
||||
files: Union[FileInput, List[FileInput]],
|
||||
extract_settings: LlamaExtractSettings,
|
||||
) -> Union[ExtractJob, List[ExtractJob]]:
|
||||
if not isinstance(files, list):
|
||||
files = [files]
|
||||
single_file = True
|
||||
else:
|
||||
single_file = False
|
||||
|
||||
upload_tasks = [self._upload_file(file) for file in files]
|
||||
with augment_async_errors():
|
||||
uploaded_files = await run_jobs(
|
||||
upload_tasks,
|
||||
workers=self.num_workers,
|
||||
desc="Uploading files",
|
||||
show_progress=self.show_progress,
|
||||
)
|
||||
|
||||
async def run_job(file: File) -> ExtractRun:
|
||||
job_queued = await self._client.llama_extract.run_job_test_user(
|
||||
job_create=ExtractJobCreate(
|
||||
extraction_agent_id=self.id,
|
||||
file_id=file.id,
|
||||
data_schema_override=self.data_schema,
|
||||
config_override=self.config,
|
||||
),
|
||||
extract_settings=extract_settings,
|
||||
)
|
||||
return await self._wait_for_job_result(job_queued.id)
|
||||
|
||||
job_tasks = [run_job(file) for file in uploaded_files]
|
||||
with augment_async_errors():
|
||||
extract_results = await run_jobs(
|
||||
job_tasks,
|
||||
workers=self.num_workers,
|
||||
desc="Running extraction jobs",
|
||||
show_progress=self.show_progress,
|
||||
)
|
||||
|
||||
if self._verbose:
|
||||
for file, job in zip(files, extract_results):
|
||||
file_repr = (
|
||||
str(file) if isinstance(file, (str, Path)) else "<bytes/buffer>"
|
||||
)
|
||||
print(f"Running extraction for file {file_repr} under job_id {job.id}")
|
||||
|
||||
return extract_results[0] if single_file else extract_results
|
||||
|
||||
async def queue_extraction(
|
||||
self,
|
||||
files: Union[FileInput, List[FileInput]],
|
||||
@@ -544,12 +492,10 @@ class ExtractionAgent:
|
||||
|
||||
job_tasks = [
|
||||
self._client.llama_extract.run_job(
|
||||
request=ExtractJobCreate(
|
||||
extraction_agent_id=self.id,
|
||||
file_id=file.id,
|
||||
data_schema_override=self.data_schema,
|
||||
config_override=self.config,
|
||||
),
|
||||
extraction_agent_id=self.id,
|
||||
file_id=file.id,
|
||||
data_schema_override=self.data_schema,
|
||||
config_override=self.config,
|
||||
)
|
||||
for file in uploaded_files
|
||||
]
|
||||
|
||||
@@ -188,6 +188,10 @@ class LlamaParse(BasePydanticReader):
|
||||
default=False,
|
||||
description="If set to true, LlamaParse will try to detect long table and adapt the output.",
|
||||
)
|
||||
aggressive_table_extraction: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, LlamaParse will try to extract tables aggressively, may lead to false positives.",
|
||||
)
|
||||
annotate_links: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Annotate links found in the document to extract their URL.",
|
||||
@@ -713,6 +717,9 @@ class LlamaParse(BasePydanticReader):
|
||||
if self.adaptive_long_table:
|
||||
data["adaptive_long_table"] = self.adaptive_long_table
|
||||
|
||||
if self.aggressive_table_extraction:
|
||||
data["aggressive_table_extraction"] = self.aggressive_table_extraction
|
||||
|
||||
if self.annotate_links:
|
||||
data["annotate_links"] = self.annotate_links
|
||||
|
||||
|
||||
@@ -1,17 +1,87 @@
|
||||
import httpx
|
||||
import os
|
||||
import re
|
||||
from pydantic import BaseModel, Field, SerializeAsAny
|
||||
from typing import Dict, Any, List, Optional
|
||||
from pydantic import BaseModel, ConfigDict, Field, SerializeAsAny, model_validator
|
||||
from typing import Dict, Any, List, Optional, get_origin, get_args
|
||||
|
||||
from llama_cloud_services.parse.utils import make_api_request
|
||||
from llama_cloud_services.parse.utils import (
|
||||
make_api_request,
|
||||
is_jupyter,
|
||||
)
|
||||
from llama_index.core.async_utils import asyncio_run
|
||||
from llama_index.core.schema import Document, ImageDocument, ImageNode, TextNode
|
||||
|
||||
PAGE_REGEX = r"page[-_](\d+)\.jpg$"
|
||||
|
||||
SAFE_MODEL_CONFIGS = ConfigDict(
|
||||
extra="allow",
|
||||
validate_assignment=False,
|
||||
arbitrary_types_allowed=True,
|
||||
validate_default=False,
|
||||
)
|
||||
|
||||
class JobMetadata(BaseModel):
|
||||
|
||||
class SafeBaseModel(BaseModel):
|
||||
"""Base model that gracefully handles None values from unstable backend responses."""
|
||||
|
||||
model_config = SAFE_MODEL_CONFIGS
|
||||
|
||||
@model_validator(mode="before")
|
||||
@classmethod
|
||||
def coerce_none_to_defaults(cls, data: Any) -> Any:
|
||||
"""
|
||||
Replace None values with appropriate defaults based on field type annotations.
|
||||
This prevents validation errors when the backend returns None for non-optional fields.
|
||||
"""
|
||||
if not isinstance(data, dict):
|
||||
return data
|
||||
|
||||
# Process each field that has a None value
|
||||
result = {}
|
||||
for key, value in data.items():
|
||||
if value is not None or key not in cls.model_fields:
|
||||
result[key] = value
|
||||
continue
|
||||
|
||||
# Value is None and field exists in model
|
||||
field_info = cls.model_fields[key]
|
||||
|
||||
# If field has a default or default_factory, let Pydantic handle it
|
||||
from pydantic_core import PydanticUndefined
|
||||
|
||||
if (
|
||||
field_info.default is not PydanticUndefined
|
||||
or field_info.default_factory is not None
|
||||
):
|
||||
continue
|
||||
|
||||
# Otherwise, provide a sensible default based on the type annotation
|
||||
annotation = field_info.annotation
|
||||
origin = get_origin(annotation)
|
||||
|
||||
# Handle List types
|
||||
if origin is list:
|
||||
result[key] = []
|
||||
# Handle Dict types
|
||||
elif origin is dict:
|
||||
result[key] = {}
|
||||
# Handle basic types
|
||||
elif annotation == str or (origin and str in get_args(annotation)):
|
||||
result[key] = ""
|
||||
elif annotation == int or (origin and int in get_args(annotation)):
|
||||
result[key] = 0
|
||||
elif annotation == float or (origin and float in get_args(annotation)):
|
||||
result[key] = 0.0
|
||||
elif annotation == bool or (origin and bool in get_args(annotation)):
|
||||
result[key] = False
|
||||
# If we can't determine a safe default, skip (let Pydantic try)
|
||||
else:
|
||||
result[key] = value
|
||||
|
||||
return result
|
||||
|
||||
|
||||
class JobMetadata(SafeBaseModel):
|
||||
"""Metadata about the job."""
|
||||
|
||||
job_pages: int = Field(default=0, description="The number of pages in the job.")
|
||||
@@ -24,19 +94,31 @@ class JobMetadata(BaseModel):
|
||||
)
|
||||
|
||||
|
||||
class BBox(BaseModel):
|
||||
class BBox(SafeBaseModel):
|
||||
"""A bounding box."""
|
||||
|
||||
x: float = Field(description="The x-coordinate of the bounding box.")
|
||||
y: float = Field(description="The y-coordinate of the bounding box.")
|
||||
w: float = Field(description="The width of the bounding box.")
|
||||
h: float = Field(description="The height of the bounding box.")
|
||||
x: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The x-coordinate of the bounding box.",
|
||||
)
|
||||
y: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The y-coordinate of the bounding box.",
|
||||
)
|
||||
w: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The width of the bounding box.",
|
||||
)
|
||||
h: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The height of the bounding box.",
|
||||
)
|
||||
|
||||
|
||||
class PageItem(BaseModel):
|
||||
class PageItem(SafeBaseModel):
|
||||
"""An item in a page."""
|
||||
|
||||
type: str = Field(description="The type of the item.")
|
||||
type: str = Field(default="", description="The type of the item.")
|
||||
lvl: Optional[int] = Field(
|
||||
default=None, description="The level of indentation of the item."
|
||||
)
|
||||
@@ -58,10 +140,10 @@ class PageItem(BaseModel):
|
||||
)
|
||||
|
||||
|
||||
class ImageItem(BaseModel):
|
||||
class ImageItem(SafeBaseModel):
|
||||
"""An image in a page."""
|
||||
|
||||
name: str = Field(description="The name of the image.")
|
||||
name: str = Field(default="", description="The name of the image.")
|
||||
height: Optional[float] = Field(
|
||||
default=None, description="The height of the image."
|
||||
)
|
||||
@@ -81,22 +163,28 @@ class ImageItem(BaseModel):
|
||||
type: Optional[str] = Field(default=None, description="The type of the image.")
|
||||
|
||||
|
||||
class LayoutItem(BaseModel):
|
||||
class LayoutItem(SafeBaseModel):
|
||||
"""The layout of a page."""
|
||||
|
||||
image: str = Field(description="The name of the image containing the layout item")
|
||||
confidence: float = Field(description="The confidence of the layout item.")
|
||||
label: str = Field(description="The label of the layout item.")
|
||||
image: str = Field(
|
||||
default="", description="The name of the image containing the layout item"
|
||||
)
|
||||
confidence: float = Field(
|
||||
default=0.0, description="The confidence of the layout item."
|
||||
)
|
||||
label: str = Field(default="", description="The label of the layout item.")
|
||||
bbox: Optional[BBox] = Field(
|
||||
default=None, description="The bounding box of the layout item."
|
||||
)
|
||||
isLikelyNoise: bool = Field(description="Whether the layout item is likely noise.")
|
||||
isLikelyNoise: bool = Field(
|
||||
default=False, description="Whether the layout item is likely noise."
|
||||
)
|
||||
|
||||
|
||||
class ChartItem(BaseModel):
|
||||
class ChartItem(SafeBaseModel):
|
||||
"""A chart in a page."""
|
||||
|
||||
name: str = Field(description="The name of the chart.")
|
||||
name: str = Field(default="", description="The name of the chart.")
|
||||
x: Optional[float] = Field(
|
||||
default=None, description="The x-coordinate of the chart."
|
||||
)
|
||||
@@ -109,7 +197,7 @@ class ChartItem(BaseModel):
|
||||
)
|
||||
|
||||
|
||||
class Page(BaseModel):
|
||||
class Page(SafeBaseModel):
|
||||
"""A page of the document."""
|
||||
|
||||
page: int = Field(default=0, description="The page number.")
|
||||
@@ -164,7 +252,7 @@ class Page(BaseModel):
|
||||
)
|
||||
|
||||
|
||||
class JobResult(BaseModel):
|
||||
class JobResult(SafeBaseModel):
|
||||
"""The raw JSON result from the LlamaParse API."""
|
||||
|
||||
pages: List[Page] = Field(
|
||||
@@ -258,6 +346,29 @@ class JobResult(BaseModel):
|
||||
documents = await self.aget_text_documents(split_by_page)
|
||||
return [TextNode(text=doc.text, metadata=doc.metadata) for doc in documents]
|
||||
|
||||
def _format_markdown_for_notebook(self, text: Optional[str]) -> Optional[str]:
|
||||
"""Format markdown text for Jupyter notebook display by escaping dollar signs."""
|
||||
if text is None:
|
||||
return None
|
||||
|
||||
def escape_single_dollar_signs(text: str) -> str:
|
||||
"""Escape single dollar signs in text to prevent Jupyter from interpreting them as LaTeX.
|
||||
|
||||
Preserves all strings of dollar signs greater than length 1,
|
||||
especially preserving double dollar signs ($$) which denote LaTeX equations.
|
||||
|
||||
Args:
|
||||
text: The text to escape
|
||||
|
||||
Returns:
|
||||
Text with single dollar signs escaped
|
||||
"""
|
||||
# Replace single $ with \$, but preserve $$
|
||||
# Use negative lookahead and lookbehind to match $ not preceded or followed by $
|
||||
return re.sub(r"(?<!\$)\$(?!\$)", r"\$", text)
|
||||
|
||||
return escape_single_dollar_signs(text)
|
||||
|
||||
def get_markdown_documents(self, split_by_page: bool = False) -> List[Document]:
|
||||
"""
|
||||
Get the markdown documents from the job.
|
||||
@@ -268,17 +379,22 @@ class JobResult(BaseModel):
|
||||
if split_by_page:
|
||||
return [
|
||||
Document(
|
||||
text=page.md,
|
||||
text=self._format_markdown_for_notebook(page.md)
|
||||
if is_jupyter()
|
||||
else page.md,
|
||||
metadata={"page_number": page.page, "file_name": self.file_name},
|
||||
)
|
||||
for page in self.pages
|
||||
]
|
||||
else:
|
||||
text = self._page_separator.join(
|
||||
[page.md if page.md is not None else "" for page in self.pages]
|
||||
)
|
||||
return [
|
||||
Document(
|
||||
text=self._page_separator.join(
|
||||
[page.md if page.md is not None else "" for page in self.pages]
|
||||
),
|
||||
text=self._format_markdown_for_notebook(text)
|
||||
if is_jupyter()
|
||||
else text,
|
||||
metadata={"file_name": self.file_name},
|
||||
)
|
||||
]
|
||||
@@ -328,7 +444,10 @@ class JobResult(BaseModel):
|
||||
"""
|
||||
url = f"{self._base_url}/api/v1/parsing/job/{self.job_id}/result/raw/markdown"
|
||||
response = await make_api_request(self._client, "GET", url)
|
||||
return response.content.decode("utf-8")
|
||||
markdown = response.content.decode("utf-8")
|
||||
return (
|
||||
self._format_markdown_for_notebook(markdown) if is_jupyter() else markdown
|
||||
)
|
||||
|
||||
def get_text(self) -> str:
|
||||
"""
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import functools
|
||||
import httpx
|
||||
import itertools
|
||||
import logging
|
||||
@@ -356,6 +357,17 @@ def partition_pages(
|
||||
return
|
||||
|
||||
|
||||
@functools.lru_cache(maxsize=1)
|
||||
def is_jupyter() -> bool:
|
||||
"""Check if we're running in a Jupyter environment."""
|
||||
try:
|
||||
from IPython import get_ipython
|
||||
|
||||
return get_ipython().__class__.__name__ == "ZMQInteractiveShell"
|
||||
except (ImportError, AttributeError):
|
||||
return False
|
||||
|
||||
|
||||
def extract_tables_from_json_results(
|
||||
json_results: List[dict], download_path: str
|
||||
) -> List[str]:
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
# llama_parse
|
||||
|
||||
## 0.6.76
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Updated dependencies [4f24f53]
|
||||
- llama-cloud-services-py@0.6.76
|
||||
|
||||
## 0.6.75
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Updated dependencies [f81532e]
|
||||
- llama-cloud-services-py@0.6.75
|
||||
|
||||
## 0.6.74
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Updated dependencies [1bf5223]
|
||||
- Updated dependencies [24166dc]
|
||||
- llama-cloud-services-py@0.6.74
|
||||
|
||||
## 0.6.73
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Updated dependencies [e6a7939]
|
||||
- llama-cloud-services-py@0.6.73
|
||||
|
||||
## 0.6.72
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Updated dependencies [ad6734b]
|
||||
- llama-cloud-services-py@0.6.72
|
||||
@@ -0,0 +1,20 @@
|
||||
{
|
||||
"name": "llama_parse",
|
||||
"version": "0.6.76",
|
||||
"description": "",
|
||||
"main": "index.js",
|
||||
"private": false,
|
||||
"scripts": {
|
||||
"test": "echo \"Error: no test specified\" && exit 1"
|
||||
},
|
||||
"dependencies": {
|
||||
"llama-cloud-services-py": "workspace:*"
|
||||
},
|
||||
"keywords": [],
|
||||
"author": "",
|
||||
"license": "ISC",
|
||||
"packageManager": "pnpm@10.11.1",
|
||||
"devDependencies": {
|
||||
"changesets": "^1.0.2"
|
||||
}
|
||||
}
|
||||
@@ -11,13 +11,13 @@ dev = [
|
||||
|
||||
[project]
|
||||
name = "llama-parse"
|
||||
version = "0.6.69"
|
||||
version = "0.6.76"
|
||||
description = "Parse files into RAG-Optimized formats."
|
||||
authors = [{name = "Logan Markewich", email = "logan@llamaindex.ai"}]
|
||||
requires-python = ">=3.9,<4.0"
|
||||
readme = "README.md"
|
||||
license = "MIT"
|
||||
dependencies = ["llama-cloud-services>=0.6.69"]
|
||||
dependencies = ["llama-cloud-services>=0.6.76"]
|
||||
|
||||
[project.scripts]
|
||||
llama-parse = "llama_parse.cli.main:parse"
|
||||
|
||||
@@ -0,0 +1,10 @@
|
||||
{
|
||||
"name": "llama-cloud-services-py",
|
||||
"version": "0.6.76",
|
||||
"private": false,
|
||||
"license": "MIT",
|
||||
"scripts": {},
|
||||
"devDependencies": {
|
||||
"changesets": "^1.0.2"
|
||||
}
|
||||
}
|
||||
+3
-3
@@ -19,7 +19,7 @@ dev = [
|
||||
|
||||
[project]
|
||||
name = "llama-cloud-services"
|
||||
version = "0.6.69"
|
||||
version = "0.6.76"
|
||||
description = "Tailored SDK clients for LlamaCloud services."
|
||||
authors = [{name = "Logan Markewich", email = "logan@runllama.ai"}]
|
||||
requires-python = ">=3.9,<4.0"
|
||||
@@ -27,14 +27,14 @@ readme = "README.md"
|
||||
license = "MIT"
|
||||
dependencies = [
|
||||
"llama-index-core>=0.12.0",
|
||||
"llama-cloud==0.1.42",
|
||||
"llama-cloud==0.1.43",
|
||||
"pydantic>=2.8,!=2.10",
|
||||
"click>=8.1.7,<9",
|
||||
"python-dotenv>=1.0.1,<2",
|
||||
"eval-type-backport>=0.2.0,<0.3 ; python_version < '3.10'",
|
||||
"platformdirs>=4.3.7,<5",
|
||||
"tenacity>=8.5.0, <10.0",
|
||||
"packaging>=25.0"
|
||||
"packaging>=23.0"
|
||||
]
|
||||
|
||||
[project.scripts]
|
||||
|
||||
@@ -1,16 +1,13 @@
|
||||
import os
|
||||
import pytest
|
||||
|
||||
from llama_cloud_services.extract import LlamaExtract, ExtractionAgent
|
||||
from time import perf_counter
|
||||
from llama_cloud_services.extract import LlamaExtract
|
||||
from collections import namedtuple
|
||||
import json
|
||||
import uuid
|
||||
from llama_cloud.types import (
|
||||
ExtractConfig,
|
||||
ExtractMode,
|
||||
LlamaParseParameters,
|
||||
LlamaExtractSettings,
|
||||
)
|
||||
from tests.extract.util import load_test_dotenv
|
||||
|
||||
@@ -122,27 +119,3 @@ def extraction_agent(test_case: BenchmarkTestCase, extractor: LlamaExtract):
|
||||
# Create new agent
|
||||
agent = extractor.create_agent(agent_name, schema, config=test_case.config)
|
||||
yield agent
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
"CI" in os.environ or not LLAMA_CLOUD_API_KEY,
|
||||
reason="LLAMA_CLOUD_API_KEY not set or CI environment not suitable for benchmarking",
|
||||
)
|
||||
@pytest.mark.parametrize("test_case", get_test_cases(), ids=lambda x: x.name)
|
||||
@pytest.mark.asyncio(loop_scope="session")
|
||||
async def test_extraction(
|
||||
test_case: BenchmarkTestCase, extraction_agent: ExtractionAgent
|
||||
) -> None:
|
||||
start = perf_counter()
|
||||
result = await extraction_agent._run_extraction_test(
|
||||
test_case.input_file,
|
||||
extract_settings=LlamaExtractSettings(
|
||||
llama_parse_params=LlamaParseParameters(
|
||||
invalidate_cache=True,
|
||||
do_not_cache=True,
|
||||
)
|
||||
),
|
||||
)
|
||||
end = perf_counter()
|
||||
print(f"Time taken: {end - start} seconds")
|
||||
print(result)
|
||||
|
||||
@@ -7,7 +7,7 @@ from pathlib import Path
|
||||
|
||||
|
||||
def load_test_dotenv():
|
||||
load_dotenv(Path(__file__).parent.parent.parent / ".env.dev", override=True)
|
||||
load_dotenv(Path(__file__).parent.parent.parent.parent / ".env.dev", override=True)
|
||||
|
||||
|
||||
def json_subset_match_score(expected: Any, actual: Any) -> float:
|
||||
|
||||
@@ -304,6 +304,9 @@ async def test_page_screenshot_retrieval(index_name: str, local_file: str):
|
||||
not base_url or not api_key, reason="No platform base url or api key set"
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.skip(
|
||||
reason="Consistently failing with FAILED tests/index/test_index.py::test_page_figure_retrieval - assert 0 > 0 + where 0 = len([])"
|
||||
)
|
||||
async def test_page_figure_retrieval(index_name: str, local_figures_file: str):
|
||||
index = await LlamaCloudIndex.acreate_index(
|
||||
name=index_name,
|
||||
|
||||
@@ -6,6 +6,40 @@ from llama_cloud_services import LlamaParse
|
||||
from llama_cloud_services.parse.types import JobResult
|
||||
|
||||
|
||||
def test_format_parse_result_markdown_for_notebook():
|
||||
"""Test the _format_markdown_for_notebook function.
|
||||
Right now, the only work it does is escape single dollar signs."""
|
||||
result = JobResult(job_id="test", file_name="test.pdf", job_result={})
|
||||
|
||||
# Test None input
|
||||
assert result._format_markdown_for_notebook(None) is None
|
||||
|
||||
# Test single dollar sign gets escaped
|
||||
assert result._format_markdown_for_notebook("This costs $5") == "This costs \\$5"
|
||||
|
||||
# Test double dollar signs are preserved (LaTeX equations)
|
||||
assert (
|
||||
result._format_markdown_for_notebook("$$x^2 + y^2 = z^2$$")
|
||||
== "$$x^2 + y^2 = z^2$$"
|
||||
)
|
||||
|
||||
# Test mixed single and double dollar signs
|
||||
text = "This costs $5, but $$E = mc^2$$ is priceless"
|
||||
expected = "This costs \\$5, but $$E = mc^2$$ is priceless"
|
||||
assert result._format_markdown_for_notebook(text) == expected
|
||||
|
||||
# Test multiple single dollar signs
|
||||
assert result._format_markdown_for_notebook("$10 and $20") == "\\$10 and \\$20"
|
||||
|
||||
# Test three or more consecutive dollar signs (preserve them)
|
||||
assert result._format_markdown_for_notebook("$$$") == "$$$"
|
||||
|
||||
# Test adjacent dollar signs with text in between
|
||||
text = "$$inline$$ and $separate"
|
||||
expected = "$$inline$$ and \\$separate"
|
||||
assert result._format_markdown_for_notebook(text) == expected
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def file_path() -> str:
|
||||
return "tests/test_files/attention_is_all_you_need.pdf"
|
||||
|
||||
@@ -118,10 +118,8 @@ async def test_extraction_agent_aextract_accepts_llama_file(
|
||||
dummy_llama_extract_iface = SimpleNamespace()
|
||||
|
||||
async def fake_run_job(**kwargs):
|
||||
# Ensure we are receiving a request with the right file_id
|
||||
request = kwargs.get("request")
|
||||
assert hasattr(request, "file_id")
|
||||
assert request.file_id == llama_file.id
|
||||
file_id = kwargs.get("file_id")
|
||||
assert file_id == llama_file.id
|
||||
return SimpleNamespace(id="job_42")
|
||||
|
||||
dummy_llama_extract_iface.run_job = fake_run_job
|
||||
|
||||
Generated
+7
-7
@@ -1,5 +1,5 @@
|
||||
version = 1
|
||||
revision = 2
|
||||
revision = 3
|
||||
requires-python = ">=3.9, <4.0"
|
||||
resolution-markers = [
|
||||
"python_full_version >= '3.14'",
|
||||
@@ -1582,21 +1582,21 @@ wheels = [
|
||||
|
||||
[[package]]
|
||||
name = "llama-cloud"
|
||||
version = "0.1.42"
|
||||
version = "0.1.43"
|
||||
source = { registry = "https://pypi.org/simple" }
|
||||
dependencies = [
|
||||
{ name = "certifi" },
|
||||
{ name = "httpx" },
|
||||
{ name = "pydantic" },
|
||||
]
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/21/04/ae0694b582d6aab4d6e7957febb7bff048897ac231ad80ba1bd71547d944/llama_cloud-0.1.42.tar.gz", hash = "sha256:485aa0e364ea648e3aaa3b2c54af7bcb6f2242c50b4f86ec022e137413fff464", size = 112480, upload-time = "2025-09-16T20:25:42.631Z" }
|
||||
sdist = { url = "https://files.pythonhosted.org/packages/9b/33/33a8bd3a617c071caf450ca2627969f8b28272d0692f122997c10a32247e/llama_cloud-0.1.43.tar.gz", hash = "sha256:00429f05aea515449d90cde91ef3ed3687fcd93e46f6246d08cbea02f9b397a9", size = 112992, upload-time = "2025-10-02T21:55:38.355Z" }
|
||||
wheels = [
|
||||
{ url = "https://files.pythonhosted.org/packages/6a/61/85d115699a59d03f0783e119aaf6d534fca95dbe1a4531a8056e6a4774ed/llama_cloud-0.1.42-py3-none-any.whl", hash = "sha256:4ed3edde4a277ff52eeb831188c8476eb079b5e4605ad3142157a0f054b27d96", size = 311857, upload-time = "2025-09-16T20:25:41.479Z" },
|
||||
{ url = "https://files.pythonhosted.org/packages/2b/54/559a67542396d5660a71115b29e0160e9dd784e570e1f4ef55ad22bf5b39/llama_cloud-0.1.43-py3-none-any.whl", hash = "sha256:540605d4dd13c6536a3b75cd4d04b211f29b16d17faee9381e3793a651f1dec1", size = 311460, upload-time = "2025-10-02T21:55:37.282Z" },
|
||||
]
|
||||
|
||||
[[package]]
|
||||
name = "llama-cloud-services"
|
||||
version = "0.6.68"
|
||||
version = "0.6.73"
|
||||
source = { editable = "." }
|
||||
dependencies = [
|
||||
{ name = "click", version = "8.1.8", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.10'" },
|
||||
@@ -1631,9 +1631,9 @@ dev = [
|
||||
requires-dist = [
|
||||
{ name = "click", specifier = ">=8.1.7,<9" },
|
||||
{ name = "eval-type-backport", marker = "python_full_version < '3.10'", specifier = ">=0.2.0,<0.3" },
|
||||
{ name = "llama-cloud", specifier = "==0.1.42" },
|
||||
{ name = "llama-cloud", specifier = "==0.1.43" },
|
||||
{ name = "llama-index-core", specifier = ">=0.12.0" },
|
||||
{ name = "packaging", specifier = ">=25.0" },
|
||||
{ name = "packaging", specifier = ">=23.0" },
|
||||
{ name = "platformdirs", specifier = ">=4.3.7,<5" },
|
||||
{ name = "pydantic", specifier = ">=2.8,!=2.10" },
|
||||
{ name = "python-dotenv", specifier = ">=1.0.1,<2" },
|
||||
|
||||
Executable
+290
@@ -0,0 +1,290 @@
|
||||
#!/usr/bin/env -S uv run --script
|
||||
# /// script
|
||||
# dependencies = ["click", "tomlkit", "packaging"]
|
||||
# ///
|
||||
|
||||
"""
|
||||
This is a script called by the changeset bot. Normally changeset can do the following things, but this is a mixed ts and python repo, so we need to do some extra things.
|
||||
|
||||
There's 2 things this does:
|
||||
- Versioning: Makes changes that may be committed with the newest version.
|
||||
- Releasing/Tagging: After versions are changed, we check each package to see if its released, and if not, we release it and tag it.
|
||||
|
||||
"""
|
||||
|
||||
from dataclasses import dataclass
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
from typing import Any, List, cast
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
import re
|
||||
|
||||
import click
|
||||
import tomlkit
|
||||
from packaging.version import Version
|
||||
|
||||
|
||||
def _run_command(
|
||||
cmd: List[str], cwd: Path | None = None, env: dict[str, str] | None = None
|
||||
) -> None:
|
||||
"""Run a command, streaming output to the console, and raise on failure."""
|
||||
subprocess.run(cmd, check=True, text=True, cwd=cwd or Path.cwd(), env=env)
|
||||
|
||||
|
||||
def _run_and_capture(
|
||||
cmd: List[str], cwd: Path | None = None, env: dict[str, str] | None = None
|
||||
) -> str:
|
||||
"""Run a command and return stdout as text, raising on failure."""
|
||||
result = subprocess.run(
|
||||
cmd,
|
||||
check=True,
|
||||
text=True,
|
||||
cwd=cwd or Path.cwd(),
|
||||
env=env,
|
||||
capture_output=True,
|
||||
)
|
||||
return result.stdout
|
||||
|
||||
|
||||
@dataclass
|
||||
class Package:
|
||||
name: str
|
||||
version: str
|
||||
path: Path
|
||||
|
||||
def python_package_name(self) -> str | None:
|
||||
if "/py/" in str(self.path) or str(self.path).endswith("/py"):
|
||||
return self.name.removesuffix("-py")
|
||||
return None
|
||||
|
||||
|
||||
def _get_pnpm_workspace_packages() -> list[Package]:
|
||||
"""Return directories for all workspace packages from pnpm list JSON output."""
|
||||
output = _run_and_capture(["pnpm", "list", "-r", "--depth=-1", "--json"])
|
||||
|
||||
data = cast(list[dict[str, Any]], json.loads(output))
|
||||
packages: list[Package] = [
|
||||
Package(name=data["name"], version=data["version"], path=Path(data["path"]))
|
||||
for data in data
|
||||
]
|
||||
return packages
|
||||
|
||||
|
||||
def _sync_package_version_with_pyproject(
|
||||
package_dir: Path, packages: dict[str, Package], js_package_name: str
|
||||
) -> None:
|
||||
"""Sync version from package.json to pyproject.toml.
|
||||
|
||||
Returns True if pyproject was changed, else False.
|
||||
"""
|
||||
pyproject_path = package_dir / "pyproject.toml"
|
||||
if not pyproject_path.exists():
|
||||
return
|
||||
|
||||
package_version = packages[js_package_name].version
|
||||
py_doc = tomlkit.parse(pyproject_path.read_text())
|
||||
|
||||
by_python_name = {
|
||||
pkg.python_package_name(): pkg
|
||||
for pkg in packages.values()
|
||||
if pkg.python_package_name()
|
||||
}
|
||||
|
||||
current_version = py_doc["project"]["version"]
|
||||
assert isinstance(current_version, str)
|
||||
|
||||
# update workspace dependency strings by replacing the first version after == or >=
|
||||
deps = py_doc["project"]["dependencies"] or []
|
||||
changed = False
|
||||
for i, dep in enumerate(deps):
|
||||
if not isinstance(dep, str):
|
||||
continue
|
||||
pkg = (cast(str, dep).split("==")[0]).split(">=")[0]
|
||||
if pkg not in by_python_name:
|
||||
continue
|
||||
target_version = by_python_name[pkg].version
|
||||
new_dep = re.sub(
|
||||
r"(==|>=)\s*([0-9A-Za-z_.+-]+)",
|
||||
lambda m: m.group(1) + target_version,
|
||||
dep,
|
||||
count=1,
|
||||
)
|
||||
if new_dep != dep:
|
||||
deps[i] = new_dep
|
||||
changed = True
|
||||
|
||||
if current_version != package_version:
|
||||
py_doc["project"]["version"] = package_version
|
||||
changed = True
|
||||
|
||||
if changed:
|
||||
pyproject_path.write_text(tomlkit.dumps(py_doc))
|
||||
click.echo(
|
||||
f"Updated {pyproject_path} version to {package_version} and synced dependency specs"
|
||||
)
|
||||
|
||||
|
||||
def lock_python_dependencies() -> None:
|
||||
"""Lock Python dependencies."""
|
||||
try:
|
||||
_run_command(["uv", "lock"])
|
||||
click.echo("Locked Python dependencies")
|
||||
except subprocess.CalledProcessError as e:
|
||||
click.echo(f"Warning: Failed to lock Python dependencies: {e}", err=True)
|
||||
|
||||
|
||||
@click.group()
|
||||
def cli() -> None:
|
||||
"""Changeset-based version management for llama-cloud-services."""
|
||||
pass
|
||||
|
||||
|
||||
@cli.command()
|
||||
def version() -> None:
|
||||
"""Apply changeset versions, then sync versions for co-located JS/Py packages.
|
||||
|
||||
- Runs changesets to bump package.json versions.
|
||||
- Discovers all workspace packages via pnpm.
|
||||
- For any directory containing both package.json and pyproject.toml, and with
|
||||
package.json private: false, set pyproject [project].version to match the JS version.
|
||||
- If a pyproject is updated, run `uv sync` in that directory to update its lock file.
|
||||
"""
|
||||
# Ensure we're at the repo root
|
||||
os.chdir(Path(__file__).parent.parent)
|
||||
|
||||
# First, run changeset version to update all package.json files
|
||||
_run_command(["npx", "@changesets/cli", "version"])
|
||||
|
||||
# Enumerate workspace packages and perform syncs
|
||||
packages = _get_pnpm_workspace_packages()
|
||||
version_map = {pkg.name: pkg for pkg in packages}
|
||||
for pkg in packages:
|
||||
_sync_package_version_with_pyproject(pkg.path, version_map, pkg.name)
|
||||
|
||||
|
||||
@cli.command()
|
||||
@click.option("--tag", is_flag=True, help="Tag the packages after publishing")
|
||||
@click.option("--dry-run", is_flag=True, help="Dry run the publish")
|
||||
@click.option("--js/--no-js", default=True, help="Publish the js package")
|
||||
@click.option("--py/--no-py", default=True, help="Publish the py package")
|
||||
def publish(tag: bool, dry_run: bool, js: bool, py: bool) -> None:
|
||||
"""Publish all packages."""
|
||||
# move to the root
|
||||
os.chdir(Path(__file__).parent.parent)
|
||||
|
||||
if js:
|
||||
if not os.getenv("NPM_TOKEN"):
|
||||
click.echo("NPM_TOKEN is not set, skipping publish", err=True)
|
||||
raise click.Abort("No token set")
|
||||
if py:
|
||||
if not os.getenv("LLAMA_PARSE_PYPI_TOKEN"):
|
||||
click.echo("LLAMA_PARSE_PYPI_TOKEN is not set, skipping publish", err=True)
|
||||
raise click.Abort("No token set")
|
||||
|
||||
# not general script. Just checks each of the 2 packages to see if they need to be published.
|
||||
if js:
|
||||
maybe_publish_npm(dry_run)
|
||||
if py:
|
||||
maybe_publish_pypi(dry_run)
|
||||
|
||||
if tag:
|
||||
if dry_run:
|
||||
click.echo("Dry run, skipping tag. Would run:")
|
||||
click.echo(" npx @changesets/cli tag")
|
||||
click.echo(" git push --tags")
|
||||
else:
|
||||
# Let changesets create JS-related tags as usual
|
||||
_run_command(["npx", "@changesets/cli", "tag"])
|
||||
_run_command(["git", "push", "--tags"])
|
||||
|
||||
|
||||
def maybe_publish_npm(dry_run: bool) -> None:
|
||||
"""Publish the ts package if it needs to be published."""
|
||||
target_dir = Path("ts/llama_cloud_services")
|
||||
ts_path_package = target_dir / "package.json"
|
||||
package_json = json.loads(ts_path_package.read_text())
|
||||
version = package_json["version"]
|
||||
|
||||
# Check if this version is already published on npm
|
||||
result = subprocess.run(
|
||||
["npm", "view", "llama-cloud-services", "versions", "--json"],
|
||||
check=True,
|
||||
capture_output=True,
|
||||
text=True,
|
||||
cwd=target_dir,
|
||||
)
|
||||
|
||||
published_versions = json.loads(result.stdout)
|
||||
if version in published_versions:
|
||||
click.echo(
|
||||
f"npm package llama-cloud-services@{version} already published, skipping"
|
||||
)
|
||||
return
|
||||
click.echo(f"Publishing npm package llama-cloud-services@{version}")
|
||||
# defer to the package.json publish script
|
||||
if dry_run:
|
||||
click.echo("Dry run, skipping publish. Would run:")
|
||||
click.echo(" pnpm run publish")
|
||||
return
|
||||
else:
|
||||
_run_command(["pnpm", "run", "build"], cwd=target_dir)
|
||||
_run_command(["pnpm", "publish"], cwd=target_dir)
|
||||
|
||||
|
||||
def maybe_publish_pypi(dry_run: bool) -> None:
|
||||
"""Publish the py packages if they need to be published."""
|
||||
for pyproject in list(Path("py").glob("*/pyproject.toml")) + [
|
||||
Path("py/pyproject.toml")
|
||||
]:
|
||||
name, version = current_version(pyproject)
|
||||
if is_published(name, version):
|
||||
click.echo(f"PyPI package {name}@{version} already published, skipping")
|
||||
continue
|
||||
click.echo(f"Publishing PyPI package {name}@{version}")
|
||||
|
||||
# Use different tokens for different packages
|
||||
env = os.environ.copy()
|
||||
token = os.environ["LLAMA_PARSE_PYPI_TOKEN"]
|
||||
env["UV_PUBLISH_TOKEN"] = token
|
||||
if dry_run:
|
||||
summary = (token[:3] + "***") if len(token) <= 6 else token[:6] + "****"
|
||||
click.echo(
|
||||
f"Dry run, skipping publish. Would run with publish token {summary}:"
|
||||
)
|
||||
click.echo(" uv build")
|
||||
click.echo(" uv publish")
|
||||
else:
|
||||
_run_command(["uv", "build"], cwd=pyproject.parent)
|
||||
_run_command(["uv", "publish"], cwd=pyproject.parent, env=env)
|
||||
|
||||
|
||||
def current_version(pyproject: Path) -> tuple[str, str]:
|
||||
"""Return (package_name, version_str) taken from the given pyproject.toml."""
|
||||
doc = tomlkit.parse(pyproject.read_text())
|
||||
name = doc["project"]["name"]
|
||||
version = str(Version(doc["project"]["version"])) # normalise
|
||||
return name, version
|
||||
|
||||
|
||||
def is_published(
|
||||
name: str, version: str, index_url: str = "https://pypi.org/pypi"
|
||||
) -> bool:
|
||||
"""
|
||||
True → `<name>==<version>` exists on the given index
|
||||
False → package missing *or* version missing
|
||||
"""
|
||||
url = f"{index_url.rstrip('/')}/{name}/json"
|
||||
try:
|
||||
data = json.load(urllib.request.urlopen(url))
|
||||
except urllib.error.HTTPError as e: # 404 → package not published at all
|
||||
if e.code == 404:
|
||||
return False
|
||||
raise # any other error should surface
|
||||
return version in data["releases"] # keys are version strings
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
cli()
|
||||
@@ -1,226 +0,0 @@
|
||||
#!/usr/bin/env -S uv run --script
|
||||
# /// script
|
||||
# dependencies = ["click", "tomlkit"]
|
||||
# ///
|
||||
|
||||
import click
|
||||
import subprocess
|
||||
import sys
|
||||
import tomlkit
|
||||
from pathlib import Path
|
||||
import json
|
||||
|
||||
|
||||
def get_current_versions() -> tuple[str, str, str, str | None]:
|
||||
"""Get current versions from both pyproject.toml files and TS package.json."""
|
||||
# Read main pyproject.toml
|
||||
main_content = Path("py/pyproject.toml").read_text()
|
||||
main_doc = tomlkit.parse(main_content)
|
||||
main_version = main_doc["project"]["version"]
|
||||
|
||||
# Read llama_parse/pyproject.toml
|
||||
llama_parse_content = Path("py/llama_parse/pyproject.toml").read_text()
|
||||
llama_parse_doc = tomlkit.parse(llama_parse_content)
|
||||
llama_parse_version = llama_parse_doc["project"]["version"]
|
||||
# Find llama-cloud-services dependency in the dependencies list
|
||||
dependency_version = None
|
||||
for dep in llama_parse_doc["project"]["dependencies"]:
|
||||
if isinstance(dep, str) and dep.startswith("llama-cloud-services"):
|
||||
dependency_version = (
|
||||
dep.split("==")[1]
|
||||
if "==" in dep
|
||||
else dep.split(">=")[1]
|
||||
if ">=" in dep
|
||||
else None
|
||||
)
|
||||
break
|
||||
|
||||
# Read TypeScript package.json version via helper
|
||||
ts_version: str = get_ts_version()
|
||||
|
||||
return (
|
||||
str(main_version),
|
||||
str(llama_parse_version),
|
||||
str(dependency_version),
|
||||
str(ts_version) if ts_version is not None else None,
|
||||
)
|
||||
|
||||
|
||||
def validate_versions(
|
||||
main_version: str,
|
||||
llama_parse_version: str,
|
||||
dependency_version: str,
|
||||
) -> list[str]:
|
||||
"""Validate that versions are consistent and return warnings."""
|
||||
warnings = []
|
||||
|
||||
if main_version != llama_parse_version:
|
||||
warnings.append(
|
||||
f"Version mismatch: main={main_version}, llama_parse={llama_parse_version}"
|
||||
)
|
||||
|
||||
# Extract version from dependency string (e.g., ">=0.6.51" -> "0.6.51")
|
||||
if dependency_version and dependency_version.startswith(">="):
|
||||
dep_ver = dependency_version[2:]
|
||||
if dep_ver != main_version:
|
||||
warnings.append(
|
||||
f"Dependency version mismatch: dependency={dep_ver}, main={main_version}"
|
||||
)
|
||||
|
||||
return warnings
|
||||
|
||||
|
||||
def set_version(version: str) -> None:
|
||||
"""Set version across Python projects (no TS change)."""
|
||||
# Update main pyproject.toml
|
||||
main_content = Path("py/pyproject.toml").read_text()
|
||||
main_doc = tomlkit.parse(main_content)
|
||||
main_doc["project"]["version"] = version
|
||||
Path("py/pyproject.toml").write_text(tomlkit.dumps(main_doc))
|
||||
|
||||
# Update llama_parse/pyproject.toml
|
||||
llama_parse_content = Path("py/llama_parse/pyproject.toml").read_text()
|
||||
llama_parse_doc = tomlkit.parse(llama_parse_content)
|
||||
llama_parse_doc["project"]["version"] = version
|
||||
for dep_index, dep in enumerate(llama_parse_doc["project"]["dependencies"]):
|
||||
if isinstance(dep, str) and dep.startswith("llama-cloud-services"):
|
||||
llama_parse_doc["project"]["dependencies"][
|
||||
dep_index
|
||||
] = f"llama-cloud-services>={version}"
|
||||
break
|
||||
Path("py/llama_parse/pyproject.toml").write_text(tomlkit.dumps(llama_parse_doc))
|
||||
|
||||
click.echo(f"Updated Python versions to {version}")
|
||||
|
||||
|
||||
def get_ts_version() -> str:
|
||||
"""Read TypeScript package.json version (if present)."""
|
||||
ts_package_path = Path("ts/llama_cloud_services/package.json")
|
||||
package_data = json.loads(ts_package_path.read_text())
|
||||
data = package_data.get("version")
|
||||
if data is None:
|
||||
raise RuntimeError("TypeScript package.json version not found")
|
||||
return data
|
||||
|
||||
|
||||
def set_ts_version(version: str) -> None:
|
||||
"""Set TypeScript package.json version only."""
|
||||
ts_package_path = Path("ts/llama_cloud_services/package.json")
|
||||
package_data = json.loads(ts_package_path.read_text())
|
||||
package_data["version"] = version
|
||||
ts_package_path.write_text(json.dumps(package_data, indent=2) + "\n")
|
||||
click.echo(f"Updated TypeScript package.json version to {version}")
|
||||
|
||||
|
||||
def get_current_branch() -> str:
|
||||
"""Get the current git branch."""
|
||||
result = subprocess.run(
|
||||
["git", "branch", "--show-current"], capture_output=True, text=True, check=True
|
||||
)
|
||||
return result.stdout.strip()
|
||||
|
||||
|
||||
def create_if_not_exists(version: str) -> str:
|
||||
"""Create a git tag and push it."""
|
||||
current_branch = get_current_branch()
|
||||
if current_branch != "main":
|
||||
click.echo(
|
||||
f"Error: Not on main branch (currently on {current_branch})", err=True
|
||||
)
|
||||
sys.exit(1)
|
||||
|
||||
tag_name = f"v{version}" if version[0].isdigit() else version
|
||||
if not tag_exists(tag_name):
|
||||
# Create tag
|
||||
subprocess.run(["git", "tag", tag_name], check=True)
|
||||
click.echo(f"Created tag {tag_name}")
|
||||
else:
|
||||
click.echo(f"Tag {tag_name} already exists")
|
||||
return tag_name
|
||||
|
||||
|
||||
def tag_exists(tag_name: str) -> bool:
|
||||
"""Check if a git tag exists."""
|
||||
result = subprocess.run(
|
||||
["git", "tag", "-l", tag_name], capture_output=True, text=True, check=True
|
||||
)
|
||||
return tag_name in result.stdout.strip()
|
||||
|
||||
|
||||
def push_tag(tag_name: str) -> None:
|
||||
"""Push a git tag."""
|
||||
subprocess.run(["git", "push", "origin", tag_name], check=True)
|
||||
click.echo(f"Pushed tag {tag_name}")
|
||||
|
||||
|
||||
@click.group()
|
||||
def cli() -> None:
|
||||
"""Version management for llama-cloud-services."""
|
||||
pass
|
||||
|
||||
|
||||
@cli.command()
|
||||
def get() -> None:
|
||||
"""Get current versions and show validation warnings."""
|
||||
(
|
||||
main_version,
|
||||
llama_parse_version,
|
||||
dependency_version,
|
||||
ts_version,
|
||||
) = get_current_versions()
|
||||
|
||||
click.echo("Current versions:")
|
||||
click.echo(f" llama-cloud-services: {main_version}")
|
||||
click.echo(f" llama-parse: {llama_parse_version}")
|
||||
click.echo(f" dependency reference: {dependency_version}")
|
||||
click.echo(f" typescript package: {ts_version}")
|
||||
|
||||
warnings = validate_versions(main_version, llama_parse_version, dependency_version)
|
||||
if warnings:
|
||||
click.echo("\nValidation warnings:")
|
||||
for warning in warnings:
|
||||
click.echo(f" ⚠️ {warning}")
|
||||
else:
|
||||
click.echo("\n✅ All versions are consistent")
|
||||
|
||||
|
||||
@cli.command()
|
||||
@click.argument("version")
|
||||
@click.option("--js", is_flag=True, help="Update TypeScript package.json only")
|
||||
def set(version: str, js: bool) -> None:
|
||||
"""Set version for Python, TypeScript, or both (default: Python only)."""
|
||||
|
||||
if js:
|
||||
set_ts_version(version)
|
||||
return
|
||||
else:
|
||||
set_version(version)
|
||||
|
||||
|
||||
@cli.command()
|
||||
@click.option(
|
||||
"--version", help="Version to tag (uses current version if not specified)"
|
||||
)
|
||||
@click.option(
|
||||
"--push",
|
||||
is_flag=True,
|
||||
help="Push the tag to the remote repository",
|
||||
)
|
||||
@click.option(
|
||||
"--js",
|
||||
is_flag=True,
|
||||
help="tag TypeScript package.json only",
|
||||
)
|
||||
def tag(version: str | None = None, push: bool = False, js: bool = False) -> None:
|
||||
"""Create and push a git tag for the current version."""
|
||||
if not version:
|
||||
main_version, _, _, js_version = get_current_versions()
|
||||
version = f"llama-cloud-services@{js_version}" if js else main_version
|
||||
|
||||
tag_name = create_if_not_exists(version)
|
||||
if push:
|
||||
push_tag(tag_name)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
cli()
|
||||
@@ -9,10 +9,12 @@ test("LlamaIndex module resolution test", async (t) => {
|
||||
const index = new LlamaCloudIndex({
|
||||
name: "test-index",
|
||||
projectName: "Default",
|
||||
apiKey: process.env.LLAMA_CLOUD_API_KEY || "test-key",
|
||||
});
|
||||
const reader = new LlamaParseReader({
|
||||
resultType: "markdown",
|
||||
verbose: false,
|
||||
apiKey: process.env.LLAMA_CLOUD_API_KEY || "test-key",
|
||||
});
|
||||
ok(index !== undefined);
|
||||
ok(reader !== undefined);
|
||||
@@ -24,6 +26,7 @@ test("LlamaIndex module resolution test", async (t) => {
|
||||
const index = new mod.LlamaCloudIndex({
|
||||
name: "test-index",
|
||||
projectName: "Default",
|
||||
apiKey: process.env.LLAMA_CLOUD_API_KEY || "test-key",
|
||||
});
|
||||
ok(index !== undefined);
|
||||
});
|
||||
|
||||
@@ -1,5 +1,23 @@
|
||||
# llama-cloud-services
|
||||
|
||||
## 0.3.9
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 5d4cabd: Add ImageNode support in TypeScript
|
||||
|
||||
## 0.3.8
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- 6e0f2f4: Agent data extraction citations can be undefined
|
||||
|
||||
## 0.3.7
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- d028397: Update llama-cloud api version, and integrate with agent data deletion
|
||||
|
||||
## v0.1.0
|
||||
|
||||
First release for `llama-cloud-services`.
|
||||
|
||||
+1714
-1270
File diff suppressed because it is too large
Load Diff
@@ -1,9 +1,10 @@
|
||||
{
|
||||
"name": "llama-cloud-services",
|
||||
"version": "0.3.6",
|
||||
"version": "0.3.9",
|
||||
"type": "module",
|
||||
"license": "MIT",
|
||||
"scripts": {
|
||||
"get-openapi": "node ./scripts/get-openapi.js",
|
||||
"generate": "./node_modules/.bin/openapi-ts",
|
||||
"build": "pnpm run generate && bunchee",
|
||||
"dev": "bunchee --watch",
|
||||
@@ -13,7 +14,8 @@
|
||||
"test": "vitest run --testTimeout=60000",
|
||||
"test:watch": "vitest --watch",
|
||||
"test:ui": "vitest --ui",
|
||||
"test:coverage": "vitest --coverage"
|
||||
"test:coverage": "vitest --coverage",
|
||||
"release": "pnpm run build && pnpm publish"
|
||||
},
|
||||
"files": [
|
||||
"openapi.json",
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
import fs from 'fs';
|
||||
|
||||
async function downloadOpenApiSpec() {
|
||||
try {
|
||||
const response = await fetch('https://api.cloud.llamaindex.ai/api/openapi.json');
|
||||
|
||||
if (!response.ok) {
|
||||
throw new Error(`HTTP error! status: ${response.status}`);
|
||||
}
|
||||
|
||||
const data = await response.json();
|
||||
|
||||
fs.writeFileSync('openapi.json', JSON.stringify(data, null, 2));
|
||||
console.log('Successfully downloaded openapi.json');
|
||||
} catch (error) {
|
||||
console.error('Error downloading OpenAPI spec:', error);
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
downloadOpenApiSpec();
|
||||
@@ -9,10 +9,16 @@ import { DEFAULT_PROJECT_NAME } from "@llamaindex/core/global";
|
||||
import type { QueryBundle } from "@llamaindex/core/query-engine";
|
||||
import { BaseRetriever } from "@llamaindex/core/retriever";
|
||||
import type { NodeWithScore } from "@llamaindex/core/schema";
|
||||
import { jsonToNode, ObjectType } from "@llamaindex/core/schema";
|
||||
import { jsonToNode, ObjectType, ImageNode } from "@llamaindex/core/schema";
|
||||
import { extractText } from "@llamaindex/core/utils";
|
||||
import type { ClientParams, CloudConstructorParams } from "./type.js";
|
||||
import { getPipelineId, initService } from "./utils.js";
|
||||
import { getPipelineId, getProjectId, initService } from "./utils.js";
|
||||
import {
|
||||
type PageScreenshotNodeWithScore,
|
||||
type PageFigureNodeWithScore,
|
||||
generateFilePageScreenshotPresignedUrlApiV1FilesIdPageScreenshotsPageIndexPresignedUrlPost,
|
||||
generateFilePageFigurePresignedUrlApiV1FilesIdPageFiguresPageIndexFigureNamePresignedUrlPost,
|
||||
} from "./api";
|
||||
|
||||
export type CloudRetrieveParams = Omit<
|
||||
RetrievalParams,
|
||||
@@ -43,6 +49,95 @@ export class LlamaCloudRetriever extends BaseRetriever {
|
||||
});
|
||||
}
|
||||
|
||||
private async fetchBase64FromPresignedUrl(url: string): Promise<string> {
|
||||
const response = await fetch(url);
|
||||
if (!response.ok) {
|
||||
throw new Error(
|
||||
`Failed to fetch media from presigned URL: ${response.status} ${response.statusText}`,
|
||||
);
|
||||
}
|
||||
const buffer = Buffer.from(await response.arrayBuffer());
|
||||
return buffer.toString("base64");
|
||||
}
|
||||
|
||||
private async pageScreenshotNodesToNodeWithScore(
|
||||
nodes: PageScreenshotNodeWithScore[] | undefined,
|
||||
projectId: string,
|
||||
): Promise<NodeWithScore[]> {
|
||||
if (!nodes || nodes.length === 0) return [];
|
||||
|
||||
const results = await Promise.all(
|
||||
nodes.map(async (n) => {
|
||||
const { data: presigned } =
|
||||
await generateFilePageScreenshotPresignedUrlApiV1FilesIdPageScreenshotsPageIndexPresignedUrlPost(
|
||||
{
|
||||
throwOnError: true,
|
||||
path: {
|
||||
id: n.node.file_id,
|
||||
page_index: n.node.page_index,
|
||||
},
|
||||
query: {
|
||||
project_id: projectId,
|
||||
organization_id: this.organizationId ?? null,
|
||||
},
|
||||
},
|
||||
);
|
||||
const base64 = await this.fetchBase64FromPresignedUrl(presigned.url);
|
||||
const imageNode = new ImageNode({
|
||||
image: base64,
|
||||
metadata: {
|
||||
...(n.node.metadata ?? {}),
|
||||
file_id: n.node.file_id,
|
||||
page_index: n.node.page_index,
|
||||
},
|
||||
});
|
||||
return { node: imageNode, score: n.score } satisfies NodeWithScore;
|
||||
}),
|
||||
);
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
private async pageFigureNodesToNodeWithScore(
|
||||
nodes: PageFigureNodeWithScore[] | undefined,
|
||||
projectId: string,
|
||||
): Promise<NodeWithScore[]> {
|
||||
if (!nodes || nodes.length === 0) return [];
|
||||
|
||||
const results = await Promise.all(
|
||||
nodes.map(async (n) => {
|
||||
const { data: presigned } =
|
||||
await generateFilePageFigurePresignedUrlApiV1FilesIdPageFiguresPageIndexFigureNamePresignedUrlPost(
|
||||
{
|
||||
throwOnError: true,
|
||||
path: {
|
||||
id: n.node.file_id,
|
||||
page_index: n.node.page_index,
|
||||
figure_name: n.node.figure_name,
|
||||
},
|
||||
query: {
|
||||
project_id: projectId,
|
||||
organization_id: this.organizationId ?? null,
|
||||
},
|
||||
},
|
||||
);
|
||||
const base64 = await this.fetchBase64FromPresignedUrl(presigned.url);
|
||||
const imageNode = new ImageNode({
|
||||
image: base64,
|
||||
metadata: {
|
||||
...(n.node.metadata ?? {}),
|
||||
file_id: n.node.file_id,
|
||||
page_index: n.node.page_index,
|
||||
figure_name: n.node.figure_name,
|
||||
},
|
||||
});
|
||||
return { node: imageNode, score: n.score } satisfies NodeWithScore;
|
||||
}),
|
||||
);
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
// LlamaCloud expects null values for filters, but LlamaIndexTS uses undefined for empty values
|
||||
// This function converts the undefined values to null
|
||||
private convertFilter(filters?: MetadataFilters): MetadataFilters | null {
|
||||
@@ -76,6 +171,35 @@ export class LlamaCloudRetriever extends BaseRetriever {
|
||||
}
|
||||
|
||||
async _retrieve(query: QueryBundle): Promise<NodeWithScore[]> {
|
||||
// Handle deprecated image retrieval flag
|
||||
const retrieveImageNodes = (this.retrieveParams as RetrievalParams)
|
||||
.retrieve_image_nodes;
|
||||
if (typeof retrieveImageNodes !== "undefined") {
|
||||
console.warn(
|
||||
"The `retrieve_image_nodes` parameter is deprecated. Use `retrieve_page_screenshot_nodes` and `retrieve_page_figure_nodes` instead.",
|
||||
);
|
||||
}
|
||||
|
||||
const retrievePageScreenshotNodes = (this.retrieveParams as RetrievalParams)
|
||||
.retrieve_page_screenshot_nodes;
|
||||
const retrievePageFigureNodes = (this.retrieveParams as RetrievalParams)
|
||||
.retrieve_page_figure_nodes;
|
||||
|
||||
if (retrieveImageNodes) {
|
||||
if (
|
||||
retrievePageScreenshotNodes === false ||
|
||||
retrievePageFigureNodes === false
|
||||
) {
|
||||
throw new Error(
|
||||
"If `retrieve_image_nodes` is set to true, both `retrieve_page_screenshot_nodes` and `retrieve_page_figure_nodes` must also be set to true or omitted.",
|
||||
);
|
||||
}
|
||||
(this.retrieveParams as RetrievalParams).retrieve_page_screenshot_nodes =
|
||||
true;
|
||||
(this.retrieveParams as RetrievalParams).retrieve_page_figure_nodes =
|
||||
true;
|
||||
}
|
||||
|
||||
const pipelineId = await getPipelineId(
|
||||
this.pipelineName,
|
||||
this.projectName,
|
||||
@@ -98,6 +222,34 @@ export class LlamaCloudRetriever extends BaseRetriever {
|
||||
},
|
||||
});
|
||||
|
||||
return this.resultNodesToNodeWithScore(results.retrieval_nodes);
|
||||
const textNodes = this.resultNodesToNodeWithScore(results.retrieval_nodes);
|
||||
|
||||
const needScreenshots = (this.retrieveParams as RetrievalParams)
|
||||
.retrieve_page_screenshot_nodes;
|
||||
const needFigures = (this.retrieveParams as RetrievalParams)
|
||||
.retrieve_page_figure_nodes;
|
||||
|
||||
if (!needScreenshots && !needFigures) {
|
||||
return textNodes;
|
||||
}
|
||||
|
||||
const projectId = await getProjectId(this.projectName, this.organizationId);
|
||||
|
||||
const [screenshotNodes, figureNodes] = await Promise.all([
|
||||
needScreenshots
|
||||
? this.pageScreenshotNodesToNodeWithScore(
|
||||
results.image_nodes,
|
||||
projectId,
|
||||
)
|
||||
: Promise.resolve([] as NodeWithScore[]),
|
||||
needFigures
|
||||
? this.pageFigureNodesToNodeWithScore(
|
||||
results.page_figure_nodes,
|
||||
projectId,
|
||||
)
|
||||
: Promise.resolve([] as NodeWithScore[]),
|
||||
]);
|
||||
|
||||
return [...textNodes, ...screenshotNodes, ...figureNodes];
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,7 @@ import {
|
||||
aggregateAgentDataApiV1BetaAgentDataAggregatePost,
|
||||
createAgentDataApiV1BetaAgentDataPost,
|
||||
deleteAgentDataApiV1BetaAgentDataItemIdDelete,
|
||||
deleteAgentDataByQueryApiV1BetaAgentDataDeletePost,
|
||||
getAgentDataApiV1BetaAgentDataItemIdGet,
|
||||
searchAgentDataApiV1BetaAgentDataSearchPost,
|
||||
updateAgentDataApiV1BetaAgentDataItemIdPut,
|
||||
@@ -12,6 +13,7 @@ import {
|
||||
} from "../../client";
|
||||
import type {
|
||||
AggregateAgentDataOptions,
|
||||
DeleteAgentDataOptions,
|
||||
SearchAgentDataOptions,
|
||||
TypedAgentData,
|
||||
TypedAgentDataItems,
|
||||
@@ -112,6 +114,24 @@ export class AgentClient<T = unknown> {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Delete all matching agent data, returns the total number of deleted items
|
||||
*/
|
||||
async delete(options: DeleteAgentDataOptions): Promise<number> {
|
||||
const response = await deleteAgentDataByQueryApiV1BetaAgentDataDeletePost({
|
||||
throwOnError: true,
|
||||
body: {
|
||||
deployment_name: this.deploymentName,
|
||||
...(this.collection !== undefined && {
|
||||
collection: this.collection,
|
||||
}),
|
||||
...(options.filter !== undefined && { filter: options.filter }),
|
||||
},
|
||||
client: this.client,
|
||||
});
|
||||
return response.data.deleted_count;
|
||||
}
|
||||
|
||||
/**
|
||||
* Search agent data
|
||||
*/
|
||||
|
||||
@@ -38,7 +38,7 @@ export interface ExtractedFieldMetadata {
|
||||
confidence?: number;
|
||||
/** The confidence score for the field based on the extracted text only */
|
||||
extraction_confidence?: number;
|
||||
citation: FieldCitation[];
|
||||
citation?: FieldCitation[];
|
||||
}
|
||||
|
||||
export interface FieldCitation {
|
||||
@@ -127,6 +127,14 @@ export interface SearchAgentDataOptions {
|
||||
includeTotal?: boolean;
|
||||
}
|
||||
|
||||
/**
|
||||
* Options for deleting agent data
|
||||
*/
|
||||
export interface DeleteAgentDataOptions {
|
||||
/** Filter options for the deletion. */
|
||||
filter?: Record<string, FilterOperation>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Options for aggregating agent data
|
||||
*/
|
||||
|
||||
@@ -1530,27 +1530,6 @@ export const Body_run_job_on_file_api_v1_extraction_jobs_file_postSchema = {
|
||||
title: "Body_run_job_on_file_api_v1_extraction_jobs_file_post",
|
||||
} as const;
|
||||
|
||||
export const Body_run_job_test_user_api_v1_extraction_jobs_test_postSchema = {
|
||||
properties: {
|
||||
job_create: {
|
||||
$ref: "#/components/schemas/ExtractJobCreate",
|
||||
},
|
||||
extract_settings: {
|
||||
anyOf: [
|
||||
{
|
||||
$ref: "#/components/schemas/LlamaExtractSettings",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: ["job_create"],
|
||||
title: "Body_run_job_test_user_api_v1_extraction_jobs_test_post",
|
||||
} as const;
|
||||
|
||||
export const Body_screenshot_api_parsing_screenshot_postSchema = {
|
||||
properties: {
|
||||
file: {
|
||||
@@ -2796,30 +2775,6 @@ export const Body_upload_file_api_v1_parsing_upload_postSchema = {
|
||||
title: "Body_upload_file_api_v1_parsing_upload_post",
|
||||
} as const;
|
||||
|
||||
export const Body_upload_file_v2_api_v2alpha1_parse_upload_postSchema = {
|
||||
properties: {
|
||||
configuration: {
|
||||
type: "string",
|
||||
title: "Configuration",
|
||||
},
|
||||
file: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "string",
|
||||
format: "binary",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "File",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: ["configuration"],
|
||||
title: "Body_upload_file_v2_api_v2alpha1_parse_upload_post",
|
||||
} as const;
|
||||
|
||||
export const BoxAuthMechanismSchema = {
|
||||
type: "string",
|
||||
enum: ["developer_token", "ccg"],
|
||||
@@ -3180,12 +3135,6 @@ export const ChatMessageSchema = {
|
||||
title: "ChatMessage",
|
||||
} as const;
|
||||
|
||||
export const ChunkModeSchema = {
|
||||
type: "string",
|
||||
enum: ["PAGE", "DOCUMENT", "SECTION", "GROUPED_PAGES"],
|
||||
title: "ChunkMode",
|
||||
} as const;
|
||||
|
||||
export const ClassificationResultSchema = {
|
||||
properties: {
|
||||
reasoning: {
|
||||
@@ -5486,6 +5435,13 @@ export const CustomClaimsSchema = {
|
||||
description: "Whether the user is allowed to delete organizations.",
|
||||
default: false,
|
||||
},
|
||||
allowed_spreadsheet: {
|
||||
type: "boolean",
|
||||
title: "Allowed Spreadsheet",
|
||||
description:
|
||||
"Whether the user is allowed to access the spreadsheet feature.",
|
||||
default: false,
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
title: "CustomClaims",
|
||||
@@ -6213,6 +6169,54 @@ export const DeleteParamsSchema = {
|
||||
description: "Schema for the parameters of a delete job.",
|
||||
} as const;
|
||||
|
||||
export const DeleteRequestSchema = {
|
||||
properties: {
|
||||
deployment_name: {
|
||||
type: "string",
|
||||
title: "Deployment Name",
|
||||
description: "The agent deployment's name to delete data for",
|
||||
},
|
||||
collection: {
|
||||
type: "string",
|
||||
title: "Collection",
|
||||
description: "The logical agent data collection to delete from",
|
||||
default: "default",
|
||||
},
|
||||
filter: {
|
||||
anyOf: [
|
||||
{
|
||||
additionalProperties: {
|
||||
$ref: "#/components/schemas/FilterOperation",
|
||||
},
|
||||
type: "object",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "Filter",
|
||||
description: "Optional filters to select which items to delete",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: ["deployment_name"],
|
||||
title: "DeleteRequest",
|
||||
description: "API request body for bulk deleting agent data by query",
|
||||
} as const;
|
||||
|
||||
export const DeleteResponseSchema = {
|
||||
properties: {
|
||||
deleted_count: {
|
||||
type: "integer",
|
||||
title: "Deleted Count",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: ["deleted_count"],
|
||||
title: "DeleteResponse",
|
||||
description: "API response for bulk delete operation",
|
||||
} as const;
|
||||
|
||||
export const DirectRetrievalParamsSchema = {
|
||||
properties: {
|
||||
mode: {
|
||||
@@ -6946,6 +6950,20 @@ export const ExtractConfigSchema = {
|
||||
description: "Whether to invalidate the cache for the extraction.",
|
||||
default: false,
|
||||
},
|
||||
num_pages_context: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "integer",
|
||||
minimum: 1,
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "Num Pages Context",
|
||||
description:
|
||||
"Number of pages to pass as context on long document extraction.",
|
||||
},
|
||||
page_range: {
|
||||
anyOf: [
|
||||
{
|
||||
@@ -7202,6 +7220,7 @@ export const ExtractModelsSchema = {
|
||||
"openai-gpt-5-mini",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.5-flash-lite",
|
||||
"gemini-2.5-pro",
|
||||
"openai-gpt-4o",
|
||||
"openai-gpt-4o-mini",
|
||||
@@ -7849,6 +7868,52 @@ export const ExtractTargetSchema = {
|
||||
title: "ExtractTarget",
|
||||
} as const;
|
||||
|
||||
export const ExtractedTableSchema = {
|
||||
properties: {
|
||||
table_id: {
|
||||
type: "integer",
|
||||
title: "Table Id",
|
||||
description: "Unique identifier for this table within the file",
|
||||
},
|
||||
sheet_name: {
|
||||
type: "string",
|
||||
title: "Sheet Name",
|
||||
description: "Worksheet name where table was found",
|
||||
},
|
||||
row_span: {
|
||||
type: "integer",
|
||||
title: "Row Span",
|
||||
description: "Number of rows in the table",
|
||||
},
|
||||
col_span: {
|
||||
type: "integer",
|
||||
title: "Col Span",
|
||||
description: "Number of columns in the table",
|
||||
},
|
||||
has_headers: {
|
||||
type: "boolean",
|
||||
title: "Has Headers",
|
||||
description: "Whether the table has header rows",
|
||||
},
|
||||
metadata_json: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "string",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "Metadata Json",
|
||||
description: "JSON metadata with detailed table information",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: ["table_id", "sheet_name", "row_span", "col_span", "has_headers"],
|
||||
title: "ExtractedTable",
|
||||
description: "A single extracted table from a spreadsheet",
|
||||
} as const;
|
||||
|
||||
export const FailPageModeSchema = {
|
||||
type: "string",
|
||||
enum: ["raw_text", "blank_page", "error_message"],
|
||||
@@ -10828,140 +10893,6 @@ export const LegacyParseJobConfigSchema = {
|
||||
description: "Configuration for llamaparse job",
|
||||
} as const;
|
||||
|
||||
export const LlamaExtractSettingsSchema = {
|
||||
properties: {
|
||||
max_file_size: {
|
||||
type: "integer",
|
||||
title: "Max File Size",
|
||||
description: "The maximum file size (in bytes) allowed for the document.",
|
||||
default: 104857600,
|
||||
},
|
||||
max_file_size_ui: {
|
||||
type: "integer",
|
||||
title: "Max File Size Ui",
|
||||
description: "The maximum file size (in bytes) allowed for the document.",
|
||||
default: 31457280,
|
||||
},
|
||||
max_pages: {
|
||||
type: "integer",
|
||||
title: "Max Pages",
|
||||
description: "The maximum number of pages allowed for the document.",
|
||||
default: 500,
|
||||
},
|
||||
chunk_mode: {
|
||||
$ref: "#/components/schemas/ChunkMode",
|
||||
description: "The mode to use for chunking the document.",
|
||||
default: "SECTION",
|
||||
},
|
||||
max_chunk_size: {
|
||||
type: "integer",
|
||||
title: "Max Chunk Size",
|
||||
description:
|
||||
"The maximum size of the chunks (in tokens) to use for chunking the document.",
|
||||
default: 10000,
|
||||
},
|
||||
extraction_agent_config: {
|
||||
additionalProperties: {
|
||||
$ref: "#/components/schemas/StructParseConf",
|
||||
},
|
||||
type: "object",
|
||||
title: "Extraction Agent Config",
|
||||
description: "The configuration for the extraction agent.",
|
||||
},
|
||||
use_multimodal_parsing: {
|
||||
type: "boolean",
|
||||
title: "Use Multimodal Parsing",
|
||||
description: "Whether to use experimental multimodal parsing.",
|
||||
default: false,
|
||||
},
|
||||
use_pixel_extraction: {
|
||||
type: "boolean",
|
||||
title: "Use Pixel Extraction",
|
||||
description:
|
||||
"DEPRECATED: Whether to use extraction over pixels for multimodal mode.",
|
||||
default: false,
|
||||
},
|
||||
llama_parse_params: {
|
||||
$ref: "#/components/schemas/LlamaParseParameters",
|
||||
description: "LlamaParse related settings.",
|
||||
default: {
|
||||
languages: ["en"],
|
||||
parsing_instruction: "",
|
||||
disable_ocr: false,
|
||||
annotate_links: true,
|
||||
adaptive_long_table: true,
|
||||
compact_markdown_table: false,
|
||||
disable_reconstruction: false,
|
||||
disable_image_extraction: false,
|
||||
invalidate_cache: false,
|
||||
outlined_table_extraction: true,
|
||||
merge_tables_across_pages_in_markdown: false,
|
||||
output_pdf_of_document: false,
|
||||
do_not_cache: false,
|
||||
fast_mode: false,
|
||||
skip_diagonal_text: false,
|
||||
preserve_layout_alignment_across_pages: false,
|
||||
preserve_very_small_text: false,
|
||||
gpt4o_mode: false,
|
||||
do_not_unroll_columns: false,
|
||||
extract_layout: false,
|
||||
high_res_ocr: false,
|
||||
html_make_all_elements_visible: false,
|
||||
layout_aware: false,
|
||||
specialized_chart_parsing_agentic: false,
|
||||
specialized_chart_parsing_plus: false,
|
||||
specialized_chart_parsing_efficient: false,
|
||||
specialized_image_parsing: false,
|
||||
precise_bounding_box: false,
|
||||
html_remove_navigation_elements: false,
|
||||
html_remove_fixed_elements: false,
|
||||
guess_xlsx_sheet_name: false,
|
||||
use_vendor_multimodal_model: false,
|
||||
page_prefix: `<<<PAGE:{pageNumber}>>>
|
||||
|
||||
`,
|
||||
page_suffix: `
|
||||
|
||||
<<<END_PAGE>>>`,
|
||||
take_screenshot: false,
|
||||
is_formatting_instruction: true,
|
||||
premium_mode: false,
|
||||
continuous_mode: false,
|
||||
auto_mode: false,
|
||||
auto_mode_trigger_on_table_in_page: false,
|
||||
auto_mode_trigger_on_image_in_page: false,
|
||||
structured_output: false,
|
||||
extract_charts: false,
|
||||
spreadsheet_extract_sub_tables: false,
|
||||
spreadsheet_force_formula_computation: false,
|
||||
inline_images_in_markdown: false,
|
||||
strict_mode_image_extraction: false,
|
||||
strict_mode_image_ocr: false,
|
||||
strict_mode_reconstruction: false,
|
||||
strict_mode_buggy_font: false,
|
||||
save_images: true,
|
||||
hide_headers: false,
|
||||
hide_footers: false,
|
||||
ignore_document_elements_for_layout_detection: false,
|
||||
output_tables_as_HTML: false,
|
||||
internal_is_screenshot_job: false,
|
||||
parse_mode: "parse_page_with_llm",
|
||||
page_error_tolerance: 0.05,
|
||||
replace_failed_page_mode: "raw_text",
|
||||
},
|
||||
},
|
||||
multimodal_parse_resolution: {
|
||||
$ref: "#/components/schemas/MultimodalParseResolution",
|
||||
description: "The resolution to use for multimodal parsing.",
|
||||
default: "medium",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
title: "LlamaExtractSettings",
|
||||
description: `All settings for the extraction agent. Only the settings in ExtractConfig
|
||||
are exposed to the user.`,
|
||||
} as const;
|
||||
|
||||
export const LlamaParseParametersSchema = {
|
||||
properties: {
|
||||
webhook_configurations: {
|
||||
@@ -12602,12 +12533,6 @@ export const MetronomeDashboardTypeSchema = {
|
||||
title: "MetronomeDashboardType",
|
||||
} as const;
|
||||
|
||||
export const MultimodalParseResolutionSchema = {
|
||||
type: "string",
|
||||
enum: ["medium", "high"],
|
||||
title: "MultimodalParseResolution",
|
||||
} as const;
|
||||
|
||||
export const NodeRelationshipSchema = {
|
||||
type: "string",
|
||||
enum: ["1", "2", "3", "4", "5"],
|
||||
@@ -13430,6 +13355,48 @@ export const PaginatedResponse_QuotaConfiguration_Schema = {
|
||||
title: "PaginatedResponse[QuotaConfiguration]",
|
||||
} as const;
|
||||
|
||||
export const PaginatedResponse_SpreadsheetJob_Schema = {
|
||||
properties: {
|
||||
items: {
|
||||
items: {
|
||||
$ref: "#/components/schemas/SpreadsheetJob",
|
||||
},
|
||||
type: "array",
|
||||
title: "Items",
|
||||
description: "The list of items.",
|
||||
},
|
||||
next_page_token: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "string",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "Next Page Token",
|
||||
description:
|
||||
"A token, which can be sent as page_token to retrieve the next page. If this field is omitted, there are no subsequent pages.",
|
||||
},
|
||||
total_size: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "integer",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "Total Size",
|
||||
description:
|
||||
"The total number of items available. This is only populated when specifically requested. The value may be an estimate and can be used for display purposes only.",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: ["items"],
|
||||
title: "PaginatedResponse[SpreadsheetJob]",
|
||||
} as const;
|
||||
|
||||
export const ParseConfigurationSchema = {
|
||||
properties: {
|
||||
id: {
|
||||
@@ -17841,69 +17808,6 @@ export const ProjectUpdateSchema = {
|
||||
description: "Schema for updating a project.",
|
||||
} as const;
|
||||
|
||||
export const PromptConfSchema = {
|
||||
properties: {
|
||||
system_prompt: {
|
||||
type: "string",
|
||||
title: "System Prompt",
|
||||
description: "The system prompt to use for the extraction.",
|
||||
default:
|
||||
"Given a JSON schema, extract the data from the provided SOURCE TEXT according to the schema. Only output information that is explicitly stated or can be inferred from the SOURCE TEXT.",
|
||||
},
|
||||
extraction_prompt: {
|
||||
type: "string",
|
||||
title: "Extraction Prompt",
|
||||
description: "The prompt to use for the extraction.",
|
||||
default: "The extracted data using the given JSON schema.",
|
||||
},
|
||||
error_handling_prompt: {
|
||||
type: "string",
|
||||
title: "Error Handling Prompt",
|
||||
description: "The prompt to use for error handling.",
|
||||
default:
|
||||
"If the source text does not contain enough information to extract the value, explain the reason very briefly. Else, output null and fill out the value__ field.",
|
||||
},
|
||||
reasoning_prompt: {
|
||||
type: "string",
|
||||
title: "Reasoning Prompt",
|
||||
description: "The prompt to use for reasoning.",
|
||||
default: `
|
||||
Provide a brief explanation for how you arrived at the extracted value based on the source text provided.
|
||||
- For inferred values, explain the reasoning behind the extraction briefly.
|
||||
- For simple verbatim extraction, output 'VERBATIM EXTRACTION'.
|
||||
- When supporting data is not present in the source text, output 'INSUFFICIENT DATA' and emit blank or null values for the value__ field.
|
||||
`,
|
||||
},
|
||||
cite_sources_prompt: {
|
||||
additionalProperties: {
|
||||
type: "string",
|
||||
},
|
||||
type: "object",
|
||||
title: "Cite Sources Prompt",
|
||||
description: "The prompt to use for citing sources.",
|
||||
default: {
|
||||
description: `
|
||||
### Citation Rules (read carefully):
|
||||
- You must ANNOTATE every value with the MOST RELEVANT short EXACT substring from the source text that supports it.
|
||||
- For inferred values, cite the text used to infer it in the matching_text field or output 'INFERRED FROM TEXT'
|
||||
- If no support exists, output 'INSUFFICIENT DATA' and leave value__ null or '', 0.0, False etc depending on the type of the field.
|
||||
`,
|
||||
page: "Cite the page number of the source text that the extracted value is from. The page number is the integer that appears right after <<<PAGE:. If no page number is present in this format, use the default value of 1.",
|
||||
matching_text:
|
||||
'Cite the **MOST RELEVANT EXACT TEXT from the SOURCE TEXT** that supports the extracted value within 80 characters. If the exact substring is >80 chars, truncate with ellipsis "...". Provide only the single most relevant citation.',
|
||||
},
|
||||
},
|
||||
scratchpad_prompt: {
|
||||
type: "string",
|
||||
title: "Scratchpad Prompt",
|
||||
description: "The prompt to use for scratchpad.",
|
||||
default: "Use for intermediate step-by-step reasoning. Be concise.",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
title: "PromptConf",
|
||||
} as const;
|
||||
|
||||
export const PublicModelNameSchema = {
|
||||
type: "string",
|
||||
enum: [
|
||||
@@ -17926,6 +17830,7 @@ export const PublicModelNameSchema = {
|
||||
"gemini-2.5-pro",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-2.0-flash-lite",
|
||||
"gemini-2.5-flash-lite",
|
||||
"gemini-1.5-flash",
|
||||
"gemini-1.5-pro",
|
||||
],
|
||||
@@ -18752,12 +18657,6 @@ export const RoleSchema = {
|
||||
description: "Schema for a role.",
|
||||
} as const;
|
||||
|
||||
export const SchemaRelaxModeSchema = {
|
||||
type: "string",
|
||||
enum: ["FULL", "TOP_LEVEL", "LEAF"],
|
||||
title: "SchemaRelaxMode",
|
||||
} as const;
|
||||
|
||||
export const SearchRequestSchema = {
|
||||
properties: {
|
||||
page_size: {
|
||||
@@ -18950,6 +18849,135 @@ BM25: Uses Qdrant's FastEmbed BM25 model for sparse embeddings
|
||||
AUTO: Automatically selects based on deployment mode (BYOC uses term frequency, Cloud uses Splade)`,
|
||||
} as const;
|
||||
|
||||
export const SpreadsheetJobSchema = {
|
||||
properties: {
|
||||
id: {
|
||||
type: "string",
|
||||
title: "Id",
|
||||
description: "The ID of the job",
|
||||
},
|
||||
user_id: {
|
||||
type: "string",
|
||||
title: "User Id",
|
||||
description: "The ID of the user",
|
||||
},
|
||||
project_id: {
|
||||
type: "string",
|
||||
format: "uuid",
|
||||
title: "Project Id",
|
||||
description: "The ID of the project",
|
||||
},
|
||||
file_id: {
|
||||
type: "string",
|
||||
format: "uuid",
|
||||
title: "File Id",
|
||||
description: "The ID of the file to parse",
|
||||
},
|
||||
config: {
|
||||
$ref: "#/components/schemas/SpreadsheetParsingConfig",
|
||||
description: "Configuration for the parsing job",
|
||||
},
|
||||
status: {
|
||||
$ref: "#/components/schemas/StatusEnum",
|
||||
description: "The status of the parsing job",
|
||||
},
|
||||
created_at: {
|
||||
type: "string",
|
||||
title: "Created At",
|
||||
description: "When the job was created",
|
||||
},
|
||||
updated_at: {
|
||||
type: "string",
|
||||
title: "Updated At",
|
||||
description: "When the job was last updated",
|
||||
},
|
||||
success: {
|
||||
anyOf: [
|
||||
{
|
||||
type: "boolean",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "Success",
|
||||
description: "Whether the job completed successfully",
|
||||
},
|
||||
tables: {
|
||||
items: {
|
||||
$ref: "#/components/schemas/ExtractedTable",
|
||||
},
|
||||
type: "array",
|
||||
title: "Tables",
|
||||
description: "All extracted tables (populated when job is complete)",
|
||||
},
|
||||
errors: {
|
||||
items: {
|
||||
type: "string",
|
||||
},
|
||||
type: "array",
|
||||
title: "Errors",
|
||||
description: "Any errors encountered",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: [
|
||||
"id",
|
||||
"user_id",
|
||||
"project_id",
|
||||
"file_id",
|
||||
"config",
|
||||
"status",
|
||||
"created_at",
|
||||
"updated_at",
|
||||
],
|
||||
title: "SpreadsheetJob",
|
||||
description: "A spreadsheet parsing job",
|
||||
} as const;
|
||||
|
||||
export const SpreadsheetJobCreateSchema = {
|
||||
properties: {
|
||||
file_id: {
|
||||
type: "string",
|
||||
format: "uuid",
|
||||
title: "File Id",
|
||||
description: "The ID of the file to parse",
|
||||
},
|
||||
config: {
|
||||
$ref: "#/components/schemas/SpreadsheetParsingConfig",
|
||||
description: "Configuration for the parsing job",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
required: ["file_id"],
|
||||
title: "SpreadsheetJobCreate",
|
||||
description: "Request to create a spreadsheet parsing job",
|
||||
} as const;
|
||||
|
||||
export const SpreadsheetParsingConfigSchema = {
|
||||
properties: {
|
||||
sheet_names: {
|
||||
anyOf: [
|
||||
{
|
||||
items: {
|
||||
type: "string",
|
||||
},
|
||||
type: "array",
|
||||
},
|
||||
{
|
||||
type: "null",
|
||||
},
|
||||
],
|
||||
title: "Sheet Names",
|
||||
description:
|
||||
"The names of the sheets to parse. If empty, all sheets will be parsed.",
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
title: "SpreadsheetParsingConfig",
|
||||
description: "Configuration for spreadsheet parsing",
|
||||
} as const;
|
||||
|
||||
export const StatusEnumSchema = {
|
||||
type: "string",
|
||||
enum: ["PENDING", "SUCCESS", "ERROR", "PARTIAL_SUCCESS", "CANCELLED"],
|
||||
@@ -18957,101 +18985,6 @@ export const StatusEnumSchema = {
|
||||
description: "Enum for representing the status of a job",
|
||||
} as const;
|
||||
|
||||
export const StructModeSchema = {
|
||||
type: "string",
|
||||
enum: [
|
||||
"STRUCT_PARSE",
|
||||
"JSON_MODE",
|
||||
"FUNC_CALL",
|
||||
"STRUCT_RELAXED",
|
||||
"UNSTRUCTURED",
|
||||
],
|
||||
title: "StructMode",
|
||||
} as const;
|
||||
|
||||
export const StructParseConfSchema = {
|
||||
properties: {
|
||||
model: {
|
||||
$ref: "#/components/schemas/ExtractModels",
|
||||
description: "The model to use for the structured parsing.",
|
||||
default: "openai-gpt-4-1",
|
||||
},
|
||||
temperature: {
|
||||
type: "number",
|
||||
title: "Temperature",
|
||||
description: "The temperature to use for the structured parsing.",
|
||||
default: 0,
|
||||
},
|
||||
relaxation_mode: {
|
||||
$ref: "#/components/schemas/SchemaRelaxMode",
|
||||
description: "The relaxation mode to use for the structured parsing.",
|
||||
default: "LEAF",
|
||||
},
|
||||
struct_mode: {
|
||||
$ref: "#/components/schemas/StructMode",
|
||||
description: "The struct mode to use for the structured parsing.",
|
||||
default: "STRUCT_PARSE",
|
||||
},
|
||||
fetch_logprobs: {
|
||||
type: "boolean",
|
||||
title: "Fetch Logprobs",
|
||||
description: "Whether to fetch logprobs for the structured parsing.",
|
||||
default: false,
|
||||
},
|
||||
handle_missing: {
|
||||
type: "boolean",
|
||||
title: "Handle Missing",
|
||||
description: "Whether to handle missing fields in the schema.",
|
||||
default: false,
|
||||
},
|
||||
use_reasoning: {
|
||||
type: "boolean",
|
||||
title: "Use Reasoning",
|
||||
description: "Whether to use reasoning for the structured extraction.",
|
||||
default: false,
|
||||
},
|
||||
cite_sources: {
|
||||
type: "boolean",
|
||||
title: "Cite Sources",
|
||||
description: "Whether to cite sources for the structured extraction.",
|
||||
default: false,
|
||||
},
|
||||
prompt_conf: {
|
||||
$ref: "#/components/schemas/PromptConf",
|
||||
description: "The prompt configuration for the structured parsing.",
|
||||
default: {
|
||||
system_prompt:
|
||||
"Given a JSON schema, extract the data from the provided SOURCE TEXT according to the schema. Only output information that is explicitly stated or can be inferred from the SOURCE TEXT.",
|
||||
extraction_prompt: "The extracted data using the given JSON schema.",
|
||||
error_handling_prompt:
|
||||
"If the source text does not contain enough information to extract the value, explain the reason very briefly. Else, output null and fill out the value__ field.",
|
||||
reasoning_prompt: `
|
||||
Provide a brief explanation for how you arrived at the extracted value based on the source text provided.
|
||||
- For inferred values, explain the reasoning behind the extraction briefly.
|
||||
- For simple verbatim extraction, output 'VERBATIM EXTRACTION'.
|
||||
- When supporting data is not present in the source text, output 'INSUFFICIENT DATA' and emit blank or null values for the value__ field.
|
||||
`,
|
||||
cite_sources_prompt: {
|
||||
description: `
|
||||
### Citation Rules (read carefully):
|
||||
- You must ANNOTATE every value with the MOST RELEVANT short EXACT substring from the source text that supports it.
|
||||
- For inferred values, cite the text used to infer it in the matching_text field or output 'INFERRED FROM TEXT'
|
||||
- If no support exists, output 'INSUFFICIENT DATA' and leave value__ null or '', 0.0, False etc depending on the type of the field.
|
||||
`,
|
||||
matching_text:
|
||||
'Cite the **MOST RELEVANT EXACT TEXT from the SOURCE TEXT** that supports the extracted value within 80 characters. If the exact substring is >80 chars, truncate with ellipsis "...". Provide only the single most relevant citation.',
|
||||
page: "Cite the page number of the source text that the extracted value is from. The page number is the integer that appears right after <<<PAGE:. If no page number is present in this format, use the default value of 1.",
|
||||
},
|
||||
scratchpad_prompt:
|
||||
"Use for intermediate step-by-step reasoning. Be concise.",
|
||||
},
|
||||
},
|
||||
},
|
||||
type: "object",
|
||||
title: "StructParseConf",
|
||||
description: "Configuration for the structured parsing agent.",
|
||||
} as const;
|
||||
|
||||
export const SupportedLLMModelSchema = {
|
||||
properties: {
|
||||
name: {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -691,115 +691,6 @@ export const zBodyRunJobOnFileApiV1ExtractionJobsFilePost = z.object({
|
||||
config_override: z.union([z.string(), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zExtractTarget = z.enum(["PER_DOC", "PER_PAGE"]);
|
||||
|
||||
export const zExtractMode = z.enum([
|
||||
"FAST",
|
||||
"BALANCED",
|
||||
"PREMIUM",
|
||||
"MULTIMODAL",
|
||||
]);
|
||||
|
||||
export const zPublicModelName = z.enum([
|
||||
"openai-gpt-4o",
|
||||
"openai-gpt-4o-mini",
|
||||
"openai-gpt-4-1",
|
||||
"openai-gpt-4-1-mini",
|
||||
"openai-gpt-4-1-nano",
|
||||
"openai-gpt-5",
|
||||
"openai-gpt-5-mini",
|
||||
"openai-gpt-5-nano",
|
||||
"openai-text-embedding-3-small",
|
||||
"openai-text-embedding-3-large",
|
||||
"openai-whisper-1",
|
||||
"anthropic-sonnet-3.5",
|
||||
"anthropic-sonnet-3.5-v2",
|
||||
"anthropic-sonnet-3.7",
|
||||
"anthropic-sonnet-4.0",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-2.0-flash-lite",
|
||||
"gemini-1.5-flash",
|
||||
"gemini-1.5-pro",
|
||||
]);
|
||||
|
||||
export const zExtractModels = z.enum([
|
||||
"openai-gpt-4-1",
|
||||
"openai-gpt-4-1-mini",
|
||||
"openai-gpt-4-1-nano",
|
||||
"openai-gpt-5",
|
||||
"openai-gpt-5-mini",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.5-pro",
|
||||
"openai-gpt-4o",
|
||||
"openai-gpt-4o-mini",
|
||||
]);
|
||||
|
||||
export const zDocumentChunkMode = z.enum(["PAGE", "SECTION"]);
|
||||
|
||||
export const zExtractConfig = z.object({
|
||||
priority: z
|
||||
.union([z.enum(["low", "medium", "high", "critical"]), z.null()])
|
||||
.optional(),
|
||||
extraction_target: zExtractTarget.optional(),
|
||||
extraction_mode: zExtractMode.optional(),
|
||||
parse_model: z.union([zPublicModelName, z.null()]).optional(),
|
||||
extract_model: z.union([zExtractModels, z.null()]).optional(),
|
||||
multimodal_fast_mode: z.boolean().optional().default(false),
|
||||
system_prompt: z.union([z.string(), z.null()]).optional(),
|
||||
use_reasoning: z.boolean().optional().default(false),
|
||||
cite_sources: z.boolean().optional().default(false),
|
||||
confidence_scores: z.boolean().optional().default(false),
|
||||
chunk_mode: zDocumentChunkMode.optional(),
|
||||
high_resolution_mode: z.boolean().optional().default(false),
|
||||
invalidate_cache: z.boolean().optional().default(false),
|
||||
page_range: z.union([z.string(), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zExtractJobCreate = z.object({
|
||||
priority: z
|
||||
.union([z.enum(["low", "medium", "high", "critical"]), z.null()])
|
||||
.optional(),
|
||||
webhook_configurations: z
|
||||
.union([z.array(zWebhookConfiguration), z.null()])
|
||||
.optional(),
|
||||
extraction_agent_id: z.string().uuid(),
|
||||
file_id: z.string().uuid(),
|
||||
data_schema_override: z
|
||||
.union([z.object({}), z.string(), z.null()])
|
||||
.optional(),
|
||||
config_override: z.union([zExtractConfig, z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zChunkMode = z.enum([
|
||||
"PAGE",
|
||||
"DOCUMENT",
|
||||
"SECTION",
|
||||
"GROUPED_PAGES",
|
||||
]);
|
||||
|
||||
export const zMultimodalParseResolution = z.enum(["medium", "high"]);
|
||||
|
||||
export const zLlamaExtractSettings = z.object({
|
||||
max_file_size: z.number().int().optional().default(104857600),
|
||||
max_file_size_ui: z.number().int().optional().default(31457280),
|
||||
max_pages: z.number().int().optional().default(500),
|
||||
chunk_mode: zChunkMode.optional(),
|
||||
max_chunk_size: z.number().int().optional().default(10000),
|
||||
extraction_agent_config: z.object({}).optional(),
|
||||
use_multimodal_parsing: z.boolean().optional().default(false),
|
||||
use_pixel_extraction: z.boolean().optional().default(false),
|
||||
llama_parse_params: zLlamaParseParameters.optional(),
|
||||
multimodal_parse_resolution: zMultimodalParseResolution.optional(),
|
||||
});
|
||||
|
||||
export const zBodyRunJobTestUserApiV1ExtractionJobsTestPost = z.object({
|
||||
job_create: zExtractJobCreate,
|
||||
extract_settings: z.union([zLlamaExtractSettings, z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zBodyScreenshotApiParsingScreenshotPost = z.object({
|
||||
file: z.union([z.string(), z.null()]).optional(),
|
||||
do_not_cache: z.boolean().optional().default(false),
|
||||
@@ -1072,11 +963,6 @@ export const zBodyUploadFileApiV1ParsingUploadPost = z.object({
|
||||
page_footer_suffix: z.string().optional(),
|
||||
});
|
||||
|
||||
export const zBodyUploadFileV2ApiV2Alpha1ParseUploadPost = z.object({
|
||||
configuration: z.string(),
|
||||
file: z.union([z.string(), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zBoxAuthMechanism = z.enum(["developer_token", "ccg"]);
|
||||
|
||||
export const zSupportedLlmModelNames = z.enum([
|
||||
@@ -1700,6 +1586,7 @@ export const zCustomClaims = z.object({
|
||||
allowed_classify: z.boolean().optional().default(true),
|
||||
api_datasource_access: z.boolean().optional().default(false),
|
||||
allow_org_deletion: z.boolean().optional().default(false),
|
||||
allowed_spreadsheet: z.boolean().optional().default(false),
|
||||
});
|
||||
|
||||
export const zCustomerPortalSessionCreatePayload = z.object({
|
||||
@@ -1855,6 +1742,16 @@ export const zDefaultOrganizationUpdate = z.object({
|
||||
organization_id: z.string().uuid(),
|
||||
});
|
||||
|
||||
export const zDeleteRequest = z.object({
|
||||
deployment_name: z.string(),
|
||||
collection: z.string().optional().default("default"),
|
||||
filter: z.union([z.object({}), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zDeleteResponse = z.object({
|
||||
deleted_count: z.number().int(),
|
||||
});
|
||||
|
||||
export const zRetrieverPipeline = z.object({
|
||||
name: z.union([z.string().min(1).max(3000), z.null()]),
|
||||
description: z.union([z.string().max(15000), z.null()]),
|
||||
@@ -1870,6 +1767,8 @@ export const zDirectRetrievalParams = z.object({
|
||||
pipelines: z.array(zRetrieverPipeline).optional(),
|
||||
});
|
||||
|
||||
export const zDocumentChunkMode = z.enum(["PAGE", "SECTION"]);
|
||||
|
||||
export const zDocumentIngestionJobParams = z.object({
|
||||
custom_metadata: z.union([z.object({}), z.null()]).optional(),
|
||||
resource_info: z.union([z.object({}), z.null()]).optional(),
|
||||
@@ -2122,6 +2021,74 @@ Query: {query_str}
|
||||
Answer: `),
|
||||
});
|
||||
|
||||
export const zExtractTarget = z.enum(["PER_DOC", "PER_PAGE"]);
|
||||
|
||||
export const zExtractMode = z.enum([
|
||||
"FAST",
|
||||
"BALANCED",
|
||||
"PREMIUM",
|
||||
"MULTIMODAL",
|
||||
]);
|
||||
|
||||
export const zPublicModelName = z.enum([
|
||||
"openai-gpt-4o",
|
||||
"openai-gpt-4o-mini",
|
||||
"openai-gpt-4-1",
|
||||
"openai-gpt-4-1-mini",
|
||||
"openai-gpt-4-1-nano",
|
||||
"openai-gpt-5",
|
||||
"openai-gpt-5-mini",
|
||||
"openai-gpt-5-nano",
|
||||
"openai-text-embedding-3-small",
|
||||
"openai-text-embedding-3-large",
|
||||
"openai-whisper-1",
|
||||
"anthropic-sonnet-3.5",
|
||||
"anthropic-sonnet-3.5-v2",
|
||||
"anthropic-sonnet-3.7",
|
||||
"anthropic-sonnet-4.0",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.5-pro",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-2.0-flash-lite",
|
||||
"gemini-2.5-flash-lite",
|
||||
"gemini-1.5-flash",
|
||||
"gemini-1.5-pro",
|
||||
]);
|
||||
|
||||
export const zExtractModels = z.enum([
|
||||
"openai-gpt-4-1",
|
||||
"openai-gpt-4-1-mini",
|
||||
"openai-gpt-4-1-nano",
|
||||
"openai-gpt-5",
|
||||
"openai-gpt-5-mini",
|
||||
"gemini-2.0-flash",
|
||||
"gemini-2.5-flash",
|
||||
"gemini-2.5-flash-lite",
|
||||
"gemini-2.5-pro",
|
||||
"openai-gpt-4o",
|
||||
"openai-gpt-4o-mini",
|
||||
]);
|
||||
|
||||
export const zExtractConfig = z.object({
|
||||
priority: z
|
||||
.union([z.enum(["low", "medium", "high", "critical"]), z.null()])
|
||||
.optional(),
|
||||
extraction_target: zExtractTarget.optional(),
|
||||
extraction_mode: zExtractMode.optional(),
|
||||
parse_model: z.union([zPublicModelName, z.null()]).optional(),
|
||||
extract_model: z.union([zExtractModels, z.null()]).optional(),
|
||||
multimodal_fast_mode: z.boolean().optional().default(false),
|
||||
system_prompt: z.union([z.string(), z.null()]).optional(),
|
||||
use_reasoning: z.boolean().optional().default(false),
|
||||
cite_sources: z.boolean().optional().default(false),
|
||||
confidence_scores: z.boolean().optional().default(false),
|
||||
chunk_mode: zDocumentChunkMode.optional(),
|
||||
high_resolution_mode: z.boolean().optional().default(false),
|
||||
invalidate_cache: z.boolean().optional().default(false),
|
||||
num_pages_context: z.union([z.number().int().gte(1), z.null()]).optional(),
|
||||
page_range: z.union([z.string(), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zExtractAgent = z.object({
|
||||
id: z.string().uuid(),
|
||||
name: z.string(),
|
||||
@@ -2167,6 +2134,21 @@ export const zExtractJob = z.object({
|
||||
file: zFile,
|
||||
});
|
||||
|
||||
export const zExtractJobCreate = z.object({
|
||||
priority: z
|
||||
.union([z.enum(["low", "medium", "high", "critical"]), z.null()])
|
||||
.optional(),
|
||||
webhook_configurations: z
|
||||
.union([z.array(zWebhookConfiguration), z.null()])
|
||||
.optional(),
|
||||
extraction_agent_id: z.string().uuid(),
|
||||
file_id: z.string().uuid(),
|
||||
data_schema_override: z
|
||||
.union([z.object({}), z.string(), z.null()])
|
||||
.optional(),
|
||||
config_override: z.union([zExtractConfig, z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zExtractJobCreateBatch = z.object({
|
||||
extraction_agent_id: z.string().uuid(),
|
||||
file_ids: z.array(z.string().uuid()).min(1),
|
||||
@@ -2234,6 +2216,15 @@ export const zExtractStatelessRequest = z.object({
|
||||
file: z.union([zFileData, z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zExtractedTable = z.object({
|
||||
table_id: z.number().int(),
|
||||
sheet_name: z.string(),
|
||||
row_span: z.number().int(),
|
||||
col_span: z.number().int(),
|
||||
has_headers: z.boolean(),
|
||||
metadata_json: z.union([z.string(), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zFileCountByStatusResponse = z.object({
|
||||
counts: z.object({}),
|
||||
total_count: z.number().int(),
|
||||
@@ -2987,6 +2978,30 @@ export const zPaginatedResponseQuotaConfiguration = z.object({
|
||||
items: z.array(zQuotaConfiguration),
|
||||
});
|
||||
|
||||
export const zSpreadsheetParsingConfig = z.object({
|
||||
sheet_names: z.union([z.array(z.string()), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zSpreadsheetJob = z.object({
|
||||
id: z.string(),
|
||||
user_id: z.string(),
|
||||
project_id: z.string().uuid(),
|
||||
file_id: z.string().uuid(),
|
||||
config: zSpreadsheetParsingConfig,
|
||||
status: zStatusEnum,
|
||||
created_at: z.string(),
|
||||
updated_at: z.string(),
|
||||
success: z.union([z.boolean(), z.null()]).optional(),
|
||||
tables: z.array(zExtractedTable).optional(),
|
||||
errors: z.array(z.string()).optional(),
|
||||
});
|
||||
|
||||
export const zPaginatedResponseSpreadsheetJob = z.object({
|
||||
items: z.array(zSpreadsheetJob),
|
||||
next_page_token: z.union([z.string(), z.null()]).optional(),
|
||||
total_size: z.union([z.number().int(), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zParseConfiguration = z.object({
|
||||
id: z.string(),
|
||||
name: z.string(),
|
||||
@@ -3400,49 +3415,6 @@ export const zProjectUpdate = z.object({
|
||||
name: z.string().min(1).max(3000),
|
||||
});
|
||||
|
||||
export const zPromptConf = z.object({
|
||||
system_prompt: z
|
||||
.string()
|
||||
.optional()
|
||||
.default(
|
||||
"Given a JSON schema, extract the data from the provided SOURCE TEXT according to the schema. Only output information that is explicitly stated or can be inferred from the SOURCE TEXT.",
|
||||
),
|
||||
extraction_prompt: z
|
||||
.string()
|
||||
.optional()
|
||||
.default("The extracted data using the given JSON schema."),
|
||||
error_handling_prompt: z
|
||||
.string()
|
||||
.optional()
|
||||
.default(
|
||||
"If the source text does not contain enough information to extract the value, explain the reason very briefly. Else, output null and fill out the value__ field.",
|
||||
),
|
||||
reasoning_prompt: z.string().optional().default(`
|
||||
Provide a brief explanation for how you arrived at the extracted value based on the source text provided.
|
||||
- For inferred values, explain the reasoning behind the extraction briefly.
|
||||
- For simple verbatim extraction, output 'VERBATIM EXTRACTION'.
|
||||
- When supporting data is not present in the source text, output 'INSUFFICIENT DATA' and emit blank or null values for the value__ field.
|
||||
`),
|
||||
cite_sources_prompt: z
|
||||
.object({})
|
||||
.optional()
|
||||
.default({
|
||||
description: `
|
||||
### Citation Rules (read carefully):
|
||||
- You must ANNOTATE every value with the MOST RELEVANT short EXACT substring from the source text that supports it.
|
||||
- For inferred values, cite the text used to infer it in the matching_text field or output 'INFERRED FROM TEXT'
|
||||
- If no support exists, output 'INSUFFICIENT DATA' and leave value__ null or '', 0.0, False etc depending on the type of the field.
|
||||
`,
|
||||
page: "Cite the page number of the source text that the extracted value is from. The page number is the integer that appears right after <<<PAGE:. If no page number is present in this format, use the default value of 1.",
|
||||
matching_text:
|
||||
'Cite the **MOST RELEVANT EXACT TEXT from the SOURCE TEXT** that supports the extracted value within 80 characters. If the exact substring is >80 chars, truncate with ellipsis "...". Provide only the single most relevant citation.',
|
||||
}),
|
||||
scratchpad_prompt: z
|
||||
.string()
|
||||
.optional()
|
||||
.default("Use for intermediate step-by-step reasoning. Be concise."),
|
||||
});
|
||||
|
||||
export const zRelatedNodeInfo = z.object({
|
||||
node_id: z.string(),
|
||||
node_type: z.union([zObjectType, z.string(), z.null()]).optional(),
|
||||
@@ -3545,8 +3517,6 @@ export const zRole = z.object({
|
||||
permissions: z.array(zPermission),
|
||||
});
|
||||
|
||||
export const zSchemaRelaxMode = z.enum(["FULL", "TOP_LEVEL", "LEAF"]);
|
||||
|
||||
export const zSearchRequest = z.object({
|
||||
page_size: z.union([z.number().int(), z.null()]).optional(),
|
||||
page_token: z.union([z.string(), z.null()]).optional(),
|
||||
@@ -3558,24 +3528,9 @@ export const zSearchRequest = z.object({
|
||||
offset: z.union([z.number().int().gte(0).lte(1000), z.null()]).optional(),
|
||||
});
|
||||
|
||||
export const zStructMode = z.enum([
|
||||
"STRUCT_PARSE",
|
||||
"JSON_MODE",
|
||||
"FUNC_CALL",
|
||||
"STRUCT_RELAXED",
|
||||
"UNSTRUCTURED",
|
||||
]);
|
||||
|
||||
export const zStructParseConf = z.object({
|
||||
model: zExtractModels.optional(),
|
||||
temperature: z.number().optional().default(0),
|
||||
relaxation_mode: zSchemaRelaxMode.optional(),
|
||||
struct_mode: zStructMode.optional(),
|
||||
fetch_logprobs: z.boolean().optional().default(false),
|
||||
handle_missing: z.boolean().optional().default(false),
|
||||
use_reasoning: z.boolean().optional().default(false),
|
||||
cite_sources: z.boolean().optional().default(false),
|
||||
prompt_conf: zPromptConf.optional(),
|
||||
export const zSpreadsheetJobCreate = z.object({
|
||||
file_id: z.string().uuid(),
|
||||
config: zSpreadsheetParsingConfig.optional(),
|
||||
});
|
||||
|
||||
export const zSupportedLlmModel = z.object({
|
||||
@@ -4017,6 +3972,33 @@ export const zCreateIntentAndCustomerSessionApiV1BillingCreateIntentAndCustomerS
|
||||
export const zGetMetronomeDashboardApiV1BillingMetronomeDashboardGetResponse =
|
||||
zMetronomeDashboardResponse;
|
||||
|
||||
export const zListJobsApiV1ExtractionJobsGetResponse = z.array(zExtractJob);
|
||||
|
||||
export const zRunJobApiV1ExtractionJobsPostResponse = zExtractJob;
|
||||
|
||||
export const zGetJobApiV1ExtractionJobsJobIdGetResponse = zExtractJob;
|
||||
|
||||
export const zRunJobOnFileApiV1ExtractionJobsFilePostResponse = zExtractJob;
|
||||
|
||||
export const zRunBatchJobsApiV1ExtractionJobsBatchPostResponse =
|
||||
z.array(zExtractJob);
|
||||
|
||||
export const zGetJobResultApiV1ExtractionJobsJobIdResultGetResponse =
|
||||
zExtractResultset;
|
||||
|
||||
export const zListExtractRunsApiV1ExtractionRunsGetResponse =
|
||||
zPaginatedExtractRunsResponse;
|
||||
|
||||
export const zGetLatestRunFromUiApiV1ExtractionRunsLatestFromUiGetResponse =
|
||||
z.union([zExtractRun, z.null()]);
|
||||
|
||||
export const zGetRunByJobIdApiV1ExtractionRunsByJobJobIdGetResponse =
|
||||
zExtractRun;
|
||||
|
||||
export const zGetRunApiV1ExtractionRunsRunIdGetResponse = zExtractRun;
|
||||
|
||||
export const zExtractStatelessApiV1ExtractionRunPostResponse = zExtractJob;
|
||||
|
||||
export const zListExtractionAgentsApiV1ExtractionExtractionAgentsGetResponse =
|
||||
z.array(zExtractAgent);
|
||||
|
||||
@@ -4041,35 +4023,6 @@ export const zGetExtractionAgentApiV1ExtractionExtractionAgentsExtractionAgentId
|
||||
export const zUpdateExtractionAgentApiV1ExtractionExtractionAgentsExtractionAgentIdPutResponse =
|
||||
zExtractAgent;
|
||||
|
||||
export const zListJobsApiV1ExtractionJobsGetResponse = z.array(zExtractJob);
|
||||
|
||||
export const zRunJobApiV1ExtractionJobsPostResponse = zExtractJob;
|
||||
|
||||
export const zGetJobApiV1ExtractionJobsJobIdGetResponse = zExtractJob;
|
||||
|
||||
export const zRunJobTestUserApiV1ExtractionJobsTestPostResponse = zExtractJob;
|
||||
|
||||
export const zRunJobOnFileApiV1ExtractionJobsFilePostResponse = zExtractJob;
|
||||
|
||||
export const zRunBatchJobsApiV1ExtractionJobsBatchPostResponse =
|
||||
z.array(zExtractJob);
|
||||
|
||||
export const zGetJobResultApiV1ExtractionJobsJobIdResultGetResponse =
|
||||
zExtractResultset;
|
||||
|
||||
export const zListExtractRunsApiV1ExtractionRunsGetResponse =
|
||||
zPaginatedExtractRunsResponse;
|
||||
|
||||
export const zGetLatestRunFromUiApiV1ExtractionRunsLatestFromUiGetResponse =
|
||||
z.union([zExtractRun, z.null()]);
|
||||
|
||||
export const zGetRunByJobIdApiV1ExtractionRunsByJobJobIdGetResponse =
|
||||
zExtractRun;
|
||||
|
||||
export const zGetRunApiV1ExtractionRunsRunIdGetResponse = zExtractRun;
|
||||
|
||||
export const zExtractStatelessApiV1ExtractionRunPostResponse = zExtractJob;
|
||||
|
||||
export const zListApiKeysApiV1BetaApiKeysGetResponse = zApiKeyQueryResponse;
|
||||
|
||||
export const zCreateApiKeyApiV1BetaApiKeysPostResponse = zApiKey;
|
||||
@@ -4100,6 +4053,9 @@ export const zSearchAgentDataApiV1BetaAgentDataSearchPostResponse =
|
||||
export const zAggregateAgentDataApiV1BetaAgentDataAggregatePostResponse =
|
||||
zPaginatedResponseAggregateGroup;
|
||||
|
||||
export const zDeleteAgentDataByQueryApiV1BetaAgentDataDeletePostResponse =
|
||||
zDeleteResponse;
|
||||
|
||||
export const zListQuotaConfigurationsApiV1BetaQuotaManagementGetResponse =
|
||||
zPaginatedResponseQuotaConfiguration;
|
||||
|
||||
@@ -4135,6 +4091,18 @@ export const zQueryParseConfigurationsApiV1BetaParseConfigurationsQueryPostRespo
|
||||
export const zGetLatestParseConfigurationApiV1BetaParseConfigurationsLatestGetResponse =
|
||||
z.union([zParseConfiguration, z.null()]);
|
||||
|
||||
export const zListSpreadsheetJobsApiV1BetaSpreadsheetJobsGetResponse =
|
||||
zPaginatedResponseSpreadsheetJob;
|
||||
|
||||
export const zCreateSpreadsheetJobApiV1BetaSpreadsheetJobsPostResponse =
|
||||
zSpreadsheetJob;
|
||||
|
||||
export const zGetSpreadsheetJobApiV1BetaSpreadsheetJobsSpreadsheetJobIdGetResponse =
|
||||
zSpreadsheetJob;
|
||||
|
||||
export const zGetTableDownloadPresignedUrlApiV1BetaSpreadsheetJobsSpreadsheetJobIdTablesTableIdResultGetResponse =
|
||||
zPresignedUrl;
|
||||
|
||||
export const zUploadFileV2ApiV2Alpha1ParseUploadPostResponse = zParsingJob;
|
||||
|
||||
export const zGetSupportedFileExtensionsApiParsingSupportedFileExtensionsGetResponse =
|
||||
|
||||
@@ -0,0 +1,246 @@
|
||||
import { describe, it, expect, beforeEach, vi } from "vitest";
|
||||
import { AgentClient, createAgentDataClient } from "../src/beta/agent/index.js";
|
||||
import * as sdk from "../src/client/index.js";
|
||||
|
||||
describe("AgentClient", () => {
|
||||
beforeEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
it("createItem sends correct payload and returns typed data", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "createAgentDataApiV1BetaAgentDataPost")
|
||||
.mockResolvedValue({
|
||||
data: {
|
||||
id: "1",
|
||||
deployment_name: "dep",
|
||||
collection: "col",
|
||||
data: { foo: "bar" },
|
||||
created_at: "2024-01-01T00:00:00Z",
|
||||
updated_at: "2024-01-01T00:00:00Z",
|
||||
},
|
||||
} as any);
|
||||
|
||||
const client = new AgentClient<{ foo: string }>({
|
||||
deploymentName: "dep",
|
||||
collection: "col",
|
||||
});
|
||||
const result = await client.createItem({ foo: "bar" });
|
||||
|
||||
expect(spy).toHaveBeenCalledOnce();
|
||||
const call = spy.mock.calls[0][0];
|
||||
expect(call.body.deployment_name).toBe("dep");
|
||||
expect(call.body.collection).toBe("col");
|
||||
expect(call.body.data).toEqual({ foo: "bar" });
|
||||
|
||||
expect(result.id).toBe("1");
|
||||
expect(result.deploymentName).toBe("dep");
|
||||
expect(result.collection).toBe("col");
|
||||
expect(result.data).toEqual({ foo: "bar" });
|
||||
expect(result.createdAt).toEqual(new Date("2024-01-01T00:00:00Z"));
|
||||
expect(result.updatedAt).toEqual(new Date("2024-01-01T00:00:00Z"));
|
||||
});
|
||||
|
||||
it("getItem returns null for 404 errors", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "getAgentDataApiV1BetaAgentDataItemIdGet")
|
||||
.mockImplementation(async () => {
|
||||
const err: any = new Error("Not found");
|
||||
err.response = { status: 404 };
|
||||
throw err;
|
||||
});
|
||||
|
||||
const client = new AgentClient({ deploymentName: "dep" });
|
||||
const res = await client.getItem("missing-id");
|
||||
|
||||
expect(spy).toHaveBeenCalledOnce();
|
||||
expect(res).toBeNull();
|
||||
});
|
||||
|
||||
it("updateItem updates and returns typed data", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "updateAgentDataApiV1BetaAgentDataItemIdPut")
|
||||
.mockResolvedValue({
|
||||
data: {
|
||||
id: "123",
|
||||
deployment_name: "dep",
|
||||
collection: "col",
|
||||
data: { foo: "baz" },
|
||||
created_at: "2024-01-01T00:00:00Z",
|
||||
updated_at: "2024-01-02T00:00:00Z",
|
||||
},
|
||||
} as any);
|
||||
|
||||
const client = new AgentClient<{ foo: string }>({
|
||||
deploymentName: "dep",
|
||||
collection: "col",
|
||||
});
|
||||
const res = await client.updateItem("123", { foo: "baz" });
|
||||
|
||||
expect(spy).toHaveBeenCalledOnce();
|
||||
const call = spy.mock.calls[0][0];
|
||||
expect(call.path.item_id).toBe("123");
|
||||
expect(call.body.data).toEqual({ foo: "baz" });
|
||||
|
||||
expect(res.id).toBe("123");
|
||||
expect(res.updatedAt).toEqual(new Date("2024-01-02T00:00:00Z"));
|
||||
});
|
||||
|
||||
it("deleteItem calls delete endpoint with correct path", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "deleteAgentDataApiV1BetaAgentDataItemIdDelete")
|
||||
.mockResolvedValue({} as any);
|
||||
|
||||
const client = new AgentClient({ deploymentName: "dep" });
|
||||
await client.deleteItem("abc");
|
||||
|
||||
expect(spy).toHaveBeenCalledOnce();
|
||||
expect(spy.mock.calls[0][0].path.item_id).toBe("abc");
|
||||
});
|
||||
|
||||
it("delete by query returns deleted count", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "deleteAgentDataByQueryApiV1BetaAgentDataDeletePost")
|
||||
.mockResolvedValue({ data: { deleted_count: 7 } } as any);
|
||||
|
||||
const client = new AgentClient({
|
||||
deploymentName: "dep",
|
||||
collection: "col",
|
||||
});
|
||||
const count = await client.delete({
|
||||
filter: { status: { op: "eq", value: "accepted" } as any },
|
||||
});
|
||||
|
||||
expect(spy).toHaveBeenCalledOnce();
|
||||
const body = spy.mock.calls[0][0].body;
|
||||
expect(body.deployment_name).toBe("dep");
|
||||
expect(body.collection).toBe("col");
|
||||
expect(count).toBe(7);
|
||||
});
|
||||
|
||||
it("search maps items and optional fields correctly", async () => {
|
||||
const now = "2024-01-01T00:00:00Z";
|
||||
const spy = vi
|
||||
.spyOn(sdk, "searchAgentDataApiV1BetaAgentDataSearchPost")
|
||||
.mockResolvedValue({
|
||||
data: {
|
||||
items: [
|
||||
{
|
||||
id: "1",
|
||||
deployment_name: "dep",
|
||||
collection: "col",
|
||||
data: { foo: "bar" },
|
||||
created_at: now,
|
||||
updated_at: now,
|
||||
},
|
||||
],
|
||||
total_size: 1,
|
||||
next_page_token: "next",
|
||||
},
|
||||
} as any);
|
||||
|
||||
const client = new AgentClient<{ foo: string }>({
|
||||
deploymentName: "dep",
|
||||
collection: "col",
|
||||
});
|
||||
const result = await client.search({
|
||||
includeTotal: true,
|
||||
orderBy: "created_at desc",
|
||||
pageSize: 1,
|
||||
offset: 0,
|
||||
});
|
||||
|
||||
expect(spy).toHaveBeenCalledOnce();
|
||||
const body = spy.mock.calls[0][0].body;
|
||||
expect(body.deployment_name).toBe("dep");
|
||||
expect(body.collection).toBe("col");
|
||||
expect(body.include_total).toBe(true);
|
||||
expect(body.order_by).toBe("created_at desc");
|
||||
expect(body.page_size).toBe(1);
|
||||
expect(body.offset).toBe(0);
|
||||
|
||||
expect(result.items).toHaveLength(1);
|
||||
expect(result.totalSize).toBe(1);
|
||||
expect(result.nextPageToken).toBe("next");
|
||||
expect(result.items[0].createdAt).toEqual(new Date(now));
|
||||
});
|
||||
|
||||
it("aggregate maps groups and optional fields correctly", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "aggregateAgentDataApiV1BetaAgentDataAggregatePost")
|
||||
.mockResolvedValue({
|
||||
data: {
|
||||
items: [
|
||||
{
|
||||
group_key: { status: "accepted" },
|
||||
count: 3,
|
||||
first_item: { foo: "bar" },
|
||||
},
|
||||
],
|
||||
total_size: 1,
|
||||
next_page_token: "tok",
|
||||
},
|
||||
} as any);
|
||||
|
||||
const client = new AgentClient<{ foo: string }>({
|
||||
deploymentName: "dep",
|
||||
collection: "col",
|
||||
});
|
||||
const result = await client.aggregate({
|
||||
groupBy: ["status"],
|
||||
count: true,
|
||||
first: true,
|
||||
pageSize: 1,
|
||||
offset: 0,
|
||||
});
|
||||
|
||||
expect(spy).toHaveBeenCalledOnce();
|
||||
const body = spy.mock.calls[0][0].body;
|
||||
expect(body.deployment_name).toBe("dep");
|
||||
expect(body.collection).toBe("col");
|
||||
expect(body.group_by).toEqual(["status"]);
|
||||
expect(body.count).toBe(true);
|
||||
expect(body.first).toBe(true);
|
||||
expect(body.page_size).toBe(1);
|
||||
expect(body.offset).toBe(0);
|
||||
|
||||
expect(result.items).toHaveLength(1);
|
||||
expect(result.totalSize).toBe(1);
|
||||
expect(result.nextPageToken).toBe("tok");
|
||||
expect(result.items[0].groupKey).toEqual({ status: "accepted" });
|
||||
expect(result.items[0].count).toBe(3);
|
||||
expect(result.items[0].firstItem).toEqual({ foo: "bar" });
|
||||
});
|
||||
|
||||
it("createAgentDataClient infers deployment name from env", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "searchAgentDataApiV1BetaAgentDataSearchPost")
|
||||
.mockResolvedValue({
|
||||
data: { items: [], total_size: 0 },
|
||||
} as any);
|
||||
|
||||
const client = createAgentDataClient({
|
||||
env: { LLAMA_DEPLOY_DEPLOYMENT_NAME: "env-dep" },
|
||||
});
|
||||
await client.search({});
|
||||
|
||||
const body = spy.mock.calls[0][0].body;
|
||||
expect(body.deployment_name).toBe("env-dep");
|
||||
});
|
||||
|
||||
it("createAgentDataClient infers deployment name from windowUrl (non-local)", async () => {
|
||||
const spy = vi
|
||||
.spyOn(sdk, "deleteAgentDataByQueryApiV1BetaAgentDataDeletePost")
|
||||
.mockResolvedValue({
|
||||
data: { deleted_count: 0 },
|
||||
} as any);
|
||||
|
||||
const client = createAgentDataClient({
|
||||
windowUrl: "https://app.llamaindex.ai/deployments/abc/ui/",
|
||||
});
|
||||
await client.delete({});
|
||||
|
||||
const body = spy.mock.calls[0][0].body;
|
||||
expect(body.deployment_name).toBe("abc");
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user