mirror of
https://github.com/run-llama/llama_cloud_services.git
synced 2026-07-21 03:55:22 -04:00
Compare commits
18 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 3690109abf | |||
| 2e322b4fc8 | |||
| 735e5f3ddc | |||
| e4cb4c75e5 | |||
| 1693deff72 | |||
| 3270f1228d | |||
| eeabf48d29 | |||
| 89348aa8e5 | |||
| 3ab2ce27b5 | |||
| 265261862f | |||
| 66cf052b8c | |||
| 2ca2d81e58 | |||
| 951ba4dfd8 | |||
| 386d210e8b | |||
| 9321602845 | |||
| 26c06353f0 | |||
| 62cf12d6eb | |||
| 253ee61463 |
@@ -7,8 +7,6 @@ assignees: ''
|
||||
|
||||
---
|
||||
|
||||
_Note: we're aware of some missing content in the output and layout issues on tables. Please refrain from opening new issues on this topic unless if you think it's different from what has already been reported._
|
||||
|
||||
**Describe the bug**
|
||||
Write a concise description of what the bug is.
|
||||
|
||||
@@ -19,19 +17,15 @@ If possible, please provide the PDF file causing the issue.
|
||||
If you have it, please provide the ID of the job you ran.
|
||||
You can find it here: https://cloud.llamaindex.ai/parse in the "History" tab.
|
||||
|
||||
**Screenshots**
|
||||
Feel free to also provide screenshots if relevant.
|
||||
|
||||
**Client:**
|
||||
Please remove untested options:
|
||||
- Frontend (cloud.llamaindex.ai)
|
||||
- Python Library
|
||||
- API
|
||||
- Frontend (cloud.llamaindex.ai)
|
||||
- Typescript Library
|
||||
- Notebook
|
||||
- API
|
||||
|
||||
**Options**
|
||||
What options did you use? Multimodal, fast mode, parsing instructions, etc.
|
||||
|
||||
**Additional context**
|
||||
Add any additional context about the problem here.
|
||||
What options did you use? Premium mode, multimodal, fast mode, parsing instructions, etc.
|
||||
Screenshots, code snippets, etc.
|
||||
|
||||
@@ -38,7 +38,22 @@ Lastly, install the package:
|
||||
|
||||
`pip install llama-parse`
|
||||
|
||||
Now you can run the following to parse your first PDF file:
|
||||
Now you can parse your first PDF file using the command line interface. Use the command `llama-parse [file_paths]`. See the help text with `llama-parse --help`.
|
||||
|
||||
```bash
|
||||
export LLAMA_CLOUD_API_KEY='llx-...'
|
||||
|
||||
# output as text
|
||||
llama-parse my_file.pdf --result-type text --output-file output.txt
|
||||
|
||||
# output as markdown
|
||||
llama-parse my_file.pdf --result-type markdown --output-file output.md
|
||||
|
||||
# output as raw json
|
||||
llama-parse my_file.pdf --output-raw-json --output-file output.json
|
||||
```
|
||||
|
||||
You can also create simple scripts:
|
||||
|
||||
```python
|
||||
import nest_asyncio
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
Binary file not shown.
|
After Width: | Height: | Size: 6.9 MiB |
@@ -342,7 +342,7 @@
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"display_name": "llama-parse-aNC435Vv-py3.10",
|
||||
"display_name": "Python 3 (ipykernel)",
|
||||
"language": "python",
|
||||
"name": "python3"
|
||||
},
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
Binary file not shown.
|
After Width: | Height: | Size: 580 KiB |
@@ -11,6 +11,8 @@
|
||||
"\n",
|
||||
"In this cookbook we show you how to build a multimodal report generation agent from a bank of research reports. We use the a set of ICLR papers (which were also used as the dataset in our [DeepLearning.ai course](https://www.deeplearning.ai/short-courses/building-agentic-rag-with-llamaindex/?utm_campaign=llamaindexC2-launch&utm_medium=headband&utm_source=dlai-homepage).\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"\n",
|
||||
"We use our workflow abstraction to define an agentic system that contains two main phases: a research phase that pulls in relevant files through chunk-level or file-level retrieval, and then a blog generation phase that synthesizes the final report."
|
||||
]
|
||||
},
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 1.5 MiB |
File diff suppressed because it is too large
Load Diff
Binary file not shown.
|
After Width: | Height: | Size: 986 KiB |
@@ -46,7 +46,7 @@
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"<LLAMA_CLOUD_API_KEY>"
|
||||
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"<LLAMA_CLOUD_API_KEY>\""
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
+454
-102
@@ -1,6 +1,6 @@
|
||||
import os
|
||||
import asyncio
|
||||
from io import TextIOWrapper
|
||||
from urllib.parse import urlparse
|
||||
|
||||
import httpx
|
||||
import mimetypes
|
||||
@@ -11,8 +11,7 @@ from contextlib import asynccontextmanager
|
||||
from io import BufferedIOBase
|
||||
|
||||
from fsspec import AbstractFileSystem
|
||||
from fsspec.spec import AbstractBufferedFile
|
||||
from llama_index.core.async_utils import run_jobs
|
||||
from llama_index.core.async_utils import asyncio_run, run_jobs
|
||||
from llama_index.core.bridge.pydantic import Field, field_validator
|
||||
from llama_index.core.constants import DEFAULT_BASE_URL
|
||||
from llama_index.core.readers.base import BasePydanticReader
|
||||
@@ -22,7 +21,6 @@ from llama_parse.utils import (
|
||||
nest_asyncio_err,
|
||||
nest_asyncio_msg,
|
||||
ResultType,
|
||||
Language,
|
||||
SUPPORTED_FILE_TYPES,
|
||||
)
|
||||
from copy import deepcopy
|
||||
@@ -37,6 +35,7 @@ _DEFAULT_SEPARATOR = "\n---\n"
|
||||
class LlamaParse(BasePydanticReader):
|
||||
"""A smart-parser for files."""
|
||||
|
||||
# Library / access specific configurations
|
||||
api_key: str = Field(
|
||||
default="",
|
||||
description="The API key for the LlamaParse API.",
|
||||
@@ -46,8 +45,20 @@ class LlamaParse(BasePydanticReader):
|
||||
default=DEFAULT_BASE_URL,
|
||||
description="The base URL of the Llama Parsing API.",
|
||||
)
|
||||
result_type: ResultType = Field(
|
||||
default=ResultType.TXT, description="The result type for the parser."
|
||||
check_interval: int = Field(
|
||||
default=1,
|
||||
description="The interval in seconds to check if the parsing is done.",
|
||||
)
|
||||
custom_client: Optional[httpx.AsyncClient] = Field(
|
||||
default=None, description="A custom HTTPX client to use for sending requests."
|
||||
)
|
||||
ignore_errors: bool = Field(
|
||||
default=True,
|
||||
description="Whether or not to ignore and skip errors raised during parsing.",
|
||||
)
|
||||
max_timeout: int = Field(
|
||||
default=2000,
|
||||
description="The maximum timeout in seconds to wait for the parsing to finish.",
|
||||
)
|
||||
num_workers: int = Field(
|
||||
default=4,
|
||||
@@ -55,63 +66,206 @@ class LlamaParse(BasePydanticReader):
|
||||
lt=10,
|
||||
description="The number of workers to use sending API requests for parsing.",
|
||||
)
|
||||
check_interval: int = Field(
|
||||
default=1,
|
||||
description="The interval in seconds to check if the parsing is done.",
|
||||
)
|
||||
max_timeout: int = Field(
|
||||
default=2000,
|
||||
description="The maximum timeout in seconds to wait for the parsing to finish.",
|
||||
)
|
||||
verbose: bool = Field(
|
||||
default=True, description="Whether to print the progress of the parsing."
|
||||
result_type: ResultType = Field(
|
||||
default=ResultType.TXT, description="The result type for the parser."
|
||||
)
|
||||
show_progress: bool = Field(
|
||||
default=True, description="Show progress when parsing multiple files."
|
||||
)
|
||||
language: Language = Field(
|
||||
default=Language.ENGLISH, description="The language of the text to parse."
|
||||
split_by_page: bool = Field(
|
||||
default=True,
|
||||
description="Whether to split by page using the page separator",
|
||||
)
|
||||
parsing_instruction: Optional[str] = Field(
|
||||
default="", description="The parsing instruction for the parser."
|
||||
verbose: bool = Field(
|
||||
default=True, description="Whether to print the progress of the parsing."
|
||||
)
|
||||
skip_diagonal_text: Optional[bool] = Field(
|
||||
|
||||
# Parsing specific configurations (Alphabetical order)
|
||||
annotate_links: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will ignore diagonal text (when the text rotation in degrees modulo 90 is not 0).",
|
||||
description="Annotate links found in the document to extract their URL.",
|
||||
)
|
||||
invalidate_cache: Optional[bool] = Field(
|
||||
auto_mode: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the cache will be ignored and the document re-processes. All document are kept in cache for 48hours after the job was completed to avoid processing the same document twice.",
|
||||
description="If set to true, the parser will automatically select the best mode to extract text from documents based on the rules provide. Will use the 'accurate' default mode by default and will upgrade page that match the rule to Premium mode.",
|
||||
)
|
||||
auto_mode_trigger_on_image_in_page: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If auto_mode is set to true, the parser will upgrade the page that contain an image to Premium mode.",
|
||||
)
|
||||
auto_mode_trigger_on_table_in_page: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If auto_mode is set to true, the parser will upgrade the page that contain a table to Premium mode.",
|
||||
)
|
||||
auto_mode_trigger_on_text_in_page: Optional[str] = Field(
|
||||
default=None,
|
||||
description="If auto_mode is set to true, the parser will upgrade the page that contain the text to Premium mode.",
|
||||
)
|
||||
auto_mode_trigger_on_regexp_in_page: Optional[str] = Field(
|
||||
default=None,
|
||||
description="If auto_mode is set to true, the parser will upgrade the page that match the regexp to Premium mode.",
|
||||
)
|
||||
azure_openai_api_version: Optional[str] = Field(
|
||||
default=None, description="Azure Openai API Version"
|
||||
)
|
||||
azure_openai_deployment_name: Optional[str] = Field(
|
||||
default=None, description="Azure Openai Deployment Name"
|
||||
)
|
||||
azure_openai_endpoint: Optional[str] = Field(
|
||||
default=None, description="Azure Openai Endpoint"
|
||||
)
|
||||
azure_openai_key: Optional[str] = Field(
|
||||
default=None, description="Azure Openai Key"
|
||||
)
|
||||
bbox_bottom: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The bottom margin of the bounding box to use to extract text from documents expressed as a float between 0 and 1 representing the percentage of the page height.",
|
||||
)
|
||||
bbox_left: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The left margin of the bounding box to use to extract text from documents expressed as a float between 0 and 1 representing the percentage of the page width.",
|
||||
)
|
||||
bbox_right: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The right margin of the bounding box to use to extract text from documents expressed as a float between 0 and 1 representing the percentage of the page width.",
|
||||
)
|
||||
bbox_top: Optional[float] = Field(
|
||||
default=None,
|
||||
description="The top margin of the bounding box to use to extract text from documents expressed as a float between 0 and 1 representing the percentage of the page height.",
|
||||
)
|
||||
continuous_mode: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Parse documents continuously, leading to better results on documents where tables span across two pages.",
|
||||
)
|
||||
disable_ocr: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Disable the OCR on the document. LlamaParse will only extract the copyable text from the document.",
|
||||
)
|
||||
disable_image_extraction: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will not extract images from the document. Make the parser faster.",
|
||||
)
|
||||
do_not_cache: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the document will not be cached. This mean that you will be re-charged it you reprocess them as they will not be cached.",
|
||||
)
|
||||
fast_mode: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Note: Non compatible with gpt-4o. If set to true, the parser will use a faster mode to extract text from documents. This mode will skip OCR of images, and table/heading reconstruction.",
|
||||
)
|
||||
premium_mode: bool = Field(
|
||||
default=False,
|
||||
description="Use our best parser mode if set to True.",
|
||||
)
|
||||
do_not_unroll_columns: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will keep column in the text according to document layout. Reduce reconstruction accuracy, and LLM's/embedings performances in most case.",
|
||||
)
|
||||
page_separator: Optional[str] = Field(
|
||||
extract_charts: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will extract/tag charts from the document.",
|
||||
)
|
||||
fast_mode: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Note: Non compatible with gpt-4o. If set to true, the parser will use a faster mode to extract text from documents. This mode will skip OCR of images, and table/heading reconstruction.",
|
||||
)
|
||||
guess_xlsx_sheet_names: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Whether to guess the sheet names of the xlsx file.",
|
||||
)
|
||||
html_make_all_elements_visible: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, when parsing HTML the parser will consider all elements display not element as display block.",
|
||||
)
|
||||
html_remove_fixed_elements: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, when parsing HTML the parser will remove fixed elements. Useful to hide cookie banners.",
|
||||
)
|
||||
http_proxy: Optional[str] = Field(
|
||||
default=None,
|
||||
description="A templated page separator to use to split the text. If it contain `{page_number}`,it will be replaced by the next page number. If not set will the default separator '\\n---\\n' will be used.",
|
||||
description="(optional) If set with input_url will use the specified http proxy to download the file.",
|
||||
)
|
||||
invalidate_cache: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the cache will be ignored and the document re-processes. All document are kept in cache for 48hours after the job was completed to avoid processing the same document twice.",
|
||||
)
|
||||
is_formatting_instruction: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Allow the parsing instruction to also format the output. Disable to have a cleaner markdown output.",
|
||||
)
|
||||
language: Optional[str] = Field(
|
||||
default="en", description="The language of the text to parse."
|
||||
)
|
||||
max_pages: Optional[int] = Field(
|
||||
default=None,
|
||||
description="The maximum number of pages to extract text from documents. If set to 0 or not set, all pages will be that should be extracted will be extracted (can work in combination with targetPages).",
|
||||
)
|
||||
output_pdf_of_document: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will also output a PDF of the document. (except for spreadsheets)",
|
||||
)
|
||||
output_s3_path_prefix: Optional[str] = Field(
|
||||
default=None,
|
||||
description="An S3 path prefix to store the output of the parsing job. If set, the parser will upload the output to S3. The bucket need to be accessible from the LlamaIndex organization.",
|
||||
)
|
||||
page_prefix: Optional[str] = Field(
|
||||
default=None,
|
||||
description="A templated prefix to add to the beginning of each page. If it contain `{page_number}`, it will be replaced by the page number.",
|
||||
)
|
||||
page_separator: Optional[str] = Field(
|
||||
default=None,
|
||||
description="A templated page separator to use to split the text. If it contain `{page_number}`,it will be replaced by the next page number. If not set will the default separator '\\n---\\n' will be used.",
|
||||
)
|
||||
page_suffix: Optional[str] = Field(
|
||||
default=None,
|
||||
description="A templated suffix to add to the beginning of each page. If it contain `{page_number}`, it will be replaced by the page number.",
|
||||
)
|
||||
gpt4o_mode: bool = Field(
|
||||
parsing_instruction: Optional[str] = Field(
|
||||
default="", description="The parsing instruction for the parser."
|
||||
)
|
||||
premium_mode: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Use our best parser mode if set to True.",
|
||||
)
|
||||
skip_diagonal_text: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will ignore diagonal text (when the text rotation in degrees modulo 90 is not 0).",
|
||||
)
|
||||
structured_output: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will output structured data based on the provided JSON Schema.",
|
||||
)
|
||||
structured_output_json_schema: Optional[str] = Field(
|
||||
default=None,
|
||||
description="A JSON Schema to use to structure the output of the parsing job. If set, the parser will output structured data based on the provided JSON Schema.",
|
||||
)
|
||||
structured_output_json_schema_name: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The named JSON Schema to use to structure the output of the parsing job. For convenience / testing, LlamaParse provides a few named JSON Schema that can be used directly. Use 'imFeelingLucky' to let llamaParse dream the schema.",
|
||||
)
|
||||
take_screenshot: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Whether to take screenshot of each page of the document.",
|
||||
)
|
||||
target_pages: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The target pages to extract text from documents. Describe as a comma separated list of page numbers. The first page of the document is page 0",
|
||||
)
|
||||
use_vendor_multimodal_model: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Whether to use the vendor multimodal API.",
|
||||
)
|
||||
vendor_multimodal_api_key: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The API key for the multimodal API.",
|
||||
)
|
||||
vendor_multimodal_model_name: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The model name for the vendor multimodal API.",
|
||||
)
|
||||
webhook_url: Optional[str] = Field(
|
||||
default=None,
|
||||
description="A URL that needs to be called at the end of the parsing job.",
|
||||
)
|
||||
|
||||
# Deprecated
|
||||
bounding_box: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The bounding box to use to extract text from documents describe as a string containing the bounding box margins",
|
||||
)
|
||||
gpt4o_mode: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Whether to use gpt-4o extract text from documents.",
|
||||
)
|
||||
@@ -119,41 +273,6 @@ class LlamaParse(BasePydanticReader):
|
||||
default=None,
|
||||
description="The API key for the GPT-4o API. Lowers the cost of parsing.",
|
||||
)
|
||||
bounding_box: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The bounding box to use to extract text from documents describe as a string containing the bounding box margins",
|
||||
)
|
||||
target_pages: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The target pages to extract text from documents. Describe as a comma separated list of page numbers. The first page of the document is page 0",
|
||||
)
|
||||
ignore_errors: bool = Field(
|
||||
default=True,
|
||||
description="Whether or not to ignore and skip errors raised during parsing.",
|
||||
)
|
||||
split_by_page: bool = Field(
|
||||
default=True,
|
||||
description="Whether to split by page using the page separator",
|
||||
)
|
||||
vendor_multimodal_api_key: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The API key for the multimodal API.",
|
||||
)
|
||||
use_vendor_multimodal_model: bool = Field(
|
||||
default=False,
|
||||
description="Whether to use the vendor multimodal API.",
|
||||
)
|
||||
vendor_multimodal_model_name: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The model name for the vendor multimodal API.",
|
||||
)
|
||||
take_screenshot: bool = Field(
|
||||
default=False,
|
||||
description="Whether to take screenshot of each page of the document.",
|
||||
)
|
||||
custom_client: Optional[httpx.AsyncClient] = Field(
|
||||
default=None, description="A custom HTTPX client to use for sending requests."
|
||||
)
|
||||
|
||||
@field_validator("api_key", mode="before", check_fields=True)
|
||||
@classmethod
|
||||
@@ -185,6 +304,38 @@ class LlamaParse(BasePydanticReader):
|
||||
async with httpx.AsyncClient(timeout=self.max_timeout) as client:
|
||||
yield client
|
||||
|
||||
def _is_input_url(self, file_path: FileInput) -> bool:
|
||||
"""Check if the input is a valid URL.
|
||||
|
||||
This method checks for:
|
||||
- Proper URL scheme (http/https)
|
||||
- Valid URL structure
|
||||
- Network location (domain)
|
||||
"""
|
||||
if not isinstance(file_path, str):
|
||||
return False
|
||||
try:
|
||||
result = urlparse(file_path)
|
||||
return all(
|
||||
[
|
||||
result.scheme in ("http", "https"),
|
||||
result.netloc, # Has domain
|
||||
result.scheme, # Has scheme
|
||||
]
|
||||
)
|
||||
except Exception:
|
||||
return False
|
||||
|
||||
def _is_s3_url(self, file_path: FileInput) -> bool:
|
||||
"""Check if the input is a valid URL.
|
||||
|
||||
This method checks for:
|
||||
- Proper S3 scheme (s3://)
|
||||
"""
|
||||
if isinstance(file_path, str):
|
||||
return file_path.startswith("s3://")
|
||||
return False
|
||||
|
||||
# upload a document and get back a job_id
|
||||
async def _create_job(
|
||||
self,
|
||||
@@ -196,6 +347,8 @@ class LlamaParse(BasePydanticReader):
|
||||
url = f"{self.base_url}/api/parsing/upload"
|
||||
files = None
|
||||
file_handle = None
|
||||
input_url = file_input if self._is_input_url(file_input) else None
|
||||
input_s3_path = file_input if self._is_s3_url(file_input) else None
|
||||
|
||||
if isinstance(file_input, (bytes, BufferedIOBase)):
|
||||
if not extra_info or "file_name" not in extra_info:
|
||||
@@ -205,6 +358,10 @@ class LlamaParse(BasePydanticReader):
|
||||
file_name = extra_info["file_name"]
|
||||
mime_type = mimetypes.guess_type(file_name)[0]
|
||||
files = {"file": (file_name, file_input, mime_type)}
|
||||
elif input_url is not None:
|
||||
files = None
|
||||
elif input_s3_path is not None:
|
||||
files = None
|
||||
elif isinstance(file_input, (str, Path, PurePosixPath, PurePath)):
|
||||
file_path = str(file_input)
|
||||
file_ext = os.path.splitext(file_path)[1].lower()
|
||||
@@ -224,40 +381,178 @@ class LlamaParse(BasePydanticReader):
|
||||
"file_input must be either a file path string, file bytes, or buffer object"
|
||||
)
|
||||
|
||||
data = {
|
||||
"language": self.language.value,
|
||||
"parsing_instruction": self.parsing_instruction,
|
||||
"invalidate_cache": self.invalidate_cache,
|
||||
"skip_diagonal_text": self.skip_diagonal_text,
|
||||
"do_not_cache": self.do_not_cache,
|
||||
"fast_mode": self.fast_mode,
|
||||
"premium_mode": self.premium_mode,
|
||||
"do_not_unroll_columns": self.do_not_unroll_columns,
|
||||
"gpt4o_mode": self.gpt4o_mode,
|
||||
"gpt4o_api_key": self.gpt4o_api_key,
|
||||
"vendor_multimodal_api_key": self.vendor_multimodal_api_key,
|
||||
"use_vendor_multimodal_model": self.use_vendor_multimodal_model,
|
||||
"vendor_multimodal_model_name": self.vendor_multimodal_model_name,
|
||||
"take_screenshot": self.take_screenshot,
|
||||
}
|
||||
data: Dict[str, Any] = {}
|
||||
|
||||
data["from_python_package"] = True
|
||||
|
||||
if self.annotate_links:
|
||||
data["annotate_links"] = self.annotate_links
|
||||
|
||||
if self.auto_mode:
|
||||
data["auto_mode"] = self.auto_mode
|
||||
|
||||
if self.auto_mode_trigger_on_image_in_page:
|
||||
data[
|
||||
"auto_mode_trigger_on_image_in_page"
|
||||
] = self.auto_mode_trigger_on_image_in_page
|
||||
|
||||
if self.auto_mode_trigger_on_table_in_page:
|
||||
data[
|
||||
"auto_mode_trigger_on_table_in_page"
|
||||
] = self.auto_mode_trigger_on_table_in_page
|
||||
|
||||
if self.auto_mode_trigger_on_text_in_page is not None:
|
||||
data[
|
||||
"auto_mode_trigger_on_text_in_page"
|
||||
] = self.auto_mode_trigger_on_text_in_page
|
||||
|
||||
if self.auto_mode_trigger_on_regexp_in_page is not None:
|
||||
data[
|
||||
"auto_mode_trigger_on_regexp_in_page"
|
||||
] = self.auto_mode_trigger_on_regexp_in_page
|
||||
|
||||
if self.azure_openai_api_version is not None:
|
||||
data["azure_openai_api_version"] = self.azure_openai_api_version
|
||||
|
||||
if self.azure_openai_deployment_name is not None:
|
||||
data["azure_openai_deployment_name"] = self.azure_openai_deployment_name
|
||||
|
||||
if self.azure_openai_endpoint is not None:
|
||||
data["azure_openai_endpoint"] = self.azure_openai_endpoint
|
||||
|
||||
if self.azure_openai_key is not None:
|
||||
data["azure_openai_key"] = self.azure_openai_key
|
||||
|
||||
if self.bbox_bottom is not None:
|
||||
data["bbox_bottom"] = self.bbox_bottom
|
||||
|
||||
if self.bbox_left is not None:
|
||||
data["bbox_left"] = self.bbox_left
|
||||
|
||||
if self.bbox_right is not None:
|
||||
data["bbox_right"] = self.bbox_right
|
||||
|
||||
if self.bbox_top is not None:
|
||||
data["bbox_top"] = self.bbox_top
|
||||
|
||||
if self.continuous_mode:
|
||||
data["continuous_mode"] = self.continuous_mode
|
||||
|
||||
if self.disable_ocr:
|
||||
data["disable_ocr"] = self.disable_ocr
|
||||
|
||||
if self.disable_image_extraction:
|
||||
data["disable_image_extraction"] = self.disable_image_extraction
|
||||
|
||||
if self.do_not_cache:
|
||||
data["do_not_cache"] = self.do_not_cache
|
||||
|
||||
if self.do_not_unroll_columns:
|
||||
data["do_not_unroll_columns"] = self.do_not_unroll_columns
|
||||
|
||||
if self.extract_charts:
|
||||
data["extract_charts"] = self.extract_charts
|
||||
|
||||
if self.fast_mode:
|
||||
data["fast_mode"] = self.fast_mode
|
||||
|
||||
if self.guess_xlsx_sheet_names:
|
||||
data["guess_xlsx_sheet_names"] = self.guess_xlsx_sheet_names
|
||||
|
||||
if self.html_make_all_elements_visible:
|
||||
data["html_make_all_elements_visible"] = self.html_make_all_elements_visible
|
||||
|
||||
if self.html_remove_fixed_elements:
|
||||
data["html_remove_fixed_elements"] = self.html_remove_fixed_elements
|
||||
|
||||
if self.http_proxy is not None:
|
||||
data["http_proxy"] = self.http_proxy
|
||||
|
||||
if input_url is not None:
|
||||
files = None
|
||||
data["input_url"] = str(input_url)
|
||||
|
||||
if input_s3_path is not None:
|
||||
files = None
|
||||
data["input_s3_path"] = str(input_s3_path)
|
||||
|
||||
if self.invalidate_cache:
|
||||
data["invalidate_cache"] = self.invalidate_cache
|
||||
|
||||
if self.is_formatting_instruction:
|
||||
data["is_formatting_instruction"] = self.is_formatting_instruction
|
||||
|
||||
if self.language:
|
||||
data["language"] = self.language
|
||||
|
||||
if self.max_pages is not None:
|
||||
data["max_pages"] = self.max_pages
|
||||
|
||||
if self.output_pdf_of_document:
|
||||
data["output_pdf_of_document"] = self.output_pdf_of_document
|
||||
|
||||
if self.output_s3_path_prefix is not None:
|
||||
data["output_s3_path_prefix"] = self.output_s3_path_prefix
|
||||
|
||||
if self.page_prefix is not None:
|
||||
data["page_prefix"] = self.page_prefix
|
||||
|
||||
# only send page separator to server if it is not None
|
||||
# as if a null, "" string is sent the server will then ignore the page separator instead of using the default
|
||||
if self.page_separator is not None:
|
||||
data["page_separator"] = self.page_separator
|
||||
|
||||
if self.page_prefix is not None:
|
||||
data["page_prefix"] = self.page_prefix
|
||||
|
||||
if self.page_suffix is not None:
|
||||
data["page_suffix"] = self.page_suffix
|
||||
|
||||
if self.bounding_box is not None:
|
||||
data["bounding_box"] = self.bounding_box
|
||||
if self.parsing_instruction is not None:
|
||||
data["parsing_instruction"] = self.parsing_instruction
|
||||
|
||||
if self.premium_mode:
|
||||
data["premium_mode"] = self.premium_mode
|
||||
|
||||
if self.skip_diagonal_text:
|
||||
data["skip_diagonal_text"] = self.skip_diagonal_text
|
||||
|
||||
if self.structured_output:
|
||||
data["structured_output"] = self.structured_output
|
||||
|
||||
if self.structured_output_json_schema is not None:
|
||||
data["structured_output_json_schema"] = self.structured_output_json_schema
|
||||
|
||||
if self.structured_output_json_schema_name is not None:
|
||||
data[
|
||||
"structured_output_json_schema_name"
|
||||
] = self.structured_output_json_schema_name
|
||||
|
||||
if self.take_screenshot:
|
||||
data["take_screenshot"] = self.take_screenshot
|
||||
|
||||
if self.target_pages is not None:
|
||||
data["target_pages"] = self.target_pages
|
||||
|
||||
if self.use_vendor_multimodal_model:
|
||||
data["use_vendor_multimodal_model"] = self.use_vendor_multimodal_model
|
||||
|
||||
if self.vendor_multimodal_api_key is not None:
|
||||
data["vendor_multimodal_api_key"] = self.vendor_multimodal_api_key
|
||||
|
||||
if self.vendor_multimodal_model_name is not None:
|
||||
data["vendor_multimodal_model_name"] = self.vendor_multimodal_model_name
|
||||
|
||||
if self.webhook_url is not None:
|
||||
data["webhook_url"] = self.webhook_url
|
||||
|
||||
# Deprecated
|
||||
if self.bounding_box is not None:
|
||||
data["bounding_box"] = self.bounding_box
|
||||
|
||||
if self.gpt4o_mode:
|
||||
data["gpt4o_mode"] = self.gpt4o_mode
|
||||
|
||||
if self.gpt4o_api_key is not None:
|
||||
data["gpt4o_api_key"] = self.gpt4o_api_key
|
||||
|
||||
try:
|
||||
async with self.client_context() as client:
|
||||
response = await client.post(
|
||||
@@ -274,12 +569,6 @@ class LlamaParse(BasePydanticReader):
|
||||
if file_handle is not None:
|
||||
file_handle.close()
|
||||
|
||||
@staticmethod
|
||||
def __get_filename(f: Union[TextIOWrapper, AbstractBufferedFile]) -> str:
|
||||
if isinstance(f, TextIOWrapper):
|
||||
return f.name
|
||||
return f.full_name
|
||||
|
||||
async def _get_job_result(
|
||||
self, job_id: str, result_type: str, verbose: bool = False
|
||||
) -> Dict[str, Any]:
|
||||
@@ -308,7 +597,8 @@ class LlamaParse(BasePydanticReader):
|
||||
continue
|
||||
|
||||
# Allowed values "PENDING", "SUCCESS", "ERROR", "CANCELED"
|
||||
status = result.json()["status"]
|
||||
result_json = result.json()
|
||||
status = result_json["status"]
|
||||
if status == "SUCCESS":
|
||||
parsed_result = await client.get(result_url, headers=headers)
|
||||
return parsed_result.json()
|
||||
@@ -320,6 +610,14 @@ class LlamaParse(BasePydanticReader):
|
||||
print(".", end="", flush=True)
|
||||
|
||||
await asyncio.sleep(self.check_interval)
|
||||
else:
|
||||
error_code = result_json.get("error_code", "No error code found")
|
||||
error_message = result_json.get(
|
||||
"error_message", "No error message found"
|
||||
)
|
||||
|
||||
exception_str = f"Job ID: {job_id} failed with status: {status}, Error code: {error_code}, Error message: {error_message}"
|
||||
raise Exception(exception_str)
|
||||
|
||||
async def _aload_data(
|
||||
self,
|
||||
@@ -364,7 +662,7 @@ class LlamaParse(BasePydanticReader):
|
||||
fs: Optional[AbstractFileSystem] = None,
|
||||
) -> List[Document]:
|
||||
"""Load data from the input path."""
|
||||
if isinstance(file_path, (str, Path, bytes, BufferedIOBase)):
|
||||
if isinstance(file_path, (str, PurePosixPath, Path, bytes, BufferedIOBase)):
|
||||
return await self._aload_data(
|
||||
file_path, extra_info=extra_info, fs=fs, verbose=self.verbose
|
||||
)
|
||||
@@ -406,7 +704,7 @@ class LlamaParse(BasePydanticReader):
|
||||
) -> List[Document]:
|
||||
"""Load data from the input path."""
|
||||
try:
|
||||
return asyncio.run(self.aload_data(file_path, extra_info, fs=fs))
|
||||
return asyncio_run(self.aload_data(file_path, extra_info, fs=fs))
|
||||
except RuntimeError as e:
|
||||
if nest_asyncio_err in str(e):
|
||||
raise RuntimeError(nest_asyncio_msg)
|
||||
@@ -473,7 +771,7 @@ class LlamaParse(BasePydanticReader):
|
||||
) -> List[dict]:
|
||||
"""Parse the input path."""
|
||||
try:
|
||||
return asyncio.run(self.aget_json(file_path, extra_info))
|
||||
return asyncio_run(self.aget_json(file_path, extra_info))
|
||||
except RuntimeError as e:
|
||||
if nest_asyncio_err in str(e):
|
||||
raise RuntimeError(nest_asyncio_msg)
|
||||
@@ -536,7 +834,61 @@ class LlamaParse(BasePydanticReader):
|
||||
def get_images(self, json_result: List[dict], download_path: str) -> List[dict]:
|
||||
"""Download images from the parsed result."""
|
||||
try:
|
||||
return asyncio.run(self.aget_images(json_result, download_path))
|
||||
return asyncio_run(self.aget_images(json_result, download_path))
|
||||
except RuntimeError as e:
|
||||
if nest_asyncio_err in str(e):
|
||||
raise RuntimeError(nest_asyncio_msg)
|
||||
else:
|
||||
raise e
|
||||
|
||||
async def aget_xlsx(
|
||||
self, json_result: List[dict], download_path: str
|
||||
) -> List[dict]:
|
||||
"""Download images from the parsed result."""
|
||||
headers = {"Authorization": f"Bearer {self.api_key}"}
|
||||
|
||||
# make the download path
|
||||
if not os.path.exists(download_path):
|
||||
os.makedirs(download_path)
|
||||
try:
|
||||
xlsx_list = []
|
||||
for result in json_result:
|
||||
job_id = result["job_id"]
|
||||
if self.verbose:
|
||||
print("> XLSX")
|
||||
|
||||
xlsx_path = os.path.join(download_path, f"{job_id}.xlsx")
|
||||
|
||||
xlsx = {}
|
||||
|
||||
xlsx["path"] = xlsx_path
|
||||
xlsx["job_id"] = job_id
|
||||
xlsx["original_file_path"] = result.get("file_path", None)
|
||||
|
||||
with open(xlsx_path, "wb") as f:
|
||||
xlsx_url = (
|
||||
f"{self.base_url}/api/parsing/job/{job_id}/result/raw/xlsx"
|
||||
)
|
||||
async with self.client_context() as client:
|
||||
res = await client.get(
|
||||
xlsx_url, headers=headers, timeout=self.max_timeout
|
||||
)
|
||||
res.raise_for_status()
|
||||
f.write(res.content)
|
||||
xlsx_list.append(xlsx)
|
||||
return xlsx_list
|
||||
|
||||
except Exception as e:
|
||||
print("Error while downloading xlsx:", e)
|
||||
if self.ignore_errors:
|
||||
return []
|
||||
else:
|
||||
raise e
|
||||
|
||||
def get_xlsx(self, json_result: List[dict], download_path: str) -> List[dict]:
|
||||
"""Download xlsx from the parsed result."""
|
||||
try:
|
||||
return asyncio_run(self.aget_xlsx(json_result, download_path))
|
||||
except RuntimeError as e:
|
||||
if nest_asyncio_err in str(e):
|
||||
raise RuntimeError(nest_asyncio_msg)
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
import click
|
||||
import json
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from pydantic.fields import FieldInfo
|
||||
from typing import Any, Callable, List
|
||||
|
||||
from llama_parse.base import LlamaParse
|
||||
|
||||
|
||||
def pydantic_field_to_click_option(name: str, field: FieldInfo) -> click.Option:
|
||||
"""Convert a Pydantic field to a Click option."""
|
||||
kwargs = {
|
||||
"default": field.default if field.default else None,
|
||||
"help": field.description,
|
||||
}
|
||||
|
||||
if isinstance(kwargs["default"], Enum):
|
||||
kwargs["default"] = kwargs["default"].value
|
||||
|
||||
if field.annotation is bool:
|
||||
kwargs["is_flag"] = True
|
||||
if field.default and field.default is True:
|
||||
name = f"no-{name}"
|
||||
return click.option(f'--{name.replace("_", "-")}', **kwargs)
|
||||
|
||||
|
||||
def add_options(options: List[click.Option]) -> Callable:
|
||||
def _add_options(func: Callable) -> Callable:
|
||||
for option in reversed(options):
|
||||
func = option(func)
|
||||
return func
|
||||
|
||||
return _add_options
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.argument("file_paths", nargs=-1, type=click.Path(exists=True, path_type=Path))
|
||||
@click.option(
|
||||
"--output-file", type=click.Path(path_type=Path), help="Path to save the output"
|
||||
)
|
||||
@click.option("--output-raw-json", is_flag=True, help="Output the raw JSON result")
|
||||
@add_options(
|
||||
[
|
||||
pydantic_field_to_click_option(name, field)
|
||||
for name, field in LlamaParse.model_fields.items()
|
||||
if name not in ["custom_client"]
|
||||
]
|
||||
)
|
||||
def parse(**kwargs: Any) -> None:
|
||||
"""Parse files using LlamaParse and output the results."""
|
||||
file_paths = kwargs.pop("file_paths")
|
||||
output_file = kwargs.pop("output_file")
|
||||
output_raw_json = kwargs.pop("output_raw_json")
|
||||
|
||||
# Remove None values to use LlamaParse defaults
|
||||
kwargs = {k: v for k, v in kwargs.items() if v is not None}
|
||||
|
||||
# Remove no- prefix for boolean flags
|
||||
kwargs = {k.replace("no_", ""): v for k, v in kwargs.items()}
|
||||
|
||||
parser = LlamaParse(**kwargs)
|
||||
if output_raw_json:
|
||||
results = parser.get_json_result(list(file_paths))
|
||||
|
||||
if output_file:
|
||||
with output_file.open("w") as f:
|
||||
json.dump(results, f)
|
||||
click.echo(f"Results saved to {output_file}")
|
||||
else:
|
||||
click.echo(results)
|
||||
else:
|
||||
results = parser.load_data(list(file_paths))
|
||||
|
||||
if output_file:
|
||||
with output_file.open("w") as f:
|
||||
for i, doc in enumerate(results):
|
||||
f.write(f"File: {doc.metadata.get('file_path', 'Unknown')}\n") # type: ignore
|
||||
f.write(doc.text) # type: ignore
|
||||
if i < len(results) - 1:
|
||||
f.write("\n\n---\n\n")
|
||||
click.echo(f"Results saved to {output_file}")
|
||||
else:
|
||||
for i, doc in enumerate(results):
|
||||
click.echo(f"File: {doc.metadata.get('file_path', 'Unknown')}") # type: ignore
|
||||
click.echo(doc.text) # type: ignore
|
||||
if i < len(results) - 1:
|
||||
click.echo("\n---\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parse()
|
||||
@@ -10,6 +10,8 @@ class ResultType(str, Enum):
|
||||
|
||||
TXT = "text"
|
||||
MD = "markdown"
|
||||
JSON = "json"
|
||||
STRUCTURED = "structured"
|
||||
|
||||
|
||||
class Language(str, Enum):
|
||||
|
||||
Generated
+1472
-1248
File diff suppressed because it is too large
Load Diff
+8
-2
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.poetry]
|
||||
name = "llama-parse"
|
||||
version = "0.5.6"
|
||||
version = "0.5.16"
|
||||
description = "Parse files into RAG-Optimized formats."
|
||||
authors = ["Logan Markewich <logan@llamaindex.ai>"]
|
||||
license = "MIT"
|
||||
@@ -12,9 +12,15 @@ readme = "README.md"
|
||||
packages = [{include = "llama_parse"}]
|
||||
|
||||
[tool.poetry.dependencies]
|
||||
python = ">=3.8.1,<4.0"
|
||||
python = ">=3.9,<4.0"
|
||||
llama-index-core = ">=0.11.0"
|
||||
pydantic = "!=2.10"
|
||||
click = "^8.1.7"
|
||||
|
||||
[tool.poetry.group.dev.dependencies]
|
||||
pytest = "^8.0.0"
|
||||
pytest-asyncio = "*"
|
||||
ipykernel = "^6.29.0"
|
||||
|
||||
[tool.poetry.scripts]
|
||||
llama-parse = "llama_parse.cli.main:parse"
|
||||
|
||||
Binary file not shown.
|
After Width: | Height: | Size: 347 KiB |
+77
-7
@@ -1,5 +1,6 @@
|
||||
import os
|
||||
import pytest
|
||||
import shutil
|
||||
from fsspec.implementations.local import LocalFileSystem
|
||||
from httpx import AsyncClient
|
||||
|
||||
@@ -76,13 +77,14 @@ def test_simple_page_markdown_buffer(markdown_parser: LlamaParse) -> None:
|
||||
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
|
||||
reason="LLAMA_CLOUD_API_KEY not set",
|
||||
)
|
||||
def test_simple_page_with_custom_fs() -> None:
|
||||
@pytest.mark.asyncio
|
||||
async def test_simple_page_with_custom_fs() -> None:
|
||||
parser = LlamaParse(result_type="markdown")
|
||||
fs = LocalFileSystem()
|
||||
filepath = os.path.join(
|
||||
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
|
||||
)
|
||||
result = parser.load_data(filepath, fs=fs)
|
||||
result = await parser.aload_data(filepath, fs=fs)
|
||||
assert len(result) == 1
|
||||
|
||||
|
||||
@@ -90,13 +92,14 @@ def test_simple_page_with_custom_fs() -> None:
|
||||
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
|
||||
reason="LLAMA_CLOUD_API_KEY not set",
|
||||
)
|
||||
def test_simple_page_progress_workers() -> None:
|
||||
@pytest.mark.asyncio
|
||||
async def test_simple_page_progress_workers() -> None:
|
||||
parser = LlamaParse(result_type="markdown", show_progress=True, verbose=True)
|
||||
|
||||
filepath = os.path.join(
|
||||
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
|
||||
)
|
||||
result = parser.load_data([filepath, filepath])
|
||||
result = await parser.aload_data([filepath, filepath])
|
||||
assert len(result) == 2
|
||||
assert len(result[0].text) > 0
|
||||
|
||||
@@ -107,7 +110,7 @@ def test_simple_page_progress_workers() -> None:
|
||||
filepath = os.path.join(
|
||||
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
|
||||
)
|
||||
result = parser.load_data([filepath, filepath])
|
||||
result = await parser.aload_data([filepath, filepath])
|
||||
assert len(result) == 2
|
||||
assert len(result[0].text) > 0
|
||||
|
||||
@@ -116,12 +119,79 @@ def test_simple_page_progress_workers() -> None:
|
||||
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
|
||||
reason="LLAMA_CLOUD_API_KEY not set",
|
||||
)
|
||||
def test_custom_client() -> None:
|
||||
@pytest.mark.asyncio
|
||||
async def test_custom_client() -> None:
|
||||
custom_client = AsyncClient(verify=False, timeout=10)
|
||||
parser = LlamaParse(result_type="markdown", custom_client=custom_client)
|
||||
filepath = os.path.join(
|
||||
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
|
||||
)
|
||||
result = parser.load_data(filepath)
|
||||
result = await parser.aload_data(filepath)
|
||||
assert len(result) == 1
|
||||
assert len(result[0].text) > 0
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
|
||||
reason="LLAMA_CLOUD_API_KEY not set",
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_input_url() -> None:
|
||||
parser = LlamaParse(result_type="markdown")
|
||||
|
||||
# links to a resume example
|
||||
input_url = "https://cdn-blog.novoresume.com/articles/google-docs-resume-templates/basic-google-docs-resume.png"
|
||||
result = await parser.aload_data(input_url)
|
||||
assert len(result) == 1
|
||||
assert "your name" in result[0].text.lower()
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
|
||||
reason="LLAMA_CLOUD_API_KEY not set",
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_input_url_with_website_input() -> None:
|
||||
parser = LlamaParse(result_type="markdown")
|
||||
input_url = "https://www.google.com"
|
||||
result = await parser.aload_data(input_url)
|
||||
assert len(result) == 1
|
||||
assert "google" in result[0].text.lower()
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
|
||||
reason="LLAMA_CLOUD_API_KEY not set",
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_mixing_input_types() -> None:
|
||||
parser = LlamaParse(result_type="markdown")
|
||||
filepath = os.path.join(
|
||||
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
|
||||
)
|
||||
input_url = "https://cdn-blog.novoresume.com/articles/google-docs-resume-templates/basic-google-docs-resume.png"
|
||||
result = await parser.aload_data([filepath, input_url])
|
||||
|
||||
assert len(result) == 2
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
|
||||
reason="LLAMA_CLOUD_API_KEY not set",
|
||||
)
|
||||
@pytest.mark.asyncio
|
||||
async def test_download_images() -> None:
|
||||
parser = LlamaParse(result_type="markdown", take_screenshot=True)
|
||||
filepath = os.path.join(
|
||||
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
|
||||
)
|
||||
json_result = await parser.aget_json([filepath])
|
||||
|
||||
assert len(json_result) == 1
|
||||
assert len(json_result[0]["pages"][0]["images"]) > 0
|
||||
|
||||
download_path = os.path.join(os.path.dirname(__file__), "test_files/images")
|
||||
shutil.rmtree(download_path, ignore_errors=True)
|
||||
|
||||
await parser.aget_images(json_result, download_path)
|
||||
assert len(os.listdir(download_path)) == len(json_result[0]["pages"][0]["images"])
|
||||
|
||||
Reference in New Issue
Block a user