mirror of
https://github.com/run-llama/llama_cloud_services.git
synced 2026-07-20 19:47:38 -04:00
Compare commits
6 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 951ba4dfd8 | |||
| 386d210e8b | |||
| 9321602845 | |||
| 26c06353f0 | |||
| 62cf12d6eb | |||
| 253ee61463 |
@@ -38,7 +38,22 @@ Lastly, install the package:
|
||||
|
||||
`pip install llama-parse`
|
||||
|
||||
Now you can run the following to parse your first PDF file:
|
||||
Now you can parse your first PDF file using the command line interface. Use the command `llama-parse [file_paths]`. See the help text with `llama-parse --help`.
|
||||
|
||||
```bash
|
||||
export LLAMA_CLOUD_API_KEY='llx-...'
|
||||
|
||||
# output as text
|
||||
llama-parse my_file.pdf --result-type text --output-file output.txt
|
||||
|
||||
# output as markdown
|
||||
llama-parse my_file.pdf --result-type markdown --output-file output.md
|
||||
|
||||
# output as raw json
|
||||
llama-parse my_file.pdf --output-raw-json --output-file output.json
|
||||
```
|
||||
|
||||
You can also create simple scripts:
|
||||
|
||||
```python
|
||||
import nest_asyncio
|
||||
|
||||
@@ -342,7 +342,7 @@
|
||||
],
|
||||
"metadata": {
|
||||
"kernelspec": {
|
||||
"display_name": "llama-parse-aNC435Vv-py3.10",
|
||||
"display_name": "Python 3 (ipykernel)",
|
||||
"language": "python",
|
||||
"name": "python3"
|
||||
},
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
Binary file not shown.
|
After Width: | Height: | Size: 580 KiB |
File diff suppressed because it is too large
Load Diff
Binary file not shown.
|
After Width: | Height: | Size: 986 KiB |
@@ -46,7 +46,7 @@
|
||||
"metadata": {},
|
||||
"outputs": [],
|
||||
"source": [
|
||||
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"<LLAMA_CLOUD_API_KEY>"
|
||||
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"<LLAMA_CLOUD_API_KEY>\""
|
||||
]
|
||||
},
|
||||
{
|
||||
|
||||
+57
-1
@@ -154,6 +154,34 @@ class LlamaParse(BasePydanticReader):
|
||||
custom_client: Optional[httpx.AsyncClient] = Field(
|
||||
default=None, description="A custom HTTPX client to use for sending requests."
|
||||
)
|
||||
disable_ocr: bool = Field(
|
||||
default=False,
|
||||
description="Disable the OCR on the document. LlamaParse will only extract the copyable text from the document.",
|
||||
)
|
||||
is_formatting_instruction: bool = Field(
|
||||
default=True,
|
||||
description="Allow the parsing instruction to also format the output. Disable to have a cleaner markdown output.",
|
||||
)
|
||||
annotate_links: bool = Field(
|
||||
default=False,
|
||||
description="Annotate links found in the document to extract their URL.",
|
||||
)
|
||||
webhook_url: Optional[str] = Field(
|
||||
default=None,
|
||||
description="A URL that needs to be called at the end of the parsing job.",
|
||||
)
|
||||
azure_openai_deployment_name: Optional[str] = Field(
|
||||
default=None, description="Azure Openai Deployment Name"
|
||||
)
|
||||
azure_openai_endpoint: Optional[str] = Field(
|
||||
default=None, description="Azure Openai Endpoint"
|
||||
)
|
||||
azure_openai_api_version: Optional[str] = Field(
|
||||
default=None, description="Azure Openai API Version"
|
||||
)
|
||||
azure_openai_key: Optional[str] = Field(
|
||||
default=None, description="Azure Openai Key"
|
||||
)
|
||||
|
||||
@field_validator("api_key", mode="before", check_fields=True)
|
||||
@classmethod
|
||||
@@ -239,6 +267,9 @@ class LlamaParse(BasePydanticReader):
|
||||
"use_vendor_multimodal_model": self.use_vendor_multimodal_model,
|
||||
"vendor_multimodal_model_name": self.vendor_multimodal_model_name,
|
||||
"take_screenshot": self.take_screenshot,
|
||||
"disable_ocr": self.disable_ocr,
|
||||
"is_formatting_instruction": self.is_formatting_instruction,
|
||||
"annotate_links": self.annotate_links,
|
||||
}
|
||||
|
||||
# only send page separator to server if it is not None
|
||||
@@ -258,6 +289,22 @@ class LlamaParse(BasePydanticReader):
|
||||
if self.target_pages is not None:
|
||||
data["target_pages"] = self.target_pages
|
||||
|
||||
if self.webhook_url is not None:
|
||||
data["webhook_url"] = self.webhook_url
|
||||
|
||||
# Azure OpenAI
|
||||
if self.azure_openai_deployment_name is not None:
|
||||
data["azure_openai_deployment_name"] = self.azure_openai_deployment_name
|
||||
|
||||
if self.azure_openai_endpoint is not None:
|
||||
data["azure_openai_endpoint"] = self.azure_openai_endpoint
|
||||
|
||||
if self.azure_openai_api_version is not None:
|
||||
data["azure_openai_api_version"] = self.azure_openai_api_version
|
||||
|
||||
if self.azure_openai_key is not None:
|
||||
data["azure_openai_key"] = self.azure_openai_key
|
||||
|
||||
try:
|
||||
async with self.client_context() as client:
|
||||
response = await client.post(
|
||||
@@ -308,7 +355,8 @@ class LlamaParse(BasePydanticReader):
|
||||
continue
|
||||
|
||||
# Allowed values "PENDING", "SUCCESS", "ERROR", "CANCELED"
|
||||
status = result.json()["status"]
|
||||
result_json = result.json()
|
||||
status = result_json["status"]
|
||||
if status == "SUCCESS":
|
||||
parsed_result = await client.get(result_url, headers=headers)
|
||||
return parsed_result.json()
|
||||
@@ -320,6 +368,14 @@ class LlamaParse(BasePydanticReader):
|
||||
print(".", end="", flush=True)
|
||||
|
||||
await asyncio.sleep(self.check_interval)
|
||||
else:
|
||||
error_code = result_json.get("error_code", "No error code found")
|
||||
error_message = result_json.get(
|
||||
"error_message", "No error message found"
|
||||
)
|
||||
|
||||
exception_str = f"Job ID: {job_id} failed with status: {status}, Error code: {error_code}, Error message: {error_message}"
|
||||
raise Exception(exception_str)
|
||||
|
||||
async def _aload_data(
|
||||
self,
|
||||
|
||||
@@ -0,0 +1,92 @@
|
||||
import click
|
||||
import json
|
||||
from enum import Enum
|
||||
from pathlib import Path
|
||||
from pydantic.fields import FieldInfo
|
||||
from typing import Any, Callable, List
|
||||
|
||||
from llama_parse.base import LlamaParse
|
||||
|
||||
|
||||
def pydantic_field_to_click_option(name: str, field: FieldInfo) -> click.Option:
|
||||
"""Convert a Pydantic field to a Click option."""
|
||||
kwargs = {
|
||||
"default": field.default if field.default else None,
|
||||
"help": field.description,
|
||||
}
|
||||
|
||||
if isinstance(kwargs["default"], Enum):
|
||||
kwargs["default"] = kwargs["default"].value
|
||||
|
||||
if field.annotation is bool:
|
||||
kwargs["is_flag"] = True
|
||||
if field.default and field.default is True:
|
||||
name = f"no-{name}"
|
||||
return click.option(f'--{name.replace("_", "-")}', **kwargs)
|
||||
|
||||
|
||||
def add_options(options: List[click.Option]) -> Callable:
|
||||
def _add_options(func: Callable) -> Callable:
|
||||
for option in reversed(options):
|
||||
func = option(func)
|
||||
return func
|
||||
|
||||
return _add_options
|
||||
|
||||
|
||||
@click.command()
|
||||
@click.argument("file_paths", nargs=-1, type=click.Path(exists=True, path_type=Path))
|
||||
@click.option(
|
||||
"--output-file", type=click.Path(path_type=Path), help="Path to save the output"
|
||||
)
|
||||
@click.option("--output-raw-json", is_flag=True, help="Output the raw JSON result")
|
||||
@add_options(
|
||||
[
|
||||
pydantic_field_to_click_option(name, field)
|
||||
for name, field in LlamaParse.model_fields.items()
|
||||
if name not in ["custom_client"]
|
||||
]
|
||||
)
|
||||
def parse(**kwargs: Any) -> None:
|
||||
"""Parse files using LlamaParse and output the results."""
|
||||
file_paths = kwargs.pop("file_paths")
|
||||
output_file = kwargs.pop("output_file")
|
||||
output_raw_json = kwargs.pop("output_raw_json")
|
||||
|
||||
# Remove None values to use LlamaParse defaults
|
||||
kwargs = {k: v for k, v in kwargs.items() if v is not None}
|
||||
|
||||
# Remove no- prefix for boolean flags
|
||||
kwargs = {k.replace("no_", ""): v for k, v in kwargs.items()}
|
||||
|
||||
parser = LlamaParse(**kwargs)
|
||||
if output_raw_json:
|
||||
results = parser.get_json_result(list(file_paths))
|
||||
|
||||
if output_file:
|
||||
with output_file.open("w") as f:
|
||||
json.dump(results, f)
|
||||
click.echo(f"Results saved to {output_file}")
|
||||
else:
|
||||
click.echo(results)
|
||||
else:
|
||||
results = parser.load_data(list(file_paths))
|
||||
|
||||
if output_file:
|
||||
with output_file.open("w") as f:
|
||||
for i, doc in enumerate(results):
|
||||
f.write(f"File: {doc.metadata.get('file_path', 'Unknown')}\n") # type: ignore
|
||||
f.write(doc.text) # type: ignore
|
||||
if i < len(results) - 1:
|
||||
f.write("\n\n---\n\n")
|
||||
click.echo(f"Results saved to {output_file}")
|
||||
else:
|
||||
for i, doc in enumerate(results):
|
||||
click.echo(f"File: {doc.metadata.get('file_path', 'Unknown')}") # type: ignore
|
||||
click.echo(doc.text) # type: ignore
|
||||
if i < len(results) - 1:
|
||||
click.echo("\n---\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parse()
|
||||
Generated
+1020
-821
File diff suppressed because it is too large
Load Diff
+5
-1
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
|
||||
|
||||
[tool.poetry]
|
||||
name = "llama-parse"
|
||||
version = "0.5.6"
|
||||
version = "0.5.10"
|
||||
description = "Parse files into RAG-Optimized formats."
|
||||
authors = ["Logan Markewich <logan@llamaindex.ai>"]
|
||||
license = "MIT"
|
||||
@@ -14,7 +14,11 @@ packages = [{include = "llama_parse"}]
|
||||
[tool.poetry.dependencies]
|
||||
python = ">=3.8.1,<4.0"
|
||||
llama-index-core = ">=0.11.0"
|
||||
click = "^8.1.7"
|
||||
|
||||
[tool.poetry.group.dev.dependencies]
|
||||
pytest = "^8.0.0"
|
||||
ipykernel = "^6.29.0"
|
||||
|
||||
[tool.poetry.scripts]
|
||||
llama-parse = "llama_parse.cli.main:parse"
|
||||
|
||||
Reference in New Issue
Block a user