Compare commits

..

1 Commits

Author SHA1 Message Date
Jerry Liu 0c440a81bf cr 2024-10-20 09:51:07 -07:00
4 changed files with 31 additions and 168 deletions
+10 -4
View File
@@ -7,6 +7,8 @@ assignees: ''
---
_Note: we're aware of some missing content in the output and layout issues on tables. Please refrain from opening new issues on this topic unless if you think it's different from what has already been reported._
**Describe the bug**
Write a concise description of what the bug is.
@@ -17,15 +19,19 @@ If possible, please provide the PDF file causing the issue.
If you have it, please provide the ID of the job you ran.
You can find it here: https://cloud.llamaindex.ai/parse in the "History" tab.
**Screenshots**
Feel free to also provide screenshots if relevant.
**Client:**
Please remove untested options:
- Python Library
- API
- Frontend (cloud.llamaindex.ai)
- Python Library
- Typescript Library
- Notebook
- API
**Options**
What options did you use? Multimodal, fast mode, parsing instructions, etc.
**Additional context**
Add any additional context about the problem here.
What options did you use? Premium mode, multimodal, fast mode, parsing instructions, etc.
Screenshots, code snippets, etc.
+13 -107
View File
@@ -1,6 +1,6 @@
import os
import asyncio
from urllib.parse import urlparse
from io import TextIOWrapper
import httpx
import mimetypes
@@ -11,7 +11,8 @@ from contextlib import asynccontextmanager
from io import BufferedIOBase
from fsspec import AbstractFileSystem
from llama_index.core.async_utils import asyncio_run, run_jobs
from fsspec.spec import AbstractBufferedFile
from llama_index.core.async_utils import run_jobs
from llama_index.core.bridge.pydantic import Field, field_validator
from llama_index.core.constants import DEFAULT_BASE_URL
from llama_index.core.readers.base import BasePydanticReader
@@ -94,10 +95,6 @@ class LlamaParse(BasePydanticReader):
default=False,
description="Use our best parser mode if set to True.",
)
continuous_mode: bool = Field(
default=False,
description="Parse documents continuously, leading to better results on documents where tables span across two pages.",
)
do_not_unroll_columns: Optional[bool] = Field(
default=False,
description="If set to true, the parser will keep column in the text according to document layout. Reduce reconstruction accuracy, and LLM's/embedings performances in most case.",
@@ -122,10 +119,6 @@ class LlamaParse(BasePydanticReader):
default=None,
description="The API key for the GPT-4o API. Lowers the cost of parsing.",
)
guess_xlsx_sheet_names: Optional[bool] = Field(
default=False,
description="Whether to guess the sheet names of the xlsx file.",
)
bounding_box: Optional[str] = Field(
default=None,
description="The bounding box to use to extract text from documents describe as a string containing the bounding box margins",
@@ -189,10 +182,6 @@ class LlamaParse(BasePydanticReader):
azure_openai_key: Optional[str] = Field(
default=None, description="Azure Openai Key"
)
http_proxy: Optional[str] = Field(
default=None,
description="(optional) If set with input_url will use the specified http proxy to download the file.",
)
@field_validator("api_key", mode="before", check_fields=True)
@classmethod
@@ -224,28 +213,6 @@ class LlamaParse(BasePydanticReader):
async with httpx.AsyncClient(timeout=self.max_timeout) as client:
yield client
def _is_input_url(self, file_path: FileInput) -> bool:
"""Check if the input is a valid URL.
This method checks for:
- Proper URL scheme (http/https)
- Valid URL structure
- Network location (domain)
"""
if not isinstance(file_path, str):
return False
try:
result = urlparse(file_path)
return all(
[
result.scheme in ("http", "https"),
result.netloc, # Has domain
result.scheme, # Has scheme
]
)
except Exception:
return False
# upload a document and get back a job_id
async def _create_job(
self,
@@ -257,7 +224,6 @@ class LlamaParse(BasePydanticReader):
url = f"{self.base_url}/api/parsing/upload"
files = None
file_handle = None
input_url = file_input if self._is_input_url(file_input) else None
if isinstance(file_input, (bytes, BufferedIOBase)):
if not extra_info or "file_name" not in extra_info:
@@ -267,8 +233,6 @@ class LlamaParse(BasePydanticReader):
file_name = extra_info["file_name"]
mime_type = mimetypes.guess_type(file_name)[0]
files = {"file": (file_name, file_input, mime_type)}
elif input_url is not None:
files = None
elif isinstance(file_input, (str, Path, PurePosixPath, PurePath)):
file_path = str(file_input)
file_ext = os.path.splitext(file_path)[1].lower()
@@ -296,7 +260,6 @@ class LlamaParse(BasePydanticReader):
"do_not_cache": self.do_not_cache,
"fast_mode": self.fast_mode,
"premium_mode": self.premium_mode,
"continuous_mode": self.continuous_mode,
"do_not_unroll_columns": self.do_not_unroll_columns,
"gpt4o_mode": self.gpt4o_mode,
"gpt4o_api_key": self.gpt4o_api_key,
@@ -305,10 +268,8 @@ class LlamaParse(BasePydanticReader):
"vendor_multimodal_model_name": self.vendor_multimodal_model_name,
"take_screenshot": self.take_screenshot,
"disable_ocr": self.disable_ocr,
"guess_xlsx_sheet_names": self.guess_xlsx_sheet_names,
"is_formatting_instruction": self.is_formatting_instruction,
"annotate_links": self.annotate_links,
"from_python_package": True,
}
# only send page separator to server if it is not None
@@ -344,13 +305,6 @@ class LlamaParse(BasePydanticReader):
if self.azure_openai_key is not None:
data["azure_openai_key"] = self.azure_openai_key
if input_url is not None:
files = None
data["input_url"] = str(input_url)
if self.http_proxy is not None:
data["http_proxy"] = self.http_proxy
try:
async with self.client_context() as client:
response = await client.post(
@@ -367,6 +321,12 @@ class LlamaParse(BasePydanticReader):
if file_handle is not None:
file_handle.close()
@staticmethod
def __get_filename(f: Union[TextIOWrapper, AbstractBufferedFile]) -> str:
if isinstance(f, TextIOWrapper):
return f.name
return f.full_name
async def _get_job_result(
self, job_id: str, result_type: str, verbose: bool = False
) -> Dict[str, Any]:
@@ -460,7 +420,7 @@ class LlamaParse(BasePydanticReader):
fs: Optional[AbstractFileSystem] = None,
) -> List[Document]:
"""Load data from the input path."""
if isinstance(file_path, (str, PurePosixPath, Path, bytes, BufferedIOBase)):
if isinstance(file_path, (str, Path, bytes, BufferedIOBase)):
return await self._aload_data(
file_path, extra_info=extra_info, fs=fs, verbose=self.verbose
)
@@ -502,7 +462,7 @@ class LlamaParse(BasePydanticReader):
) -> List[Document]:
"""Load data from the input path."""
try:
return asyncio_run(self.aload_data(file_path, extra_info, fs=fs))
return asyncio.run(self.aload_data(file_path, extra_info, fs=fs))
except RuntimeError as e:
if nest_asyncio_err in str(e):
raise RuntimeError(nest_asyncio_msg)
@@ -569,7 +529,7 @@ class LlamaParse(BasePydanticReader):
) -> List[dict]:
"""Parse the input path."""
try:
return asyncio_run(self.aget_json(file_path, extra_info))
return asyncio.run(self.aget_json(file_path, extra_info))
except RuntimeError as e:
if nest_asyncio_err in str(e):
raise RuntimeError(nest_asyncio_msg)
@@ -632,61 +592,7 @@ class LlamaParse(BasePydanticReader):
def get_images(self, json_result: List[dict], download_path: str) -> List[dict]:
"""Download images from the parsed result."""
try:
return asyncio_run(self.aget_images(json_result, download_path))
except RuntimeError as e:
if nest_asyncio_err in str(e):
raise RuntimeError(nest_asyncio_msg)
else:
raise e
async def aget_xlsx(
self, json_result: List[dict], download_path: str
) -> List[dict]:
"""Download images from the parsed result."""
headers = {"Authorization": f"Bearer {self.api_key}"}
# make the download path
if not os.path.exists(download_path):
os.makedirs(download_path)
try:
xlsx_list = []
for result in json_result:
job_id = result["job_id"]
if self.verbose:
print("> XLSX")
xlsx_path = os.path.join(download_path, f"{job_id}.xlsx")
xlsx = {}
xlsx["path"] = xlsx_path
xlsx["job_id"] = job_id
xlsx["original_file_path"] = result.get("file_path", None)
with open(xlsx_path, "wb") as f:
xlsx_url = (
f"{self.base_url}/api/parsing/job/{job_id}/result/raw/xlsx"
)
async with self.client_context() as client:
res = await client.get(
xlsx_url, headers=headers, timeout=self.max_timeout
)
res.raise_for_status()
f.write(res.content)
xlsx_list.append(xlsx)
return xlsx_list
except Exception as e:
print("Error while downloading xlsx:", e)
if self.ignore_errors:
return []
else:
raise e
def get_xlsx(self, json_result: List[dict], download_path: str) -> List[dict]:
"""Download xlsx from the parsed result."""
try:
return asyncio_run(self.aget_xlsx(json_result, download_path))
return asyncio.run(self.aget_images(json_result, download_path))
except RuntimeError as e:
if nest_asyncio_err in str(e):
raise RuntimeError(nest_asyncio_msg)
+1 -1
View File
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
[tool.poetry]
name = "llama-parse"
version = "0.5.14"
version = "0.5.10"
description = "Parse files into RAG-Optimized formats."
authors = ["Logan Markewich <logan@llamaindex.ai>"]
license = "MIT"
+7 -56
View File
@@ -76,14 +76,13 @@ def test_simple_page_markdown_buffer(markdown_parser: LlamaParse) -> None:
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.asyncio
async def test_simple_page_with_custom_fs() -> None:
def test_simple_page_with_custom_fs() -> None:
parser = LlamaParse(result_type="markdown")
fs = LocalFileSystem()
filepath = os.path.join(
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
)
result = await parser.aload_data(filepath, fs=fs)
result = parser.load_data(filepath, fs=fs)
assert len(result) == 1
@@ -91,14 +90,13 @@ async def test_simple_page_with_custom_fs() -> None:
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.asyncio
async def test_simple_page_progress_workers() -> None:
def test_simple_page_progress_workers() -> None:
parser = LlamaParse(result_type="markdown", show_progress=True, verbose=True)
filepath = os.path.join(
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
)
result = await parser.aload_data([filepath, filepath])
result = parser.load_data([filepath, filepath])
assert len(result) == 2
assert len(result[0].text) > 0
@@ -109,7 +107,7 @@ async def test_simple_page_progress_workers() -> None:
filepath = os.path.join(
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
)
result = await parser.aload_data([filepath, filepath])
result = parser.load_data([filepath, filepath])
assert len(result) == 2
assert len(result[0].text) > 0
@@ -118,59 +116,12 @@ async def test_simple_page_progress_workers() -> None:
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.asyncio
async def test_custom_client() -> None:
def test_custom_client() -> None:
custom_client = AsyncClient(verify=False, timeout=10)
parser = LlamaParse(result_type="markdown", custom_client=custom_client)
filepath = os.path.join(
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
)
result = await parser.aload_data(filepath)
result = parser.load_data(filepath)
assert len(result) == 1
assert len(result[0].text) > 0
@pytest.mark.skipif(
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.asyncio
async def test_input_url() -> None:
parser = LlamaParse(result_type="markdown")
# links to a resume example
input_url = "https://cdn-blog.novoresume.com/articles/google-docs-resume-templates/basic-google-docs-resume.png"
result = await parser.aload_data(input_url)
assert len(result) == 1
assert "your name" in result[0].text.lower()
@pytest.mark.skipif(
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.asyncio
async def test_input_url_with_website_input() -> None:
parser = LlamaParse(result_type="markdown")
input_url = "https://www.google.com"
result = await parser.aload_data(input_url)
assert len(result) == 1
assert "google" in result[0].text.lower()
@pytest.mark.skipif(
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.asyncio
async def test_mixing_input_types() -> None:
parser = LlamaParse(result_type="markdown")
filepath = os.path.join(
os.path.dirname(__file__), "test_files/attention_is_all_you_need.pdf"
)
input_url = "https://www.google.com"
result = await parser.aload_data([filepath, input_url])
assert len(result) == 2
assert "table 2" in result[0].text.lower()
assert "google" in result[1].text.lower()