Compare commits

..

6 Commits

Author SHA1 Message Date
Pierre-Loic Doulcet f3ee205456 add type hint for FailedPageMode and ParsingMode 2025-04-24 10:47:20 +08:00
Pierre-Loic Doulcet 9cbce746bf add type hint for FailedPageMode and ParsingMode 2025-04-24 10:45:38 +08:00
Pierre-Loic Doulcet fcc8f4f566 update description 2025-04-24 10:30:11 +08:00
Pierre-Loic Doulcet 2f3a5ce2ac mergefix 2025-04-24 09:45:51 +08:00
Pierre-Loic Doulcet 0943625579 add page_error_parameters 2025-04-24 09:44:51 +08:00
Pierre-Loic Doulcet 2192ad4a8e add compact md tables 2025-04-09 12:38:23 -07:00
33 changed files with 912 additions and 19532 deletions
-11
View File
@@ -1,11 +0,0 @@
# Please see the documentation for all configuration options:
# https://docs.github.com/github/administering-a-repository/configuration-options-for-dependency-updates
# and
# https://docs.github.com/code-security/dependabot/dependabot-version-updates/configuration-options-for-the-dependabot.yml-file
version: 2
updates:
- package-ecosystem: "github-actions"
directory: "/"
schedule:
interval: "weekly"
+2 -2
View File
@@ -21,9 +21,9 @@ jobs:
os: [ubuntu-latest, windows-latest]
python-version: ["3.9"]
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v3
- name: Set up python ${{ matrix.python-version }}
uses: actions/setup-python@v5
uses: actions/setup-python@v4
with:
python-version: ${{ matrix.python-version }}
- name: Install Poetry
+48 -8
View File
@@ -1,3 +1,14 @@
# For most projects, this workflow file will not need changing; you simply need
# to commit it to your repository.
#
# You may wish to alter this file to override the set of languages analyzed,
# or to provide custom queries or build logic.
#
# ******** NOTE ********
# We have attempted to detect the languages in your repository. Please check
# the `language` matrix defined below to confirm you have the correct set of
# supported CodeQL languages.
#
name: "CodeQL"
on:
@@ -17,25 +28,54 @@ jobs:
# - https://gh.io/supported-runners-and-hardware-resources
# - https://gh.io/using-larger-runners
# Consider using larger runners for possible analysis time improvements.
runs-on: "ubuntu-latest"
timeout-minutes: 360
runs-on: ${{ (matrix.language == 'swift' && 'macos-latest') || 'ubuntu-latest' }}
timeout-minutes: ${{ (matrix.language == 'swift' && 120) || 360 }}
permissions:
actions: read
contents: read
security-events: write
strategy:
fail-fast: false
matrix:
language: ["python"]
# CodeQL supports [ 'cpp', 'csharp', 'go', 'java', 'javascript', 'python', 'ruby', 'swift' ]
# Use only 'java' to analyze code written in Java, Kotlin or both
# Use only 'javascript' to analyze code written in JavaScript, TypeScript or both
# Learn more about CodeQL language support at https://aka.ms/codeql-docs/language-support
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@v3
# Initializes the CodeQL tools for scanning.
- name: Initialize CodeQL
uses: github/codeql-action/init@v3
uses: github/codeql-action/init@v2
with:
languages: python
dependency-caching: true
languages: ${{ matrix.language }}
# If you wish to specify custom queries, you can do so here or in a config file.
# By default, queries listed here will override any specified in a config file.
# Prefix the list here with "+" to use these queries and those in the config file.
# For more details on CodeQL's query packs, refer to: https://docs.github.com/en/code-security/code-scanning/automatically-scanning-your-code-for-vulnerabilities-and-errors/configuring-code-scanning#using-queries-in-ql-packs
# queries: security-extended,security-and-quality
# Autobuild attempts to build any compiled languages (C/C++, C#, Go, Java, or Swift).
# If this step fails, then you should remove it and run the build manually (see below)
- name: Autobuild
uses: github/codeql-action/autobuild@v2
# ️ Command-line programs to run using the OS shell.
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
# If the Autobuild fails above, remove it and uncomment the following three lines.
# modify them (or add more) to build your code if your project, please refer to the EXAMPLE below for guidance.
# - run: |
# echo "Run, Build Application using script"
# ./location_of_script_within_repo/buildscript.sh
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@v3
uses: github/codeql-action/analyze@v2
with:
category: "/language:python"
category: "/language:${{matrix.language}}"
+2 -2
View File
@@ -18,11 +18,11 @@ jobs:
matrix:
python-version: ["3.9"]
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v3
with:
fetch-depth: ${{ github.event_name == 'pull_request' && 2 || 0 }}
- name: Set up python ${{ matrix.python-version }}
uses: actions/setup-python@v5
uses: actions/setup-python@v4
with:
python-version: ${{ matrix.python-version }}
- name: Install Poetry
+3 -11
View File
@@ -18,9 +18,9 @@ jobs:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v3
- name: Set up python ${{ env.PYTHON_VERSION }}
uses: actions/setup-python@v5
uses: actions/setup-python@v4
with:
python-version: ${{ env.PYTHON_VERSION }}
@@ -39,18 +39,10 @@ jobs:
pypi_token: ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
poetry_install_options: "--without dev"
- name: Wait for PyPI to update
run: |
sleep 120
- name: Update llama-parse lock file
run: |
cd llama_parse && poetry lock
- name: Build and publish llama-parse
uses: JRubics/poetry-publish@v2.1
with:
package_directory: "./llama_parse"
working_directory: "llama_parse"
pypi_token: ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
poetry_install_options: "--without dev"
+2 -2
View File
@@ -19,11 +19,11 @@ jobs:
matrix:
python-version: ["3.9", "3.10", "3.11", "3.12"]
steps:
- uses: actions/checkout@v4
- uses: actions/checkout@v3
with:
fetch-depth: 0
- name: Set up python ${{ matrix.python-version }}
uses: actions/setup-python@v5
uses: actions/setup-python@v4
with:
python-version: ${{ matrix.python-version }}
- name: Install Poetry
File diff suppressed because it is too large Load Diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 3.3 MiB

@@ -1 +0,0 @@
sec_form_4_dump.json
File diff suppressed because it is too large Load Diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 202 KiB

@@ -1,440 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Extract Data from Financial Reports - with Citations and Reasoning\n",
"\n",
"Given complex files like financial reports, contracts, invoices etc, Llama Extract allows you to make use of an LLM to extract the information relevant to you, in a structured format.\n",
"\n",
"In this example, we'll be using [LlamaExtract](https://docs.cloud.llamaindex.ai/llamaextract/getting_started?utm_campaign=extract&utm_medium=recipe) to extract structured data from an SEC filing (specifically, the filing by Nvidia for fiscal year 2025).\n",
"\n",
"On top of simple data extraction, we'll ask our extraction agent to provide citations and reasoning for each extracted field. This allows us to:\n",
"- Confirm the accuracy of the extracted field\n",
"- Understand the reasoning behind why the LLM extracted a given piece of information\n",
"- This last point allows us an opportunity to adjust the system prompt or field descriptions and improve on results where needed.\n",
"\n",
"\n",
"The example we go through below is also replicable within Llama Cloud as well, where you will also be able to pick between a number of pre-defined schemas, instead of building your own."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install llama-cloud-services"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Connect to Llama Cloud\n",
"\n",
"To get started, make sure you provide your [Llama Cloud](https://cloud.llamaindex.ai?utm_campaign=extract&utm_medium=recipe) API key."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Enter your Llama Cloud API Key: ··········\n"
]
}
],
"source": [
"import os\n",
"from getpass import getpass\n",
"\n",
"if \"LLAMA_CLOUD_API_KEY\" not in os.environ:\n",
" os.environ[\"LLAMA_CLOUD_API_KEY\"] = getpass(\"Enter your Llama Cloud API Key: \")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Extract Data with Llama Extract Agent"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"No project_id provided, fetching default project.\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaExtract\n",
"\n",
"# Optionally, provide your project id, if not, it will use the 'Default' project\n",
"llama_extract = LlamaExtract()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Provide Your Custom Schema\n",
"\n",
"When using LlamaExtract via the API, you provide your own schema that describes what you want extracted from files and data provided to your agent. Here, we are essentially building an SEC filings extraction agent."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pydantic import BaseModel, Field\n",
"from enum import Enum\n",
"\n",
"\n",
"class FilingType(str, Enum):\n",
" ten_k = \"10 K\"\n",
" ten_q = \"10-Q\"\n",
" ten_ka = \"10-K/A\"\n",
" ten_qa = \"10-Q/A\"\n",
"\n",
"\n",
"class FinancialReport(BaseModel):\n",
" company_name: str = Field(description=\"The name of the company\")\n",
" description: str = Field(\n",
" description=\"Short description of the filing and what it contains\"\n",
" )\n",
" filing_type: FilingType = Field(description=\"Type of SEC filing\")\n",
" filing_date: str = Field(description=\"Date when filing was submitted to SEC\")\n",
" fiscal_year: int = Field(description=\"Fiscal year\")\n",
" unit: str = Field(\n",
" description=\"Unit of financial figures (thousands, millions, etc.)\"\n",
" )\n",
" revenue: int = Field(description=\"Total revenue for period\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Set Up Citations and Reasoning\n",
"\n",
"Optionally, we can set the `ExtractConfig` to extract citations for each field the agent extracts. These cications will cite the specific pages and sections of the file from which a given field was extractedd.\n",
"\n",
"By setting `use_reasoning` to True, we als ask the agent to do an additional reasoning step, explaining why a given field was extracted."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud.types import ExtractConfig, ExtractMode\n",
"\n",
"config = ExtractConfig(\n",
" use_reasoning=True, cite_sources=True, extraction_mode=ExtractMode.MULTIMODAL\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"/usr/local/lib/python3.11/dist-packages/llama_cloud_services/extract/extract.py:127: ExperimentalWarning: `use_reasoning` is an experimental feature. Results will be available in the `extraction_metadata` field for the extraction run.\n",
" warnings.warn(\n",
"/usr/local/lib/python3.11/dist-packages/llama_cloud_services/extract/extract.py:133: ExperimentalWarning: `cite_sources` is an experimental feature. This may greatly increase the size of the response, and slow down the extraction. Results will be available in the `extraction_metadata` field for the extraction run.\n",
" warnings.warn(\n"
]
}
],
"source": [
"agent = llama_extract.create_agent(\n",
" name=\"filing-parser\", data_schema=FinancialReport, config=config\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Demo Time - Download a PDF and Extract Data with Citations"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"PDF downloaded successfully.\n"
]
}
],
"source": [
"import requests\n",
"\n",
"url = \"https://raw.githubusercontent.com/run-llama/llama_cloud_services/refs/heads/main/examples/extract/data/sec_filings/nvda_10k.pdf\"\n",
"\n",
"response = requests.get(url)\n",
"\n",
"if response.status_code == 200:\n",
" with open(\"/content/nvda_10k.pdf\", \"wb\") as f:\n",
" f.write(response.content)\n",
" print(\"PDF downloaded successfully.\")\n",
"else:\n",
" print(f\"Failed to download. Status code: {response.status_code}\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"Uploading files: 100%|██████████| 1/1 [00:00<00:00, 1.83it/s]\n",
"Creating extraction jobs: 100%|██████████| 1/1 [00:00<00:00, 4.38it/s]\n",
"Extracting files: 100%|██████████| 1/1 [02:03<00:00, 123.40s/it]\n"
]
}
],
"source": [
"filing_info = agent.extract(\"/content/nvda_10k.pdf\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'company_name': 'NVIDIA Corporation',\n",
" 'description': \"The filing provides a detailed overview of NVIDIA's business as a full-stack computing infrastructure company, discusses various technologies including digital avatars and autonomous vehicles, outlines numerous risk factors affecting operations such as supply chain issues and geopolitical tensions, and describes employee stock purchase plans and related compliance requirements.\",\n",
" 'filing_type': '10 K',\n",
" 'filing_date': 'February 26, 2025',\n",
" 'fiscal_year': 2025,\n",
" 'unit': 'millions',\n",
" 'revenue': 130497}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"filing_info.data"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Inspect Citations and Reasoning"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'field_metadata': {'company_name': {'reasoning': 'VERBATIM EXTRACTION',\n",
" 'citation': [{'page': 1, 'matching_text': 'NVIDIA CORPORATION'},\n",
" {'page': 2, 'matching_text': 'NVIDIA Corporation'},\n",
" {'page': 3,\n",
" 'matching_text': 'All references to \"NVIDIA,\" \"we,\" \"us,\" \"our,\" or the \"Company\" mean NVIDIA Corporation and its subsidiaries.'},\n",
" {'page': 35,\n",
" 'matching_text': 'Comparison of 5 Year Cumulative Total Return* Among NVIDIA Corporation'},\n",
" {'page': 49,\n",
" 'matching_text': 'To the Board of Directors and Shareholders of NVIDIA Corporation'},\n",
" {'page': 90, 'matching_text': 'NVIDIA Corporation'},\n",
" {'page': 119,\n",
" 'matching_text': '*\"Company\"* means NVIDIA Corporation, a Delaware corporation.'},\n",
" {'page': 126,\n",
" 'matching_text': 'Annual Report on Form 10-K of NVIDIA Corporation'}]},\n",
" 'filing_type': {'reasoning': \"VERBATIM EXTRACTION from multiple sources confirming the filing type as '10 K'.\",\n",
" 'citation': [{'page': 1, 'matching_text': 'FORM 10-K'},\n",
" {'page': 2, 'matching_text': 'Item 16. | Form 10-K Summary'},\n",
" {'page': 3,\n",
" 'matching_text': 'This Annual Report on Form 10-K contains forward-looking statements...'},\n",
" {'page': 13, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 15, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 32,\n",
" 'matching_text': 'Annual Report on Form 10-K, which information is hereby incorporated by reference.'},\n",
" {'page': 36, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 43,\n",
" 'matching_text': 'Annual Report on Form 10-K for additional information'},\n",
" {'page': 45, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 46, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 62, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 83,\n",
" 'matching_text': 'Restated Certificate of Incorporation | 10-K'},\n",
" {'page': 84, 'matching_text': 'Item 16. Form 10-K Summary'},\n",
" {'page': 126, 'matching_text': 'which appears in this Form 10-K'},\n",
" {'page': 127, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 128, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 129, 'matching_text': \"The Company's Annual Report on Form 10-K\"},\n",
" {'page': 130,\n",
" 'matching_text': \"The Company's Annual Report on Form 10-K for the year ended January 26, 2025\"}]},\n",
" 'fiscal_year': {'reasoning': 'The fiscal year ended January 26, 2025, indicates the fiscal year is 2025. Additionally, multiple references throughout the text confirm the fiscal year 2025 in various contexts.',\n",
" 'citation': [{'page': 1,\n",
" 'matching_text': 'For the fiscal year ended January 26, 2025'},\n",
" {'page': 6,\n",
" 'matching_text': 'In fiscal year 2025, we launched the NVIDIA Blackwell architecture'},\n",
" {'page': 12, 'matching_text': 'fiscal year 2025'},\n",
" {'page': 17,\n",
" 'matching_text': 'our gross margins in the second quarter of fiscal year 2025 were negatively impacted'},\n",
" {'page': 20,\n",
" 'matching_text': 'we generated 53% of our revenue in fiscal year 2025 from sales outside the United States.'},\n",
" {'page': 23,\n",
" 'matching_text': 'For fiscal year 2025, an indirect customer which primarily purchases our products through system integrators...'},\n",
" {'page': 33,\n",
" 'matching_text': 'In fiscal year 2025, we repurchased 310 million shares of our common stock for $34.0 billion.'},\n",
" {'page': 37,\n",
" 'matching_text': 'Our Data Center revenue in China grew in fiscal year 2025.'},\n",
" {'page': 44,\n",
" 'matching_text': 'Cash provided by operating activities increased in fiscal year 2025 compared to fiscal year 2024'},\n",
" {'page': 57,\n",
" 'matching_text': 'Fiscal years 2025, 2024 and 2023 were all 52-week years.'},\n",
" {'page': 65,\n",
" 'matching_text': 'Beginning in the second quarter of fiscal year 2025'},\n",
" {'page': 69, 'matching_text': 'In the fourth quarter of fiscal year 2025'},\n",
" {'page': 78,\n",
" 'matching_text': 'Depreciation and amortization expense attributable to our Compute and Networking segment for fiscal years 2025'},\n",
" {'page': 129, 'matching_text': 'for the year ended January 26, 2025'}]},\n",
" 'description': {'reasoning': 'The extracted data combines multiple descriptions from the source text, ensuring no duplication while maintaining the order and context of the information. Each section of the filing is summarized to reflect the key points without losing the essence of the original text.',\n",
" 'citation': [{'page': 4,\n",
" 'matching_text': 'NVIDIA is now a full-stack computing infrastructure company with data-center-scale offerings that are reshaping industry.'},\n",
" {'page': 8,\n",
" 'matching_text': 'a suite of technologies that help developers bring digital avatars to life with generative Al...autonomous vehicles, or AV, and electric vehicles, or EV, is revolutionizing the transportation industry...Our worldwide sales and marketing strategy is key to achieving our objective of providing markets with our high-performance and efficient computing platforms and software.'},\n",
" {'page': 14, 'matching_text': 'Risk Factors Summary'},\n",
" {'page': 16,\n",
" 'matching_text': 'Risks Related to Demand, Supply, and Manufacturing\\n\\nLong manufacturing lead times and uncertain supply and component availability...'},\n",
" {'page': 18,\n",
" 'matching_text': 'cryptocurrency mining, on demand for our products. Volatility in the cryptocurrency market, including new compute technologies...'},\n",
" {'page': 21,\n",
" 'matching_text': 'supply-chain attacks or other business disruptions. We cannot guarantee that third parties and infrastructure in our supply chain...'},\n",
" {'page': 22,\n",
" 'matching_text': 'We are monitoring the impact of the geopolitical conflict in and around Israel on our operations... Climate change may have a long-term impact on our business.'},\n",
" {'page': 25,\n",
" 'matching_text': 'We are subject to complex laws, rules, regulations, and political and other actions, including restrictions on the export of our products, which may adversely impact our business.'},\n",
" {'page': 28,\n",
" 'matching_text': 'Our competitive position has been harmed by the existing export controls, and our competitive position and future results may be further harmed'},\n",
" {'page': 29,\n",
" 'matching_text': 'restrictions imposed by the Chinese government on the duration of gaming activities and access to games may adversely affect our Gaming revenue'},\n",
" {'page': 29,\n",
" 'matching_text': 'our business depends on our ability to receive consistent and reliable supply from our overseas partners, especially in Taiwan and South Korea'},\n",
" {'page': 29,\n",
" 'matching_text': 'Increased scrutiny from shareholders, regulators and others regarding our corporate sustainability practices could result in additional costs'},\n",
" {'page': 29,\n",
" 'matching_text': 'Concerns relating to the responsible use of new and evolving technologies, such as Al, in our products and services may result in reputational or financial harm'},\n",
" {'page': 31,\n",
" 'matching_text': 'Data protection laws around the world are quickly changing and may be interpreted and applied in an increasingly stringent fashion...'}]},\n",
" 'filing_date': {'reasoning': 'The filing date is consistently mentioned as February 26, 2025 across multiple entries, making it the most reliable date for the filing.',\n",
" 'citation': [{'page': 51, 'matching_text': 'February 26, 2025'},\n",
" {'page': 86, 'matching_text': 'on February 26, 2025.'},\n",
" {'page': 87, 'matching_text': 'February 26, 2025'},\n",
" {'page': 126, 'matching_text': 'our report dated February 26, 2025'},\n",
" {'page': 127, 'matching_text': 'Date: February 26, 2025'},\n",
" {'page': 128, 'matching_text': 'Date: February 26, 2025'},\n",
" {'page': 129, 'matching_text': 'Date: February 26, 2025'},\n",
" {'page': 130, 'matching_text': 'Date: February 26, 2025'}]},\n",
" 'unit': {'reasoning': \"The unit of financial figures is explicitly mentioned multiple times in the text as 'millions', including in table headers and notes. This is confirmed by various citations from pages 38, 42, 43, 52, 53, 54, 56, 65, 71, 72, 73, 75, 77, 79, 80, and 82.\",\n",
" 'citation': [{'page': 38,\n",
" 'matching_text': '($ in millions, except per share data)'},\n",
" {'page': 42, 'matching_text': '($ in millions)'},\n",
" {'page': 43, 'matching_text': '($ in millions)'},\n",
" {'page': 52, 'matching_text': '(In millions, except per share data)'},\n",
" {'page': 53,\n",
" 'matching_text': 'Consolidated Statements of Comprehensive Income (In millions)'},\n",
" {'page': 54,\n",
" 'matching_text': 'Consolidated Balance Sheets (In millions, except par value)'},\n",
" {'page': 55, 'matching_text': '(In millions, except per share data)'},\n",
" {'page': 56,\n",
" 'matching_text': 'Consolidated Statements of Cash Flows (In millions)'},\n",
" {'page': 65,\n",
" 'matching_text': 'Year Ended<br/>Jan 26, 2025<br/>(In millions, except per share data)'},\n",
" {'page': 71, 'matching_text': '(In millions) | (In millions)'},\n",
" {'page': 72, 'matching_text': '(In millions)'}]},\n",
" 'revenue': {'reasoning': 'The total revenue for fiscal year 2025 is extracted from multiple sources within the text, all confirming the same figure of $130,497 million. The revenue recognized for fiscal year 2025 is also noted as $4,607 million, which is a separate figure. However, the primary focus is on the total revenue figure, which is consistently cited.',\n",
" 'citation': [{'page': 38,\n",
" 'matching_text': 'Revenue for fiscal year 2025 was $130.5 billion'},\n",
" {'page': 41,\n",
" 'matching_text': 'Total | $ 130,497 | $ | 60,922'},\n",
" {'page': 52, 'matching_text': 'Revenue | $ 130,497'},\n",
" {'page': 78,\n",
" 'matching_text': 'Revenue | $ 116,193 | $ 14,304 | $ - | $ 130,497'},\n",
" {'page': 79, 'matching_text': 'Total revenue | $ 130,497'},\n",
" {'page': 80, 'matching_text': 'Total revenue | $ 130,497'}]}},\n",
" 'usage': {'num_pages_extracted': 130,\n",
" 'num_document_tokens': 105932,\n",
" 'num_output_tokens': 31306}}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"filing_info.extraction_metadata"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## What's Next?\n",
"\n",
"In this example, we built an Extraction Agent that is capable of citing it's sources from the document it's extracting data from, and reasoning about its reponse. To further customize and improve on the results, you can also try to customize the `system_prompt` in the `ExtractConfig`.\n",
"\n",
"#### Learn More\n",
"\n",
"- [LlamaExtract Documentation](https://docs.cloud.llamaindex.ai/llamaextract/getting_started)\n",
"- [Example Notebooks](https://github.com/run-llama/llama_cloud_services/tree/main/examples/extract)"
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "Python 3",
"name": "python3"
},
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 0
}
File diff suppressed because it is too large Load Diff
+1 -11
View File
@@ -1,17 +1,7 @@
from llama_cloud_services.extract.extract import (
LlamaExtract,
ExtractConfig,
ExtractionAgent,
SourceText,
ExtractTarget,
ExtractMode,
)
__all__ = [
"LlamaExtract",
"ExtractionAgent",
"SourceText",
"ExtractConfig",
"ExtractTarget",
"ExtractMode",
]
__all__ = ["LlamaExtract", "ExtractionAgent", "SourceText"]
+58 -101
View File
@@ -8,32 +8,25 @@ import secrets
import warnings
import httpx
from pydantic import BaseModel
from tenacity import (
retry_if_exception,
stop_after_attempt,
wait_exponential_jitter,
AsyncRetrying,
)
from llama_cloud import (
ExtractAgent as CloudExtractAgent,
ExtractAgentCreate,
ExtractConfig,
ExtractJob,
ExtractJobCreate,
ExtractRun,
ExtractSchemaValidateRequest,
ExtractAgentUpdate,
File,
ExtractMode,
StatusEnum,
Project,
ExtractTarget,
LlamaExtractSettings,
PaginatedExtractRunsResponse,
)
from llama_cloud.client import AsyncLlamaCloud
from llama_cloud.core.api_error import ApiError
from llama_cloud_services.extract.utils import (
JSONObjectType,
augment_async_errors,
ExperimentalWarning,
)
from llama_cloud_services.extract.utils import JSONObjectType, augment_async_errors
from llama_index.core.schema import BaseComponent
from llama_index.core.async_utils import run_jobs
from llama_index.core.bridge.pydantic import Field, PrivateAttr
@@ -51,17 +44,6 @@ DEFAULT_EXTRACT_CONFIG = ExtractConfig(
)
def _is_retryable_error(exception: BaseException) -> bool:
"""Check if an exception is retryable."""
if isinstance(exception, ApiError):
return exception.status_code in (502, 503, 504, 425, 408)
elif isinstance(
exception, (httpx.HTTPStatusError, httpx.RequestError, httpx.TimeoutException)
):
return True
return False
class SourceText:
def __init__(
self,
@@ -136,25 +118,6 @@ def run_in_thread(
return thread_pool.submit(run_coro).result()
def _extraction_config_warning(config: ExtractConfig) -> None:
if config.extraction_mode == ExtractMode.ACCURATE:
warnings.warn("ACCURATE extraction mode is deprecated. Using BALANCED instead.")
config.extraction_mode = ExtractMode.BALANCED
if config.use_reasoning:
warnings.warn(
"`use_reasoning` is an experimental feature. Results will be available in "
"the `extraction_metadata` field for the extraction run.",
ExperimentalWarning,
)
if config.cite_sources:
warnings.warn(
"`cite_sources` is an experimental feature. This may greatly increase the "
"size of the response, and slow down the extraction. Results will be "
"available in the `extraction_metadata` field for the extraction run.",
ExperimentalWarning,
)
class ExtractionAgent:
"""Class representing a single extraction agent with methods for extraction operations."""
@@ -215,7 +178,7 @@ class ExtractionAgent:
)
validated_schema = self._run_in_thread(
self._client.llama_extract.validate_extraction_schema(
data_schema=processed_schema
request=ExtractSchemaValidateRequest(data_schema=processed_schema)
)
)
self._data_schema = validated_schema.data_schema
@@ -226,7 +189,6 @@ class ExtractionAgent:
@config.setter
def config(self, config: ExtractConfig) -> None:
_extraction_config_warning(config)
self._config = config
def _run_in_thread(self, coro: Coroutine[Any, Any, T]) -> T:
@@ -249,8 +211,9 @@ class ExtractionAgent:
ValueError: If filename is not provided for bytes input or for file-like objects
without a name attribute.
"""
file_contents: Optional[Union[BufferedIOBase, BytesIO]] = None
try:
file_contents: Union[BufferedIOBase, BytesIO]
if file_input.text_content is not None:
# Handle direct text content
file_contents = BytesIO(file_input.text_content.encode("utf-8"))
@@ -277,7 +240,7 @@ class ExtractionAgent:
project_id=self._project_id, upload_file=file_contents
)
finally:
if file_contents is not None and isinstance(file_contents, BufferedReader):
if isinstance(file_contents, BufferedReader):
file_contents.close()
async def _upload_file(self, file_input: FileInput) -> File:
@@ -305,60 +268,35 @@ class ExtractionAgent:
return await self.upload_file(source_text)
async def _get_job_with_retry(self, job_id: str) -> ExtractJob:
"""Get job with retry logic for transient errors."""
async for attempt in AsyncRetrying(
retry=retry_if_exception(_is_retryable_error),
stop=stop_after_attempt(5),
wait=wait_exponential_jitter(initial=1, max=60, jitter=5),
reraise=True,
):
with attempt:
return await self._client.llama_extract.get_job(job_id=job_id)
async def _get_run_with_retry(self, job_id: str) -> ExtractRun:
"""Get extraction run with retry logic for transient errors."""
async for attempt in AsyncRetrying(
retry=retry_if_exception(_is_retryable_error),
stop=stop_after_attempt(3),
wait=wait_exponential_jitter(initial=1, max=20, jitter=3),
reraise=True,
):
with attempt:
return await self._client.llama_extract.get_run_by_job_id(job_id=job_id)
async def _wait_for_job_result(self, job_id: str) -> Optional[ExtractRun]:
"""Wait for and return the results of an extraction job."""
start = time.perf_counter()
tries = 0
while True:
await asyncio.sleep(self.check_interval)
tries += 1
job = await self._client.llama_extract.get_job(
job_id=job_id,
)
try:
job = await self._get_job_with_retry(job_id)
if job.status == StatusEnum.SUCCESS:
return await self._get_run_with_retry(job_id)
elif job.status == StatusEnum.PENDING:
end = time.perf_counter()
if end - start > self.max_timeout:
raise Exception(f"Timeout while extracting the file: {job_id}")
if self._verbose and tries % 10 == 0:
print(".", end="", flush=True)
continue
else:
warnings.warn(
f"Failure in job: {job_id}, status: {job.status}, error: {job.error}"
)
return await self._get_run_with_retry(job_id)
except Exception as e:
# If we get a non-retryable error or all retries are exhausted, re-raise
if self._verbose:
print(f"\nError in job polling for {job_id}: {e}")
raise e
if job.status == StatusEnum.SUCCESS:
return await self._client.llama_extract.get_run_by_job_id(
job_id=job_id,
)
elif job.status == StatusEnum.PENDING:
end = time.perf_counter()
if end - start > self.max_timeout:
raise Exception(f"Timeout while extracting the file: {job_id}")
if self._verbose and tries % 10 == 0:
print(".", end="", flush=True)
continue
else:
warnings.warn(
f"Failure in job: {job_id}, status: {job.status}, error: {job.error}"
)
return await self._client.llama_extract.get_run_by_job_id(
job_id=job_id,
)
def save(self) -> None:
"""Persist the extraction agent's schema and config to the database.
@@ -369,8 +307,10 @@ class ExtractionAgent:
self._agent = self._run_in_thread(
self._client.llama_extract.update_extraction_agent(
extraction_agent_id=self.id,
data_schema=self.data_schema,
config=self.config,
request=ExtractAgentUpdate(
data_schema=self.data_schema,
config=self.config,
),
)
)
@@ -662,7 +602,7 @@ class LlamaExtract(BaseComponent):
httpx_timeout=httpx_timeout,
verbose=verbose,
)
self._httpx_client = httpx.AsyncClient(verify=verify, timeout=httpx_timeout) # type: ignore
self._httpx_client = httpx.AsyncClient(verify=verify, timeout=httpx_timeout)
self.verify = verify
self.httpx_timeout = httpx_timeout
@@ -674,8 +614,21 @@ class LlamaExtract(BaseComponent):
self._thread_pool = ThreadPoolExecutor(
max_workers=min(10, (os.cpu_count() or 1) + 4)
)
# Fetch default project id if not provided
if not project_id:
project_id = os.getenv("LLAMA_CLOUD_PROJECT_ID", None)
if not project_id:
print("No project_id provided, fetching default project.")
projects: List[Project] = self._run_in_thread(
self._async_client.projects.list_projects()
)
default_project = [p for p in projects if p.is_default]
if not default_project:
raise ValueError(
"No default project found. Please provide a project_id."
)
project_id = default_project[0].id
self._project_id = project_id
self._organization_id = organization_id
@@ -706,7 +659,11 @@ class LlamaExtract(BaseComponent):
ExtractionAgent: The created extraction agent
"""
if config is not None:
_extraction_config_warning(config)
if config.extraction_mode == ExtractMode.ACCURATE:
warnings.warn(
"ACCURATE extraction mode is deprecated. Using BALANCED instead."
)
config.extraction_mode = ExtractMode.BALANCED
else:
config = DEFAULT_EXTRACT_CONFIG
@@ -723,9 +680,11 @@ class LlamaExtract(BaseComponent):
self._async_client.llama_extract.create_extraction_agent(
project_id=self._project_id,
organization_id=self._organization_id,
name=name,
data_schema=data_schema,
config=config,
request=ExtractAgentCreate(
name=name,
data_schema=data_schema,
config=config,
),
)
)
@@ -739,8 +698,6 @@ class LlamaExtract(BaseComponent):
num_workers=self.num_workers,
show_progress=self.show_progress,
verbose=self.verbose,
verify=self.verify,
httpx_timeout=self.httpx_timeout,
)
def get_agent(
-6
View File
@@ -32,9 +32,3 @@ def augment_async_errors() -> Generator[None, None, None]:
JSONType = Union[Dict[str, Any], List[Any], str, int, float, bool, None]
JSONObjectType = Dict[str, JSONType]
class ExperimentalWarning(Warning):
"""Warning for experimental features."""
pass
+89 -308
View File
@@ -2,41 +2,32 @@ import asyncio
import mimetypes
import os
import time
import warnings
from contextlib import asynccontextmanager
from copy import deepcopy
from enum import Enum
from io import BufferedIOBase
from pathlib import Path, PurePath, PurePosixPath
from typing import Any, AsyncGenerator, Dict, List, Optional, Tuple, Union
from typing import Any, AsyncGenerator, Dict, List, Optional, Union
from urllib.parse import urlparse
import httpx
from fsspec import AbstractFileSystem
from llama_index.core.async_utils import asyncio_run, run_jobs
from llama_index.core.bridge.pydantic import (
Field,
PrivateAttr,
field_validator,
model_validator,
)
from llama_index.core.bridge.pydantic import Field, PrivateAttr, field_validator
from llama_index.core.constants import DEFAULT_BASE_URL
from llama_index.core.readers.base import BasePydanticReader
from llama_index.core.readers.file.base import get_default_fs
from llama_index.core.schema import Document
from llama_cloud_services.utils import check_extra_params
from llama_cloud_services.parse.types import JobResult
from llama_cloud_services.parse.utils import (
SUPPORTED_FILE_TYPES,
ResultType,
ParsingMode,
FailedPageMode,
expand_target_pages,
nest_asyncio_err,
nest_asyncio_msg,
make_api_request,
partition_pages,
)
# can put in a path to the file or the file bytes itself
@@ -66,36 +57,6 @@ def build_url(
return base_url
class JobFailedException(Exception):
"""Parse job failed exception."""
def __init__(
self,
job_id: str,
status: str,
error_code: Optional[str] = None,
error_message: Optional[str] = None,
):
exception_str = (
f"Job ID: {job_id} failed with status: {status}, "
f'Error code: {error_code or "No error code found"}, '
f'Error message: {error_message or "No error message found"}'
)
super().__init__(exception_str)
self.job_id = job_id
self.status = status
self.error_code = error_code
self.error_message = error_message
@classmethod
def from_result(cls, result_json: Dict[str, Any]) -> "JobFailedException":
job_id = result_json["id"]
status = result_json["status"]
error_code = result_json.get("error_code")
error_message = result_json.get("error_message")
return cls(job_id, status, error_code=error_code, error_message=error_message)
class BackoffPattern(str, Enum):
"""Backoff pattern for polling."""
@@ -154,7 +115,7 @@ class LlamaParse(BasePydanticReader):
num_workers: int = Field(
default=4,
gt=0,
lt=20,
lt=10,
description="The number of workers to use sending API requests for parsing.",
)
result_type: ResultType = Field(
@@ -184,10 +145,6 @@ class LlamaParse(BasePydanticReader):
default=False,
description="If set to true, the parser will automatically select the best mode to extract text from documents based on the rules provide. Will use the 'accurate' default mode by default and will upgrade page that match the rule to Premium mode.",
)
auto_mode_configuration_json: Optional[str] = Field(
default=None,
description="A JSON string containing the configuration for the auto mode. If set, the parser will use the provided configuration for the auto mode.",
)
auto_mode_trigger_on_image_in_page: Optional[bool] = Field(
default=False,
description="If auto_mode is set to true, the parser will upgrade the page that contain an image to Premium mode.",
@@ -273,10 +230,6 @@ class LlamaParse(BasePydanticReader):
default=False,
description="Whether to guess the sheet names of the xlsx file.",
)
high_res_ocr: Optional[bool] = Field(
default=False,
description="If set to true, the parser will use high resolution OCR to extract text from images. This will increase the accuracy of the parsing job, but reduce the speed.",
)
html_make_all_elements_visible: Optional[bool] = Field(
default=False,
description="If set to true, when parsing HTML the parser will consider all elements display not element as display block.",
@@ -340,10 +293,6 @@ class LlamaParse(BasePydanticReader):
default=False,
description="If set to true, the parser will output tables as HTML in the markdown.",
)
outlined_table_extraction: Optional[bool] = Field(
default=False,
description="If set to true, the parser will use a dedicated approach to extract tables with outlined cells. This is useful for documents with spreadsheet-like tables where cells are outlined with borders. This could lead to false positives, so use with caution.",
)
page_error_tolerance: Optional[float] = Field(
default=None,
description="The error tolerance for the number of pages with error in a doc (percentage express as 0-1). If we fail to parse a greater percentage of pages than the tolerance value we fail the job.",
@@ -368,15 +317,11 @@ class LlamaParse(BasePydanticReader):
default=False,
description="Use our best parser mode if set to True.",
)
preset: Optional[str] = Field(
default=None,
description="The preset to use for the parser. If set, the parser will use the preset configuration. See LlamaParse documentation for available presets. Preset override most other parameters.",
)
preserve_layout_alignment_across_pages: Optional[bool] = Field(
default=False,
description="Preserve grid alignment across page in text mode.",
)
replace_failed_page_mode: Optional[FailedPageMode] = Field(
replace_failed_page_mode: Optional[Union[FailedPageMode, str]] = Field(
default=None,
description="The mode to use to replace the failed page, see FailedPageMode enum for possible value. If set, the parser will replace the failed page with the specified mode. If not set, the default mode (raw_text) will be used.",
)
@@ -457,10 +402,6 @@ class LlamaParse(BasePydanticReader):
default=None,
description="The model name for the vendor multimodal API.",
)
model: Optional[str] = Field(
default=None,
description="The document model name to be used with `parse_with_agent`.",
)
webhook_url: Optional[str] = Field(
default=None,
description="A URL that needs to be called at the end of the parsing job.",
@@ -504,26 +445,6 @@ class LlamaParse(BasePydanticReader):
description="Whether to use the vendor multimodal API.",
)
partition_pages: Optional[int] = Field(
default=None,
description="If set, documents will automatically be partitioned into segments containing the specified number of pages at most. Parsing will be split into separate jobs for each partition segment. Can be used in combination with targetPages and maxPages.",
)
@model_validator(mode="before")
@classmethod
def warn_extra_params(cls, data: Dict[str, Any]) -> Dict[str, Any]:
extra_params, suggestions = check_extra_params(cls, data)
if extra_params:
suggestions = [f"\n - {suggestion}" for suggestion in suggestions]
suggestions_str = "".join(suggestions)
warnings.warn(
"The following parameters are unused: "
+ ", ".join(extra_params)
+ f".\n{suggestions_str}",
)
return data
@field_validator("api_key", mode="before", check_fields=True)
@classmethod
def validate_api_key(cls, v: str) -> str:
@@ -611,7 +532,6 @@ class LlamaParse(BasePydanticReader):
file_input: FileInput,
extra_info: Optional[dict] = None,
fs: Optional[AbstractFileSystem] = None,
partition_target_pages: Optional[str] = None,
) -> str:
files = None
file_handle = None
@@ -662,9 +582,6 @@ class LlamaParse(BasePydanticReader):
if self.auto_mode:
data["auto_mode"] = self.auto_mode
if self.auto_mode_configuration_json is not None:
data["auto_mode_configuration_json"] = self.auto_mode_configuration_json
if self.auto_mode_trigger_on_image_in_page:
data[
"auto_mode_trigger_on_image_in_page"
@@ -762,9 +679,6 @@ class LlamaParse(BasePydanticReader):
if self.html_make_all_elements_visible:
data["html_make_all_elements_visible"] = self.html_make_all_elements_visible
if self.high_res_ocr:
data["high_res_ocr"] = self.high_res_ocr
if self.html_remove_fixed_elements:
data["html_remove_fixed_elements"] = self.html_remove_fixed_elements
@@ -827,9 +741,6 @@ class LlamaParse(BasePydanticReader):
if self.output_tables_as_HTML:
data["output_tables_as_HTML"] = self.output_tables_as_HTML
if self.outlined_table_extraction:
data["outlined_table_extraction"] = self.outlined_table_extraction
if self.page_error_tolerance is not None:
data["page_error_tolerance"] = self.page_error_tolerance
@@ -861,12 +772,8 @@ class LlamaParse(BasePydanticReader):
"preserve_layout_alignment_across_pages"
] = self.preserve_layout_alignment_across_pages
if self.preset is not None:
data["preset"] = self.preset
if self.replace_failed_page_mode is not None:
data["replace_failed_page_mode"] = self.replace_failed_page_mode.value
data["replace_failed_page_mode"] = self.replace_failed_page_mode
if self.replace_failed_page_with_error_message_prefix is not None:
data[
"replace_failed_page_with_error_message_prefix"
@@ -912,9 +819,7 @@ class LlamaParse(BasePydanticReader):
if self.take_screenshot:
data["take_screenshot"] = self.take_screenshot
if partition_target_pages is not None:
data["target_pages"] = partition_target_pages
elif self.target_pages is not None:
if self.target_pages is not None:
data["target_pages"] = self.target_pages
if self.user_prompt is not None:
data["user_prompt"] = self.user_prompt
@@ -927,9 +832,6 @@ class LlamaParse(BasePydanticReader):
if self.vendor_multimodal_model_name is not None:
data["vendor_multimodal_model_name"] = self.vendor_multimodal_model_name
if self.model is not None:
data["model"] = self.model
if self.webhook_url is not None:
data["webhook_url"] = self.webhook_url
@@ -1011,7 +913,15 @@ class LlamaParse(BasePydanticReader):
print(".", end="", flush=True)
current_interval = self._calculate_backoff(current_interval)
else:
raise JobFailedException.from_result(result_json)
error_code = result_json.get("error_code", "No error code found")
error_message = result_json.get(
"error_message", "No error message found"
)
exception_str = (
f"Job ID: {job_id} failed with status: {status}, "
f"Error code: {error_code}, Error message: {error_message}"
)
raise Exception(exception_str)
except (
httpx.ConnectError,
httpx.ReadError,
@@ -1020,7 +930,6 @@ class LlamaParse(BasePydanticReader):
httpx.ReadTimeout,
httpx.WriteTimeout,
httpx.HTTPStatusError,
httpx.RemoteProtocolError,
) as err:
error_count += 1
end = time.time()
@@ -1035,151 +944,26 @@ class LlamaParse(BasePydanticReader):
)
current_interval = self._calculate_backoff(current_interval)
async def _parse_one(
self,
file_path: FileInput,
extra_info: Optional[dict] = None,
fs: Optional[AbstractFileSystem] = None,
result_type: Optional[str] = None,
num_workers: Optional[int] = None,
) -> List[Tuple[str, Dict[str, Any]]]:
if self.partition_pages is None:
job_results = [
await self._parse_one_unpartitioned(
file_path,
extra_info=extra_info,
fs=fs,
result_type=result_type,
)
]
else:
job_results = await self._parse_one_partitioned(
file_path,
extra_info,
fs=fs,
result_type=result_type,
num_workers=num_workers,
)
return job_results
async def _parse_one_unpartitioned(
self,
file_path: FileInput,
extra_info: Optional[dict] = None,
fs: Optional[AbstractFileSystem] = None,
result_type: Optional[str] = None,
**create_kwargs: Any,
) -> Tuple[str, Dict[str, Any]]:
"""Create one parse job and wait for the result."""
job_id = await self._create_job(
file_path, extra_info=extra_info, fs=fs, **create_kwargs
)
if self.verbose:
print("Started parsing the file under job_id %s" % job_id)
result = await self._get_job_result(
job_id, result_type or self.result_type.value, verbose=self.verbose
)
return job_id, result
async def _parse_one_partitioned(
self,
file_path: FileInput,
extra_info: Optional[dict] = None,
fs: Optional[AbstractFileSystem] = None,
result_type: Optional[str] = None,
num_workers: Optional[int] = None,
) -> List[Tuple[str, Dict[str, Any]]]:
"""Partition a file and run separate parse jobs per partition segment."""
assert self.partition_pages is not None
num_workers = num_workers or self.num_workers
if num_workers < 1:
raise ValueError("Invalid number of workers")
if self.target_pages is not None:
jobs = [
self._parse_one_unpartitioned(
file_path,
extra_info=extra_info,
fs=fs,
result_type=result_type,
partition_target_pages=target_pages,
)
for target_pages in partition_pages(
expand_target_pages(self.target_pages),
self.partition_pages,
max_pages=self.max_pages,
)
]
return await run_jobs(
jobs,
workers=num_workers,
desc="Getting job results",
show_progress=self.show_progress,
)
total = 0
results: List[Tuple[str, Dict[str, Any]]] = []
while self.max_pages is None or total < self.max_pages:
if (
self.max_pages is not None
and total + self.partition_pages >= self.max_pages
):
size = self.max_pages - total
else:
size = self.partition_pages
if not size:
break
try:
# Fetch JSON result type first to get accurate pagination data
# and then fetch the user's desired result type if needed
job_id, json_result = await self._parse_one_unpartitioned(
file_path,
extra_info=extra_info,
fs=fs,
result_type=ResultType.JSON.value,
partition_target_pages=f"{total}-{total + size - 1}",
)
result_type = result_type or self.result_type.value
if result_type == ResultType.JSON.value:
job_result = json_result
else:
job_result = await self._get_job_result(
job_id, result_type, verbose=self.verbose
)
except JobFailedException as e:
if results and e.error_code == "NO_DATA_FOUND_IN_FILE":
# Expected when we try to read past the end of the file
return results
raise
results.append((job_id, job_result))
if len(json_result["pages"]) < size:
break
total += size
return results
async def _aload_data(
self,
file_path: FileInput,
extra_info: Optional[dict] = None,
fs: Optional[AbstractFileSystem] = None,
verbose: bool = False,
num_workers: Optional[int] = None,
) -> List[Document]:
"""Load data from the input path."""
try:
results = [
job_result
for _, job_result in await self._parse_one(
file_path, extra_info, fs=fs, num_workers=num_workers
)
]
# Flatten the resulting doc if it was partitioned
separator = self.page_separator or _DEFAULT_SEPARATOR
job_id = await self._create_job(file_path, extra_info=extra_info, fs=fs)
if verbose:
print("Started parsing the file under job_id %s" % job_id)
result = await self._get_job_result(
job_id, self.result_type.value, verbose=verbose
)
docs = [
Document(
text=separator.join(
result[self.result_type.value] for result in results
),
text=result[self.result_type.value],
metadata=extra_info or {},
)
]
@@ -1202,11 +986,7 @@ class LlamaParse(BasePydanticReader):
extra_info: Optional[dict] = None,
fs: Optional[AbstractFileSystem] = None,
) -> List[Document]:
"""Load data from the input path.
File(s) which were partitioned before parsing will be loaded as a single
re-assembled Document.
"""
"""Load data from the input path."""
if isinstance(file_path, (str, PurePosixPath, Path, bytes, BufferedIOBase)):
return await self._aload_data(
file_path, extra_info=extra_info, fs=fs, verbose=self.verbose
@@ -1218,7 +998,6 @@ class LlamaParse(BasePydanticReader):
extra_info=extra_info,
fs=fs,
verbose=self.verbose and not self.show_progress,
num_workers=1,
)
for f in file_path
]
@@ -1257,34 +1036,6 @@ class LlamaParse(BasePydanticReader):
else:
raise e
async def _aparse_one(
self,
file_path: FileInput,
file_name: str,
extra_info: Optional[dict] = None,
fs: Optional[AbstractFileSystem] = None,
num_workers: Optional[int] = None,
) -> List[JobResult]:
job_results = await self._parse_one(
file_path,
extra_info,
fs=fs,
result_type=ResultType.JSON.value,
num_workers=num_workers,
)
return [
JobResult(
job_id=job_id,
file_name=file_name,
job_result=job_result,
api_key=self.api_key,
base_url=self.base_url,
client=self.aclient,
page_separator=self.page_separator or _DEFAULT_SEPARATOR,
)
for job_id, job_result in job_results
]
async def aparse(
self,
file_path: Union[List[FileInput], FileInput],
@@ -1303,10 +1054,14 @@ class LlamaParse(BasePydanticReader):
fs: Optional filesystem to use for reading files.
Returns:
JobResult object or list of JobResult objects if either multiple files were provided or file(s) were partitioned before parsing.
JobResult object or list of JobResult objects if multiple files were provided
"""
if isinstance(file_path, (str, PurePosixPath, Path, bytes, BufferedIOBase)):
job_id = await self._create_job(file_path, extra_info=extra_info, fs=fs)
if self.verbose:
print("Started parsing the file under job_id %s" % job_id)
if isinstance(file_path, (bytes, BufferedIOBase)):
if not extra_info or "file_name" not in extra_info:
raise ValueError(
@@ -1315,12 +1070,29 @@ class LlamaParse(BasePydanticReader):
file_name = extra_info["file_name"]
else:
file_name = str(file_path)
result = await self._aparse_one(
file_path, file_name, extra_info=extra_info, fs=fs
job_result = await self._get_job_result(
job_id, ResultType.JSON.value, verbose=self.verbose
)
return JobResult(
job_id=job_id,
file_name=file_name,
job_result=job_result,
api_key=self.api_key,
base_url=self.base_url,
client=self.aclient,
page_separator=self.page_separator or _DEFAULT_SEPARATOR,
)
return result[0] if len(result) == 1 else result
elif isinstance(file_path, list):
jobs = [
self._create_job(
f,
extra_info=extra_info,
fs=fs,
)
for f in file_path
]
file_names = []
for f in file_path:
if isinstance(f, (bytes, BufferedIOBase)):
@@ -1332,24 +1104,40 @@ class LlamaParse(BasePydanticReader):
else:
file_names.append(str(f))
job_results = []
try:
for result in await run_jobs(
job_ids = await run_jobs(
jobs,
workers=self.num_workers,
desc="Creating parsing jobs",
show_progress=self.show_progress,
)
job_results = await run_jobs(
[
self._aparse_one(
f,
file_names[i],
extra_info=extra_info,
fs=fs,
num_workers=1,
self._get_job_result(
job_id, ResultType.JSON.value, verbose=self.verbose
)
for i, f in enumerate(file_path)
for job_id in job_ids
],
workers=self.num_workers,
desc="Getting job results",
show_progress=self.show_progress,
):
job_results.extend(result)
)
# Create JobResults just using the job_ids and job_results
job_results = [
JobResult(
job_id=job_id,
file_name=file_names[i],
job_result=job_results[i],
api_key=self.api_key,
base_url=self.base_url,
client=self.aclient,
page_separator=self.page_separator or _DEFAULT_SEPARATOR,
)
for i, job_id in enumerate(job_ids)
]
return job_results
except RuntimeError as e:
@@ -1391,27 +1179,20 @@ class LlamaParse(BasePydanticReader):
raise e
async def _aget_json(
self,
file_path: FileInput,
extra_info: Optional[dict] = None,
num_workers: Optional[int] = None,
self, file_path: FileInput, extra_info: Optional[dict] = None
) -> List[dict]:
"""Load data from the input path."""
try:
job_results = await self._parse_one(
file_path,
extra_info=extra_info,
result_type=ResultType.JSON.value,
num_workers=num_workers,
)
job_id = await self._create_job(file_path, extra_info=extra_info)
if self.verbose:
print("Started parsing the file under job_id %s" % job_id)
result = await self._get_job_result(job_id, "json")
result["job_id"] = job_id
results = []
for job_id, job_result in job_results:
job_result["job_id"] = job_id
if not isinstance(file_path, (bytes, BufferedIOBase)):
job_result["file_path"] = str(file_path)
results.append(job_result)
return results
if not isinstance(file_path, (bytes, BufferedIOBase)):
result["file_path"] = str(file_path)
return [result]
except Exception as e:
file_repr = file_path if isinstance(file_path, str) else "<bytes/buffer>"
print(f"Error while parsing the file '{file_repr}':", e)
@@ -1426,7 +1207,7 @@ class LlamaParse(BasePydanticReader):
extra_info: Optional[dict] = None,
) -> List[dict]:
"""Load data from the input path."""
if isinstance(file_path, (str, PurePosixPath, Path, bytes, BufferedIOBase)):
if isinstance(file_path, (str, Path)):
return await self._aget_json(file_path, extra_info=extra_info)
elif isinstance(file_path, list):
jobs = [self._aget_json(f, extra_info=extra_info) for f in file_path]
@@ -1447,7 +1228,7 @@ class LlamaParse(BasePydanticReader):
raise e
else:
raise ValueError(
"The input file_path must be a string, Path, bytes, BufferedIOBase, or a list of these types."
"The input file_path must be a string or a list of strings."
)
def get_json_result(
+27 -53
View File
@@ -14,6 +14,9 @@ PAGE_REGEX = r"page[-_](\d+)\.jpg$"
class JobMetadata(BaseModel):
"""Metadata about the job."""
job_credits_usage: int = Field(
default_factory=dict, description="The credits usage for the job."
)
job_pages: int = Field(description="The number of pages in the job.")
job_auto_mode_triggered_pages: int = Field(
description="The number of pages that triggered auto mode (thus increasing the cost)."
@@ -46,31 +49,19 @@ class PageItem(BaseModel):
rows: Optional[List[List[str]]] = Field(
default=None, description="The rows of the item."
)
bBox: Optional[BBox] = Field(
default=None, description="The bounding box of the item."
)
bBox: BBox = Field(description="The bounding box of the item.")
class ImageItem(BaseModel):
"""An image in a page."""
name: str = Field(description="The name of the image.")
height: Optional[float] = Field(
default=None, description="The height of the image."
)
width: Optional[float] = Field(default=None, description="The width of the image.")
x: Optional[float] = Field(
default=None, description="The x-coordinate of the image."
)
y: Optional[float] = Field(
default=None, description="The y-coordinate of the image."
)
original_width: Optional[int] = Field(
default=None, description="The original width of the image."
)
original_height: Optional[int] = Field(
default=None, description="The original height of the image."
)
height: float = Field(description="The height of the image.")
width: float = Field(description="The width of the image.")
x: float = Field(description="The x-coordinate of the image.")
y: float = Field(description="The y-coordinate of the image.")
original_width: int = Field(description="The original width of the image.")
original_height: int = Field(description="The original height of the image.")
type: Optional[str] = Field(default=None, description="The type of the image.")
@@ -80,9 +71,7 @@ class LayoutItem(BaseModel):
image: str = Field(description="The name of the image containing the layout item")
confidence: float = Field(description="The confidence of the layout item.")
label: str = Field(description="The label of the layout item.")
bbox: Optional[BBox] = Field(
default=None, description="The bounding box of the layout item."
)
bbox: BBox = Field(description="The bounding box of the layout item.")
isLikelyNoise: bool = Field(description="Whether the layout item is likely noise.")
@@ -90,24 +79,18 @@ class ChartItem(BaseModel):
"""A chart in a page."""
name: str = Field(description="The name of the chart.")
x: Optional[float] = Field(
default=None, description="The x-coordinate of the chart."
)
y: Optional[float] = Field(
default=None, description="The y-coordinate of the chart."
)
width: Optional[float] = Field(default=None, description="The width of the chart.")
height: Optional[float] = Field(
default=None, description="The height of the chart."
)
x: float = Field(description="The x-coordinate of the chart.")
y: float = Field(description="The y-coordinate of the chart.")
width: float = Field(description="The width of the chart.")
height: float = Field(description="The height of the chart.")
class Page(BaseModel):
"""A page of the document."""
page: int = Field(description="The page number.")
text: Optional[str] = Field(default=None, description="The text of the page.")
md: Optional[str] = Field(default=None, description="The markdown of the page.")
text: str = Field(description="The text of the page.")
md: str = Field(description="The markdown of the page.")
images: List[ImageItem] = Field(
default_factory=list,
description="The names of the image IDs in the page, including both objects and page screenshots.",
@@ -124,28 +107,23 @@ class Page(BaseModel):
items: List[PageItem] = Field(
default_factory=list, description="The items in the page."
)
status: Optional[str] = Field(default=None, description="The status of the page.")
status: str = Field(description="The status of the page.")
links: List[SerializeAsAny[Any]] = Field(
default_factory=list, description="The links in the page."
)
width: Optional[float] = Field(default=None, description="The width of the page.")
height: Optional[float] = Field(default=None, description="The height of the page.")
width: float = Field(description="The width of the page.")
height: float = Field(description="The height of the page.")
triggeredAutoMode: bool = Field(
default=False,
description="Whether the page triggered auto mode (thus increasing the cost).",
)
parsingMode: str = Field(
default="", description="The parsing mode used for the page."
description="Whether the page triggered auto mode (thus increasing the cost)."
)
parsingMode: str = Field(description="The parsing mode used for the page.")
structuredData: Optional[Dict[str, Any]] = Field(
default=None, description="The structured data of the page."
description="The structured data of the page."
)
noStructuredContent: bool = Field(
default=True, description="Whether the page has no structured data."
)
noTextContent: bool = Field(
default=False, description="Whether the page has no text content."
description="Whether the page has no structured data."
)
noTextContent: bool = Field(description="Whether the page has no text content.")
class JobResult(BaseModel):
@@ -207,9 +185,7 @@ class JobResult(BaseModel):
for page in self.pages
]
else:
text = self._page_separator.join(
[page.text if page.text is not None else "" for page in self.pages]
)
text = self._page_separator.join([page.text for page in self.pages])
return [Document(text=text, metadata={"file_name": self.file_name})]
async def aget_text_documents(self, split_by_page: bool = False) -> List[Document]:
@@ -254,9 +230,7 @@ class JobResult(BaseModel):
else:
return [
Document(
text=self._page_separator.join(
[page.md if page.md is not None else "" for page in self.pages]
),
text=self._page_separator.join([page.md for page in self.pages]),
metadata={"file_name": self.file_name},
)
]
+1 -55
View File
@@ -1,5 +1,4 @@
import httpx
import itertools
import logging
from enum import Enum
from tenacity import (
@@ -9,7 +8,7 @@ from tenacity import (
retry_if_exception,
before_sleep_log,
)
from typing import Any, Iterable, Iterator, Optional
from typing import Any
logger = logging.getLogger(__name__)
@@ -298,56 +297,3 @@ async def make_api_request(
return response
return await _make_request(url, **httpx_kwargs)
def expand_target_pages(target_pages: str) -> Iterator[int]:
"""Yield all values in target_pages."""
for target in target_pages.strip().split(","):
if "-" in target:
try:
start, end = map(int, target.strip().split("-"))
if start > end:
raise ValueError
yield from range(start, end + 1)
except ValueError as e:
raise ValueError(f"Invalid page range: {target}") from e
else:
try:
yield int(target)
except ValueError as e:
raise ValueError(f"Invalid page number: {target}") from e
def partition_pages(
pages: Iterable[int], size: int, max_pages: Optional[int] = None
) -> Iterator[str]:
"""Yield partitioned target_pages segments."""
if size < 1:
raise ValueError(f"Invalid partition segment size: {size}")
if max_pages is not None and max_pages < 1:
raise ValueError("Max pages must be > 0")
it = iter(pages)
total = 0
while max_pages is None or total < max_pages:
segment = tuple(itertools.islice(it, size))
if segment:
targets = []
for _k, g in itertools.groupby(enumerate(segment), lambda x: x[0] - x[1]):
group = [item[1] for item in g]
if len(group) > 1:
start, end = group[0], group[-1]
group_size = end - start + 1
if max_pages is not None and total + group_size > max_pages:
end -= total + group_size - max_pages
group_size = end - start + 1
if group_size > 1:
targets.append(f"{start}-{end}")
else:
targets.append(str(start))
total += group_size
else:
targets.append(str(group[0]))
total += 1
yield ",".join(targets)
else:
return
-29
View File
@@ -1,29 +0,0 @@
import difflib
from pydantic import BaseModel
from typing import Any, Dict, List, Tuple, Type
def check_extra_params(
model_cls: Type[BaseModel], data: Dict[str, Any]
) -> Tuple[List[str], List[str]]:
# check if one of the parameters is unused, and warn the user
model_attributes = set(model_cls.model_fields.keys())
extra_params = [param for param in data.keys() if param not in model_attributes]
suggestions: List[str] = []
if extra_params:
# for each unused parameter, check if it is similar to a valid parameter and suggest a typo correction, else suggest to check the documentation / update the package
for param in extra_params:
similar_params = difflib.get_close_matches(
param, model_attributes, n=1, cutoff=0.8
)
if similar_params:
suggestions.append(
f"'{param}' is not a valid parameter. Did you mean '{similar_params[0]}' instead of '{param}'?"
)
else:
suggestions.append(
f"'{param}' is not a valid parameter. Please check the documentation or update the package."
)
return extra_params, suggestions
-3088
View File
File diff suppressed because it is too large Load Diff
+2 -2
View File
@@ -4,7 +4,7 @@ build-backend = "poetry.core.masonry.api"
[tool.poetry]
name = "llama-parse"
version = "0.6.38"
version = "0.6.14"
description = "Parse files into RAG-Optimized formats."
authors = ["Logan Markewich <logan@llamaindex.ai>"]
license = "MIT"
@@ -13,7 +13,7 @@ packages = [{include = "llama_parse"}]
[tool.poetry.dependencies]
python = ">=3.9,<4.0"
llama-cloud-services = ">=0.6.37"
llama-cloud-services = ">=0.6.14"
[tool.poetry.group.dev.dependencies]
pytest = "^8.0.0"
Generated
+652 -849
View File
File diff suppressed because it is too large Load Diff
+4 -5
View File
@@ -8,7 +8,7 @@ python_version = "3.10"
[tool.poetry]
name = "llama-cloud-services"
version = "0.6.38"
version = "0.6.14"
description = "Tailored SDK clients for LlamaCloud services."
authors = ["Logan Markewich <logan@runllama.ai>"]
license = "MIT"
@@ -17,14 +17,13 @@ packages = [{include = "llama_cloud_services"}]
[tool.poetry.dependencies]
python = ">=3.9,<4.0"
llama-index-core = ">=0.12.0"
llama-cloud = "==0.1.29"
pydantic = ">=2.8,!=2.10"
llama-index-core = ">=0.11.0"
llama-cloud = "^0.1.18"
pydantic = "!=2.10"
click = "^8.1.7"
python-dotenv = "^1.0.1"
eval-type-backport = {python = "<3.10", version = "^0.2.0"}
platformdirs = "^4.3.7"
tenacity = ">=8.5.0, <10.0"
[tool.poetry.group.dev.dependencies]
pytest = "^8.0.0"
-41
View File
@@ -1,41 +0,0 @@
import os
from typing import List
from llama_cloud_services.extract import LlamaExtract
# Global storage for agents to cleanup
_TEST_AGENTS_TO_CLEANUP: List[str] = []
def pytest_sessionfinish(session, exitstatus):
"""Hook that runs after all tests complete - cleanup agents here"""
print(
f"pytest_sessionfinish hook called! Agents to cleanup: {_TEST_AGENTS_TO_CLEANUP}"
)
if _TEST_AGENTS_TO_CLEANUP:
print("Creating cleanup client...")
# Create a fresh client just for cleanup
cleanup_client = LlamaExtract(
api_key=os.getenv("LLAMA_CLOUD_API_KEY"),
base_url=os.getenv("LLAMA_CLOUD_BASE_URL"),
project_id=os.getenv("LLAMA_CLOUD_PROJECT_ID"),
verbose=True,
)
for agent_id in _TEST_AGENTS_TO_CLEANUP:
try:
print(f"Deleting agent {agent_id}...")
cleanup_client.delete_agent(agent_id)
print(f"Cleaned up agent {agent_id}")
except Exception as e:
print(f"Warning: Failed to delete agent {agent_id}: {e}")
_TEST_AGENTS_TO_CLEANUP.clear()
print("Agent cleanup completed")
else:
print("No agents to cleanup")
def register_agent_for_cleanup(agent_id: str):
"""Register an agent ID for cleanup at the end of the test session"""
_TEST_AGENTS_TO_CLEANUP.append(agent_id)
Binary file not shown.
+8 -7
View File
@@ -5,7 +5,6 @@ from pydantic import BaseModel
from llama_cloud_services.extract import LlamaExtract, ExtractionAgent, SourceText
from tests.extract.util import load_test_dotenv
from .conftest import register_agent_for_cleanup
load_test_dotenv()
@@ -28,7 +27,7 @@ class TestSchema(BaseModel):
# Test data paths
TEST_DIR = Path(__file__).parent / "data"
TEST_PDF = TEST_DIR / "api_test" / "noisebridge_receipt.pdf"
TEST_PDF = TEST_DIR / "slide" / "saas_slide.pdf"
@pytest.fixture
@@ -59,7 +58,7 @@ def test_schema_dict():
@pytest.fixture
def test_agent(llama_extract, test_agent_name, test_schema_dict, request):
"""Creates a test agent and collects it for cleanup at the end of all tests"""
"""Creates a test agent and cleans it up after the test"""
test_id = request.node.nodeid
test_hash = hex(hash(test_id))[-8:]
base_name = test_agent_name
@@ -87,12 +86,14 @@ def test_agent(llama_extract, test_agent_name, test_schema_dict, request):
print(f"Warning: Failed to cleanup existing agent: {e}")
agent = llama_extract.create_agent(name=name, data_schema=schema)
# Add agent to cleanup list via conftest helper
register_agent_for_cleanup(agent.id)
yield agent
# Cleanup after test
try:
llama_extract.delete_agent(agent.id)
except Exception as e:
print(f"Warning: Failed to delete agent {agent.id}: {e}")
class TestLlamaExtract:
def test_init_without_api_key(self):
+9 -27
View File
@@ -1,7 +1,6 @@
import os
import pytest
import shutil
from typing import Optional, cast
from fsspec.implementations.local import LocalFileSystem
from httpx import AsyncClient
@@ -21,15 +20,11 @@ def test_simple_page_text() -> None:
assert len(result[0].text) > 0
@pytest.fixture(params=[None, 2])
def markdown_parser(request: pytest.FixtureRequest) -> LlamaParse:
@pytest.fixture
def markdown_parser() -> LlamaParse:
if os.environ.get("LLAMA_CLOUD_API_KEY", "") == "":
pytest.skip("LLAMA_CLOUD_API_KEY not set")
return LlamaParse(
result_type="markdown",
ignore_errors=False,
partition_pages=cast(Optional[int], request.param),
)
return LlamaParse(result_type="markdown", ignore_errors=False)
def test_simple_page_markdown(markdown_parser: LlamaParse) -> None:
@@ -40,6 +35,8 @@ def test_simple_page_markdown(markdown_parser: LlamaParse) -> None:
def test_simple_page_markdown_bytes(markdown_parser: LlamaParse) -> None:
markdown_parser = LlamaParse(result_type="markdown", ignore_errors=False)
filepath = "tests/test_files/attention_is_all_you_need.pdf"
with open(filepath, "rb") as f:
file_bytes = f.read()
@@ -54,6 +51,8 @@ def test_simple_page_markdown_bytes(markdown_parser: LlamaParse) -> None:
def test_simple_page_markdown_buffer(markdown_parser: LlamaParse) -> None:
markdown_parser = LlamaParse(result_type="markdown", ignore_errors=False)
filepath = "tests/test_files/attention_is_all_you_need.pdf"
with open(filepath, "rb") as f:
# client must provide extra_info with file_name
@@ -162,12 +161,9 @@ async def test_mixing_input_types() -> None:
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.parametrize("partition_pages", [None, 2])
@pytest.mark.asyncio
async def test_download_images(partition_pages: Optional[int]) -> None:
parser = LlamaParse(
result_type="markdown", take_screenshot=True, partition_pages=partition_pages
)
async def test_download_images() -> None:
parser = LlamaParse(result_type="markdown", take_screenshot=True)
filepath = "tests/test_files/attention_is_all_you_need.pdf"
json_result = await parser.aget_json([filepath])
@@ -179,17 +175,3 @@ async def test_download_images(partition_pages: Optional[int]) -> None:
await parser.aget_images(json_result, download_path)
assert len(os.listdir(download_path)) == len(json_result[0]["pages"][0]["images"])
@pytest.mark.asyncio
@pytest.mark.parametrize("split_by_page,expected", [(True, 4), (False, 1)])
async def test_multiple_page_markdown(
markdown_parser: LlamaParse,
split_by_page: bool,
expected: int,
) -> None:
markdown_parser.split_by_page = split_by_page
filepath = "tests/test_files/TOS.pdf"
result = await markdown_parser.aload_data(filepath)
assert len(result) == expected
assert all(len(doc.text) > 0 for doc in result)
+3 -53
View File
@@ -1,7 +1,6 @@
import tempfile
import os
import pytest
from typing import Optional
from llama_cloud_services import LlamaParse
from llama_cloud_services.parse.types import JobResult
@@ -16,23 +15,16 @@ def chart_file_path() -> str:
return "tests/test_files/attention_is_all_you_need_chart.pdf"
@pytest.fixture
def multiple_page_path() -> str:
return "tests/test_files/TOS.pdf"
@pytest.mark.asyncio
@pytest.mark.skipif(
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.parametrize("partition_pages", [None, 2])
async def test_basic_parse_result(file_path: str, partition_pages: Optional[int]):
async def test_basic_parse_result(file_path: str):
parser = LlamaParse(
take_screenshot=True,
auto_mode=True,
fast_mode=False,
partition_pages=partition_pages,
)
result = await parser.aparse(file_path)
@@ -104,12 +96,10 @@ async def test_link_parse_result(file_path: str):
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.skip(reason="TODO: Needs to be fixed in prod. Raising 500 error.")
async def test_parse_structured_output(file_path: str):
parser = LlamaParse(
structured_output=True,
structured_output_json_schema_name="imFeelingLucky",
invalidate_cache=True,
)
result = await parser.aparse(file_path)
assert isinstance(result, JobResult)
@@ -152,11 +142,8 @@ async def test_parse_layout(file_path: str):
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.parametrize("partition_pages", [None, 2])
def test_parse_multiple_files(
file_path: str, chart_file_path: str, partition_pages: Optional[int]
):
parser = LlamaParse(partition_pages=partition_pages)
def test_parse_multiple_files(file_path: str, chart_file_path: str):
parser = LlamaParse()
result = parser.parse([file_path, chart_file_path])
assert isinstance(result, list)
@@ -165,40 +152,3 @@ def test_parse_multiple_files(
assert isinstance(result[1], JobResult)
assert result[0].file_name == file_path
assert result[1].file_name == chart_file_path
@pytest.mark.asyncio
@pytest.mark.skipif(
os.environ.get("LLAMA_CLOUD_API_KEY", "") == "",
reason="LLAMA_CLOUD_API_KEY not set",
)
@pytest.mark.parametrize("partition_pages", [None, 2])
async def test_multiple_page_parse_result(
multiple_page_path: str, partition_pages: Optional[int]
):
parser = LlamaParse(
take_screenshot=True,
auto_mode=True,
fast_mode=False,
partition_pages=partition_pages,
)
results = await parser.aparse(multiple_page_path)
if partition_pages is None:
assert isinstance(results, JobResult)
results = [results]
else:
assert isinstance(results, list)
for result in results:
assert isinstance(result, JobResult)
assert result.job_id is not None
assert result.file_name == multiple_page_path
assert len(result.pages) > 0
assert result.pages[0].text is not None
assert len(result.pages[0].text) > 0
assert result.pages[0].md is not None
assert len(result.pages[0].md) > 0
assert result.pages[0].md != result.pages[0].text
-30
View File
@@ -1,30 +0,0 @@
import pytest
from llama_cloud_services.parse.utils import expand_target_pages, partition_pages
def test_expand_target_pages() -> None:
with pytest.raises(ValueError):
list(expand_target_pages("x"))
with pytest.raises(ValueError):
list(expand_target_pages("1-2-3"))
with pytest.raises(ValueError):
list(expand_target_pages("2-1"))
result = list(expand_target_pages("0,2-3,5,8-10"))
assert result == [0, 2, 3, 5, 8, 9, 10]
def test_partion_pages() -> None:
pages = [0, 2, 3, 5, 8, 9, 10]
with pytest.raises(ValueError):
list(partition_pages(pages, 0))
result = list(partition_pages(pages, 3))
assert result == ["0,2-3", "5,8-9", "10"]
with pytest.raises(ValueError):
list(partition_pages(pages, 3, 0))
result = list(partition_pages(pages, 3, max_pages=5))
assert result == ["0,2-3", "5,8"]
result = list(partition_pages(pages, 3, max_pages=10))
assert result == ["0,2-3", "5,8-9", "10"]
+1 -2
View File
@@ -7,8 +7,7 @@ from llama_cloud_services.report import LlamaReport, ReportClient
# Skip tests if no API key is set
pytestmark = pytest.mark.skipif(
not os.getenv("LLAMA_CLOUD_API_KEY") or os.getenv("CI") == "true",
reason="No API key provided",
not os.getenv("LLAMA_CLOUD_API_KEY"), reason="No API key provided"
)
Binary file not shown.
-66
View File
@@ -1,66 +0,0 @@
from pydantic import BaseModel
from llama_cloud_services.utils import check_extra_params
class MyModel(BaseModel):
name: str
age: int
email: str
is_active: bool
def test_check_extra_params_no_extra():
"""Test when all parameters are valid - should return empty lists."""
data = {"name": "John", "age": 25, "email": "john@example.com", "is_active": True}
extra_params, suggestions = check_extra_params(MyModel, data)
assert extra_params == []
assert suggestions == []
def test_check_extra_params_with_typos():
"""Test when there are extra parameters that are close to valid ones (typos)."""
data = {
"name": "John",
"age": 25,
"emial": "john@example.com", # typo: emial instead of email
"is_activ": True, # typo: is_activ instead of is_active
"address": "123 Main St", # completely different parameter
}
extra_params, suggestions = check_extra_params(MyModel, data)
assert len(extra_params) == 3
assert "emial" in extra_params
assert "is_activ" in extra_params
assert "address" in extra_params
# Check that typo suggestions are provided
assert len(suggestions) == 3
assert "Did you mean 'email' instead of 'emial'?" in suggestions[0]
assert "Did you mean 'is_active' instead of 'is_activ'?" in suggestions[1]
assert "check the documentation or update the package" in suggestions[2]
def test_check_extra_params_completely_invalid():
"""Test when there are extra parameters with no close matches."""
data = {
"name": "John",
"xyz": "invalid",
"random_field": 123,
"completely_different": True,
}
extra_params, suggestions = check_extra_params(MyModel, data)
assert len(extra_params) == 3
assert "xyz" in extra_params
assert "random_field" in extra_params
assert "completely_different" in extra_params
# All suggestions should be generic (no close matches)
assert len(suggestions) == 3
for suggestion in suggestions:
assert "check the documentation or update the package" in suggestion
assert "Did you mean" not in suggestion