mirror of
https://github.com/run-llama/llama_cloud_services.git
synced 2026-07-21 12:05:23 -04:00
Compare commits
10 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7bfd3b2e0a | |||
| 3877c42b94 | |||
| 3863b6f76a | |||
| 0bfbcb61b9 | |||
| 6f334251ea | |||
| 54561e2dd2 | |||
| bfaec79a8f | |||
| 3e0e522a6b | |||
| f70b6d87ec | |||
| 693b5b83b1 |
@@ -24,8 +24,8 @@ from workflows import Context
|
||||
|
||||
dotenv.load_dotenv()
|
||||
|
||||
# Global context for loaded dataframes
|
||||
_dataframe_context: Dict[str, Any] = {}
|
||||
# Global context for executed code
|
||||
_code_context: Dict[str, Any] = {}
|
||||
|
||||
|
||||
# Helper function for initial agent context
|
||||
@@ -79,36 +79,30 @@ def list_extracted_data(data_dir: str = "data") -> str:
|
||||
|
||||
|
||||
# Agent tool for code execution against dataframes
|
||||
def execute_dataframe_code(
|
||||
code: str, load_files: Optional[Dict[str, str]] = None
|
||||
) -> str:
|
||||
def execute_code(code: str) -> str:
|
||||
"""
|
||||
Execute Python pandas code against LlamaSheets extracted data.
|
||||
Execute Python pandas code against LlamaSheets extracted data.
|
||||
|
||||
This tool allows flexible data analysis by executing arbitrary pandas code.
|
||||
You can load parquet files, manipulate dataframes, and return results.
|
||||
This tool allows flexible data analysis by executing arbitrary pandas code.
|
||||
You can load parquet files, manipulate dataframes, and return results.
|
||||
|
||||
The code executes in a context where:
|
||||
- pandas is available as 'pd'
|
||||
- json is available for formatting output
|
||||
- Previously loaded dataframes are accessible by their variable names
|
||||
The code executes in a context where:
|
||||
- pandas is available as 'pd'
|
||||
- json is available for formatting output
|
||||
|
||||
Args:
|
||||
code: Python code to execute. Any print() statements or stdout/stderr
|
||||
will be captured and returned. Optionally set a 'result' variable
|
||||
for structured output.
|
||||
load_files: Optional dict mapping variable names to file paths to load
|
||||
Example: {"df": "data/sales_region_1.parquet",
|
||||
"meta": "data/sales_metadata_1.parquet"}
|
||||
Args:
|
||||
code: Python code to execute. Any print() statements or stdout/stderr
|
||||
will be captured and returned. Optionally set a 'result' variable
|
||||
for structured output.
|
||||
|
||||
Returns:
|
||||
String containing:
|
||||
- Any stdout/stderr output from the code execution
|
||||
- The 'result' variable if it was set (formatted appropriately)
|
||||
- Error message if execution failed
|
||||
Returns:
|
||||
String containing:
|
||||
- Any stdout/stderr output from the code execution
|
||||
- The 'result' variable if it was set (formatted appropriately)
|
||||
- Error message if execution failed
|
||||
|
||||
Example usage:
|
||||
code = '''
|
||||
Example usage:
|
||||
code = '''
|
||||
# Load and inspect data
|
||||
df = pd.read_parquet("data/sales_region_1.parquet")
|
||||
print(f"Loaded {len(df)} rows")
|
||||
@@ -118,9 +112,9 @@ def execute_dataframe_code(
|
||||
"columns": list(df.columns),
|
||||
"sample": df.head(3).to_dict(orient="records")
|
||||
}
|
||||
'''
|
||||
'''
|
||||
"""
|
||||
global _dataframe_context
|
||||
global _code_context
|
||||
|
||||
# Capture stdout and stderr
|
||||
stdout_capture = io.StringIO()
|
||||
@@ -138,24 +132,17 @@ def execute_dataframe_code(
|
||||
"pd": pd,
|
||||
"json": json,
|
||||
"Path": Path,
|
||||
**_dataframe_context, # Include previously loaded dataframes
|
||||
**_code_context, # Include previously loaded dataframes
|
||||
}
|
||||
|
||||
# Load any requested files into context
|
||||
if load_files:
|
||||
for var_name, file_path in load_files.items():
|
||||
if file_path.endswith(".parquet"):
|
||||
exec_context[var_name] = pd.read_parquet(file_path)
|
||||
# Also save to global context for future calls
|
||||
_dataframe_context[var_name] = exec_context[var_name]
|
||||
elif file_path.endswith(".json"):
|
||||
with open(file_path, "r") as f:
|
||||
exec_context[var_name] = json.load(f)
|
||||
_dataframe_context[var_name] = exec_context[var_name]
|
||||
|
||||
# Execute the code
|
||||
exec(code, exec_context)
|
||||
|
||||
# Update global context with any new variables (excluding built-ins and modules)
|
||||
for key, value in exec_context.items():
|
||||
if not key.startswith("_") and key not in ["pd", "json", "Path"]:
|
||||
_code_context[key] = value
|
||||
|
||||
# Restore stdout/stderr
|
||||
sys.stdout = old_stdout
|
||||
sys.stderr = old_stderr
|
||||
@@ -223,8 +210,8 @@ def create_llamasheets_agent(
|
||||
# Initialize LLM
|
||||
llm = OpenAI(model=llm_model, api_key=api_key)
|
||||
|
||||
# Create tools - just 4 simple but powerful tools
|
||||
tools = [execute_dataframe_code]
|
||||
# Create tools list
|
||||
tools = [execute_code]
|
||||
|
||||
# System prompt to guide the agent
|
||||
available_regions = list_extracted_data()
|
||||
@@ -238,11 +225,8 @@ LlamaSheets extracts messy spreadsheets into clean parquet files with two types
|
||||
- Type detection: data_type, is_date_like, is_percentage, is_currency
|
||||
- Layout: is_in_first_row, is_merged_cell, horizontal_alignment
|
||||
|
||||
Your approach:
|
||||
1. Use list_extracted_data() to discover available files
|
||||
2. Use execute_dataframe_code() to load and analyze data with pandas
|
||||
3. Use metadata to understand structure (bold = headers, colors = groups)
|
||||
4. Use save_dataframe() to export results
|
||||
You have access to tools that allow you to execute Python pandas code against these files.
|
||||
Use these tools to load the parquet files, analyze the data, and return results.
|
||||
|
||||
Key tips:
|
||||
- Bold cells in metadata often indicate headers
|
||||
@@ -299,7 +283,7 @@ async def main():
|
||||
print(ev.delta, end="", flush=True)
|
||||
|
||||
_ = await handler
|
||||
print("=== End Query ===\n")
|
||||
print("\n=== End Query ===\n")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -1,5 +1,11 @@
|
||||
# llama-cloud-services-py
|
||||
|
||||
## 0.6.82
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- bfaec79: Update for new page number params
|
||||
|
||||
## 0.6.81
|
||||
|
||||
### Patch Changes
|
||||
|
||||
@@ -285,7 +285,7 @@ class LlamaParse(BasePydanticReader):
|
||||
description="Note: Non compatible with gpt-4o. If set to true, the parser will use a faster mode to extract text from documents. This mode will skip OCR of images, and table/heading reconstruction.",
|
||||
)
|
||||
|
||||
guess_xlsx_sheet_names: Optional[bool] = Field(
|
||||
guess_xlsx_sheet_name: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="Whether to guess the sheet names of the xlsx file.",
|
||||
)
|
||||
@@ -313,6 +313,10 @@ class LlamaParse(BasePydanticReader):
|
||||
default=False,
|
||||
description="If set to true, the parser will ignore document elements for layout detection and only rely on a vision model.",
|
||||
)
|
||||
inline_images_in_markdown: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will inline images in the markdown output.",
|
||||
)
|
||||
input_s3_region: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The region of the input S3 bucket if input_s3_path is specified.",
|
||||
@@ -329,6 +333,10 @@ class LlamaParse(BasePydanticReader):
|
||||
default=None,
|
||||
description="The maximum timeout in seconds to wait for the parsing to finish. Override default timeout of 30 minutes. Minimum is 120 seconds.",
|
||||
)
|
||||
keep_page_separator_when_merging_tables: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will keep the page separator when merging tables across pages.",
|
||||
)
|
||||
language: Optional[str] = Field(
|
||||
default="en", description="The language of the text to parse."
|
||||
)
|
||||
@@ -400,6 +408,10 @@ class LlamaParse(BasePydanticReader):
|
||||
default=False,
|
||||
description="If set, the parser will try to preserve very small text lines. This can be useful for documents containing vector graphics with very small text lines that may not be recognized by OCR or a vision model (such as in CAD drawings).",
|
||||
)
|
||||
presentation_out_of_bounds_content: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will include out-of-bounds content in presentation files.",
|
||||
)
|
||||
precise_bounding_box: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will use a more precise bounding box to extract text from documents. This will increase the accuracy of the parsing job, but reduce the speed.",
|
||||
@@ -416,6 +428,14 @@ class LlamaParse(BasePydanticReader):
|
||||
default=None,
|
||||
description="A suffix to add after error message in failed pages. If not set, no suffix will be used.",
|
||||
)
|
||||
remove_hidden_text: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will remove hidden text from the document.",
|
||||
)
|
||||
save_images: Optional[bool] = Field(
|
||||
default=True,
|
||||
description="If set to true, the parser will save images extracted from the document.",
|
||||
)
|
||||
skip_diagonal_text: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will ignore diagonal text (when the text rotation in degrees modulo 90 is not 0).",
|
||||
@@ -440,6 +460,10 @@ class LlamaParse(BasePydanticReader):
|
||||
default=False,
|
||||
description="If set to true, the parser will use a specialized one-shot chart parsing model to extract data from charts. This model is able to understand the chart type and extract the data accordingly. It is more accurate than the efficient model, but also more expensive.",
|
||||
)
|
||||
specialized_image_parsing: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will use a specialized image parsing model to extract data from images.",
|
||||
)
|
||||
strict_mode_buggy_font: Optional[bool] = Field(
|
||||
default=False,
|
||||
description="If set to true, the parser will fail if it can't extract text from a document because of a buggy font.",
|
||||
@@ -536,6 +560,10 @@ class LlamaParse(BasePydanticReader):
|
||||
default=None,
|
||||
description="A prefix to add to the page footer in the output markdown.",
|
||||
)
|
||||
extract_printed_page_number: Optional[bool] = Field(
|
||||
default=None,
|
||||
description="Whether to extract the printed page numbers from pages in the document.",
|
||||
)
|
||||
|
||||
# Deprecated
|
||||
bounding_box: Optional[str] = Field(
|
||||
@@ -580,6 +608,23 @@ class LlamaParse(BasePydanticReader):
|
||||
description="Automatically check for Python SDK updates.",
|
||||
)
|
||||
|
||||
@model_validator(mode="before")
|
||||
@classmethod
|
||||
def handle_deprecated_params(cls, data: Dict[str, Any]) -> Dict[str, Any]:
|
||||
# Handle deprecated guess_xlsx_sheet_names -> guess_xlsx_sheet_name
|
||||
if "guess_xlsx_sheet_names" in data:
|
||||
warnings.warn(
|
||||
"The parameter 'guess_xlsx_sheet_names' is deprecated and will be removed in a future release. "
|
||||
"Use 'guess_xlsx_sheet_name' instead.",
|
||||
DeprecationWarning,
|
||||
stacklevel=2,
|
||||
)
|
||||
# Only set the new parameter if it's not already explicitly set
|
||||
if "guess_xlsx_sheet_name" not in data:
|
||||
data["guess_xlsx_sheet_name"] = data["guess_xlsx_sheet_names"]
|
||||
del data["guess_xlsx_sheet_names"]
|
||||
return data
|
||||
|
||||
@model_validator(mode="before")
|
||||
@classmethod
|
||||
def warn_extra_params(cls, data: Dict[str, Any]) -> Dict[str, Any]:
|
||||
@@ -695,11 +740,9 @@ class LlamaParse(BasePydanticReader):
|
||||
file_path = str(file_input)
|
||||
file_ext = os.path.splitext(file_path)[1].lower()
|
||||
if file_ext not in SUPPORTED_FILE_TYPES:
|
||||
raise Exception(
|
||||
f"Currently, only the following file types are supported: {SUPPORTED_FILE_TYPES}\n"
|
||||
f"Current file type: {file_ext}"
|
||||
)
|
||||
mime_type = mimetypes.guess_type(file_path)[0]
|
||||
mime_type = "application/octet-stream"
|
||||
else:
|
||||
mime_type = mimetypes.guess_type(file_path)[0]
|
||||
# Open the file here for the duration of the async context
|
||||
# load data, set the mime type
|
||||
fs = fs or get_default_fs()
|
||||
@@ -820,8 +863,8 @@ class LlamaParse(BasePydanticReader):
|
||||
)
|
||||
data["formatting_instruction"] = self.formatting_instruction
|
||||
|
||||
if self.guess_xlsx_sheet_names:
|
||||
data["guess_xlsx_sheet_names"] = self.guess_xlsx_sheet_names
|
||||
if self.guess_xlsx_sheet_name:
|
||||
data["guess_xlsx_sheet_name"] = self.guess_xlsx_sheet_name
|
||||
|
||||
if self.html_make_all_elements_visible:
|
||||
data["html_make_all_elements_visible"] = self.html_make_all_elements_visible
|
||||
@@ -845,6 +888,9 @@ class LlamaParse(BasePydanticReader):
|
||||
"ignore_document_elements_for_layout_detection"
|
||||
] = self.ignore_document_elements_for_layout_detection
|
||||
|
||||
if self.inline_images_in_markdown:
|
||||
data["inline_images_in_markdown"] = self.inline_images_in_markdown
|
||||
|
||||
if input_url is not None:
|
||||
files = None
|
||||
data["input_url"] = str(input_url)
|
||||
@@ -873,6 +919,11 @@ class LlamaParse(BasePydanticReader):
|
||||
if self.job_timeout_in_seconds is not None:
|
||||
data["job_timeout_in_seconds"] = self.job_timeout_in_seconds
|
||||
|
||||
if self.keep_page_separator_when_merging_tables:
|
||||
data[
|
||||
"keep_page_separator_when_merging_tables"
|
||||
] = self.keep_page_separator_when_merging_tables
|
||||
|
||||
if self.language:
|
||||
data["language"] = self.language
|
||||
|
||||
@@ -951,6 +1002,11 @@ class LlamaParse(BasePydanticReader):
|
||||
if self.preserve_very_small_text:
|
||||
data["preserve_very_small_text"] = self.preserve_very_small_text
|
||||
|
||||
if self.presentation_out_of_bounds_content:
|
||||
data[
|
||||
"presentation_out_of_bounds_content"
|
||||
] = self.presentation_out_of_bounds_content
|
||||
|
||||
if self.preset is not None:
|
||||
data["preset"] = self.preset
|
||||
|
||||
@@ -970,6 +1026,11 @@ class LlamaParse(BasePydanticReader):
|
||||
"replace_failed_page_with_error_message_suffix"
|
||||
] = self.replace_failed_page_with_error_message_suffix
|
||||
|
||||
if self.remove_hidden_text:
|
||||
data["remove_hidden_text"] = self.remove_hidden_text
|
||||
|
||||
data["save_images"] = self.save_images
|
||||
|
||||
if self.skip_diagonal_text:
|
||||
data["skip_diagonal_text"] = self.skip_diagonal_text
|
||||
|
||||
@@ -994,6 +1055,9 @@ class LlamaParse(BasePydanticReader):
|
||||
if self.specialized_chart_parsing_plus:
|
||||
data["specialized_chart_parsing_plus"] = self.specialized_chart_parsing_plus
|
||||
|
||||
if self.specialized_image_parsing:
|
||||
data["specialized_image_parsing"] = self.specialized_image_parsing
|
||||
|
||||
if self.strict_mode_buggy_font:
|
||||
data["strict_mode_buggy_font"] = self.strict_mode_buggy_font
|
||||
|
||||
@@ -1049,6 +1113,9 @@ class LlamaParse(BasePydanticReader):
|
||||
"markdown_table_multiline_header_separator"
|
||||
] = self.markdown_table_multiline_header_separator
|
||||
|
||||
if self.extract_printed_page_number is not None:
|
||||
data["extract_printed_page_number"] = self.extract_printed_page_number
|
||||
|
||||
# Deprecated
|
||||
if self.bounding_box is not None:
|
||||
data["bounding_box"] = self.bounding_box
|
||||
|
||||
@@ -250,6 +250,19 @@ class Page(SafeBaseModel):
|
||||
slideSpeakerNotes: Optional[str] = Field(
|
||||
default=None, description="The speaker notes for the slide."
|
||||
)
|
||||
confidence: Optional[float] = Field(
|
||||
default=None, description="The confidence of the page parsing."
|
||||
)
|
||||
printedPageNumber: Optional[str] = Field(
|
||||
default=None,
|
||||
description="The printed page number on the page, if found and extractPrintedPageNumber is set to true.",
|
||||
)
|
||||
pageHeaderMarkdown: Optional[str] = Field(
|
||||
default=None, description="The page header in markdown format."
|
||||
)
|
||||
pageFooterMarkdown: Optional[str] = Field(
|
||||
default=None, description="The page footer in markdown format."
|
||||
)
|
||||
|
||||
|
||||
class JobResult(SafeBaseModel):
|
||||
|
||||
@@ -1,5 +1,12 @@
|
||||
# llama_parse
|
||||
|
||||
## 0.6.82
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- Updated dependencies [bfaec79]
|
||||
- llama-cloud-services-py@0.6.82
|
||||
|
||||
## 0.6.81
|
||||
|
||||
### Patch Changes
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "llama_parse",
|
||||
"version": "0.6.81",
|
||||
"version": "0.6.82",
|
||||
"description": "",
|
||||
"main": "index.js",
|
||||
"private": false,
|
||||
|
||||
@@ -11,13 +11,13 @@ dev = [
|
||||
|
||||
[project]
|
||||
name = "llama-parse"
|
||||
version = "0.6.81"
|
||||
version = "0.6.83"
|
||||
description = "Parse files into RAG-Optimized formats."
|
||||
authors = [{name = "Logan Markewich", email = "logan@llamaindex.ai"}]
|
||||
requires-python = ">=3.9,<4.0"
|
||||
readme = "README.md"
|
||||
license = "MIT"
|
||||
dependencies = ["llama-cloud-services>=0.6.81"]
|
||||
dependencies = ["llama-cloud-services>=0.6.82"]
|
||||
|
||||
[project.scripts]
|
||||
llama-parse = "llama_parse.cli.main:parse"
|
||||
|
||||
+1
-1
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "llama-cloud-services-py",
|
||||
"version": "0.6.81",
|
||||
"version": "0.6.82",
|
||||
"private": false,
|
||||
"license": "MIT",
|
||||
"scripts": {},
|
||||
|
||||
+1
-1
@@ -22,7 +22,7 @@ dev = [
|
||||
|
||||
[project]
|
||||
name = "llama-cloud-services"
|
||||
version = "0.6.81"
|
||||
version = "0.6.84"
|
||||
description = "Tailored SDK clients for LlamaCloud services."
|
||||
authors = [{name = "Logan Markewich", email = "logan@runllama.ai"}]
|
||||
requires-python = ">=3.9,<4.0"
|
||||
|
||||
@@ -1,5 +1,11 @@
|
||||
# llama-cloud-services
|
||||
|
||||
## 0.4.2
|
||||
|
||||
### Patch Changes
|
||||
|
||||
- bfaec79: Update for new page number params
|
||||
|
||||
## 0.4.1
|
||||
|
||||
### Patch Changes
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "llama-cloud-services",
|
||||
"version": "0.4.1",
|
||||
"version": "0.4.2",
|
||||
"type": "module",
|
||||
"license": "MIT",
|
||||
"scripts": {
|
||||
|
||||
@@ -185,6 +185,7 @@ export class LlamaParseReader extends FileReader {
|
||||
page_footer_prefix?: string | undefined;
|
||||
page_footer_suffix?: string | undefined;
|
||||
merge_tables_across_pages_in_markdown?: boolean | undefined;
|
||||
extract_printed_page_number?: boolean | undefined;
|
||||
|
||||
constructor(
|
||||
params: Partial<Omit<LlamaParseReader, "language" | "apiKey">> & {
|
||||
@@ -381,6 +382,7 @@ export class LlamaParseReader extends FileReader {
|
||||
page_footer_suffix: this.page_footer_suffix,
|
||||
merge_tables_across_pages_in_markdown:
|
||||
this.merge_tables_across_pages_in_markdown,
|
||||
extract_printed_page_number: this.extract_printed_page_number,
|
||||
} satisfies {
|
||||
[Key in keyof BodyUploadFileApiParsingUploadPost]-?:
|
||||
| BodyUploadFileApiParsingUploadPost[Key]
|
||||
|
||||
Reference in New Issue
Block a user