Compare commits

...

4 Commits

Author SHA1 Message Date
Adrian Lyjak d2b5fcd690 ignore npmrc 2025-10-03 00:20:18 -04:00
Adrian Lyjak d028397603 version and release via changesets (#849) 2025-10-03 00:08:52 -04:00
Emanuel Ferreira 35ea8476db docs: parse -> classify -> extract (#931) 2025-09-24 18:52:15 -03:00
Logan 3e5f7c4f1e Update parse.md 2025-09-24 11:35:13 -06:00
17 changed files with 1693 additions and 360 deletions
+8
View File
@@ -0,0 +1,8 @@
# Changesets
Hello and welcome! This folder has been automatically generated by `@changesets/cli`, a build tool that works
with multi-package repos, or single-package repos to help you version and publish your code. You can
find the full documentation for it [in our repository](https://github.com/changesets/changesets)
We have a quick list of common questions to get you started engaging with this project in
[our documentation](https://github.com/changesets/changesets/blob/main/docs/common-questions.md)
+11
View File
@@ -0,0 +1,11 @@
{
"$schema": "https://unpkg.com/@changesets/config@3.1.1/schema.json",
"changelog": "@changesets/cli/changelog",
"commit": false,
"fixed": [],
"linked": [],
"access": "restricted",
"baseBranch": "main",
"updateInternalDependencies": "patch",
"ignore": []
}
+6
View File
@@ -0,0 +1,6 @@
---
"llama-cloud-services": patch
"llama-cloud-services-py": patch
---
Update llama-cloud api version, and integrate with agent data deletion
-66
View File
@@ -1,66 +0,0 @@
name: Publish Release - Python
on:
push:
tags:
- "v*"
workflow_dispatch:
env:
UV_VERSION: "0.7.20"
jobs:
build-n-publish:
name: Build and publish to PyPI
if: github.repository == 'run-llama/llama_cloud_services'
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v5
- name: Install uv
uses: astral-sh/setup-uv@v6
with:
version: ${{ env.UV_VERSION }}
- name: Set up Python
run: uv python install
- name: Display Python version
run: python --version
- name: Build
working-directory: py
run: uv build
- name: Test installing built package
shell: bash
working-directory: py
run: |
uv venv
uv pip install dist/*.whl
- name: Publish package
shell: bash
working-directory: py
run: uv publish --token ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
- name: Build and publish llama-parse
working-directory: py/llama_parse/
run: |
uv build
uv publish --token ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
- name: Create GitHub Release
id: create_release
uses: actions/create-release@v1
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} # This token is provided by Actions, you do not need to create your own token
with:
tag_name: ${{ github.ref }}
release_name: ${{ github.ref }} - LlamaCloud Services PY
artifacts: "py/**/dist/*"
generateReleaseNotes: true
draft: false
prerelease: false
-52
View File
@@ -1,52 +0,0 @@
name: Publish Release - TypeScript
on:
push:
tags:
- "llama-cloud-services@*"
jobs:
build-and-publish:
runs-on: ubuntu-latest
steps:
- name: Checkout Repo
uses: actions/checkout@v5
- uses: pnpm/action-setup@v4
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version-file: "ts/llama_cloud_services/.nvmrc"
- name: Install dependencies
run: pnpm install --no-frozen-lockfile
- name: Run Build
working-directory: ts/llama_cloud_services/
run: pnpm build
- name: Build tarball
run: |
pnpm pack
working-directory: ts/llama_cloud_services
- name: Setup npm authentication
run: echo "//registry.npmjs.org/:_authToken=${NPM_TOKEN}" > ~/.npmrc
env:
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
- name: Release
working-directory: ts/llama_cloud_services
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
run: pnpm publish --access public --no-git-checks
- name: Create release
uses: ncipollo/release-action@v1
with:
artifacts: "ts/llama_cloud_services/llama-cloud-services*.tgz"
name: Release ${{ github.ref_name }} - LlamaCloud Services TS
generateReleaseNotes: true
token: ${{ secrets.GITHUB_TOKEN }}
@@ -0,0 +1,61 @@
name: Version Bump and Release
on:
push:
branches:
- main
concurrency: ${{ github.workflow }}-${{ github.ref }}
jobs:
release:
name: Release
runs-on: ubuntu-latest
# Only run on main branch pushes
if: github.ref == 'refs/heads/main'
steps:
- name: Checkout Repo
uses: actions/checkout@v4
- uses: pnpm/action-setup@v3
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version: "22"
cache: "pnpm"
- name: Setup Python
uses: actions/setup-python@v5
with:
python-version: "3.11"
- name: Install uv
uses: astral-sh/setup-uv@v3
- name: Install dependencies
run: pnpm install
- name: Add auth token to .npmrc file
run: |
cat << EOF >> ".npmrc"
//registry.npmjs.org/:_authToken=$NPM_TOKEN
EOF
env:
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
- name: Create Release Pull Request or Publish packages
id: changesets
uses: changesets/action@v1
with:
commit: "chore: version packages"
title: "chore: version packages"
# Custom version script
version: pnpm -w run version
# Custom publish script
publish: pnpm -w run publish
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
UV_PUBLISH_TOKEN: ${{ secrets.PYPI_TOKEN }}
LLAMA_PARSE_PYPI_TOKEN: ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
+1
View File
@@ -9,3 +9,4 @@ __pycache__/
node_modules/
.turbo/
dist/
.npmrc
+2 -2
View File
@@ -1035,7 +1035,7 @@
],
"metadata": {
"kernelspec": {
"display_name": ".venv",
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
@@ -1052,5 +1052,5 @@
}
},
"nbformat": 4,
"nbformat_minor": 2
"nbformat_minor": 4
}
@@ -0,0 +1,765 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Complete Parse → Classify → Extract Workflow with LlamaCloud Services\n",
"\n",
"This notebook demonstrates the complete workflow for processing documents using LlamaCloud services:\n",
"1. **Parse** - Extract and convert documents to markdown\n",
"2. **Classify** - Categorize documents based on their content\n",
"3. **Extract** - Extract structured data using the markdown as input via SourceText\n",
"\n",
"## Overview of the Workflow\n",
"\n",
"### 1. Parse Phase\n",
"- Use `LlamaParse` to convert documents (PDFs, Word docs, etc.) into structured formats\n",
"- Extract markdown content that preserves document structure\n",
"- Get both raw text and markdown representations\n",
"\n",
"### 2. Classify Phase\n",
"- Use `ClassifyClient` to categorize documents based on content\n",
"- Apply classification rules to route documents appropriately\n",
"- Handle different document types with specific processing logic\n",
"\n",
"### 3. Extract Phase\n",
"- Use `LlamaExtract` with `SourceText` to extract structured data\n",
"- Pass the markdown content as input for more accurate extraction\n",
"- Define custom schemas for structured data extraction\n",
"\n",
"Let's walk through each step with practical examples."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Setup and Installation"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"# Install required packages\n",
"!pip install llama-cloud-services\n",
"!pip install python-dotenv"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"✅ API key configured\n"
]
}
],
"source": [
"import os\n",
"import nest_asyncio\n",
"from getpass import getpass\n",
"from dotenv import load_dotenv\n",
"\n",
"# Load environment variables\n",
"load_dotenv()\n",
"nest_asyncio.apply()\n",
"\n",
"# Set up API key\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"\" # edit it\n",
"\n",
"# Setup Base URL\n",
"# os.envrion[\"LLAMA_CLOUD_BASE_URL\"] = \"https://api.cloud.eu.llamaindex.ai/\" # update if necessay\n",
"\n",
"print(\"✅ API key configured\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Download Sample Documents\n",
"\n",
"Let's download some sample documents to work with:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"📁 financial_report.pdf already exists\n",
"📁 technical_spec.pdf already exists\n",
"\n",
"📂 Sample documents ready!\n"
]
}
],
"source": [
"import requests\n",
"import os\n",
"\n",
"# Create directory for sample documents\n",
"os.makedirs(\"sample_docs\", exist_ok=True)\n",
"\n",
"# Download sample documents\n",
"docs_to_download = {\n",
" \"financial_report.pdf\": \"https://raw.githubusercontent.com/run-llama/llama_index/main/docs/docs/examples/data/10k/uber_2021.pdf\",\n",
" \"technical_spec.pdf\": \"https://www.ti.com/lit/ds/symlink/lm317.pdf\",\n",
"}\n",
"\n",
"for filename, url in docs_to_download.items():\n",
" filepath = f\"sample_docs/{filename}\"\n",
" if not os.path.exists(filepath):\n",
" print(f\"Downloading {filename}...\")\n",
" response = requests.get(url)\n",
" if response.status_code == 200:\n",
" with open(filepath, \"wb\") as f:\n",
" f.write(response.content)\n",
" print(f\"✅ Downloaded {filename}\")\n",
" else:\n",
" print(f\"❌ Failed to download {filename}\")\n",
" else:\n",
" print(f\"📁 {filename} already exists\")\n",
"\n",
"print(\"\\n📂 Sample documents ready!\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Phase 1: Document Parsing\n",
"\n",
"First, let's parse our documents using LlamaParse to extract clean markdown content."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"🔄 Parsing documents...\n",
"Started parsing the file under job_id 8a8c76f9-354d-4275-91d8-312ff1adc762\n",
"...✅ Parsed financial report (Job ID: 8a8c76f9-354d-4275-91d8-312ff1adc762)\n",
"Started parsing the file under job_id 7e603448-ed80-4d18-948b-6801ed51c41b\n",
"✅ Parsed technical spec (Job ID: 7e603448-ed80-4d18-948b-6801ed51c41b)\n",
"\n",
"📄 Parsing complete!\n"
]
}
],
"source": [
"from llama_cloud_services.parse.base import LlamaParse\n",
"from llama_cloud_services.parse.utils import ResultType\n",
"import asyncio\n",
"\n",
"# Initialize the parser\n",
"parser = LlamaParse(\n",
" result_type=ResultType.MD, # Get markdown output\n",
" verbose=True,\n",
" language=\"en\",\n",
" # Premium mode for better accuracy\n",
" premium_mode=True,\n",
" # Extract tables as HTML for better structure\n",
" output_tables_as_HTML=True,\n",
" # Parse only first few pages for demo\n",
")\n",
"\n",
"print(\"🔄 Parsing documents...\")\n",
"\n",
"# Parse the financial report\n",
"financial_result = await parser.aparse(\"sample_docs/financial_report.pdf\")\n",
"print(f\"✅ Parsed financial report (Job ID: {financial_result.job_id})\")\n",
"\n",
"# Parse the technical specification\n",
"technical_result = await parser.aparse(\"sample_docs/technical_spec.pdf\")\n",
"print(f\"✅ Parsed technical spec (Job ID: {technical_result.job_id})\")\n",
"\n",
"print(\"\\n📄 Parsing complete!\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Extract Markdown Content\n",
"\n",
"Now let's get the markdown content from our parsed documents:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"📋 Financial Report Markdown (first 500 chars):\n",
"\n",
"\n",
"# UNITED STATES\n",
"# SECURITIES AND EXCHANGE COMMISSION\n",
"Washington, D.C. 20549\n",
"\n",
"## FORM 10-K\n",
"\n",
"(Mark One)\n",
"\n",
"☒ ANNUAL REPORT PURSUANT TO SECTION 13 OR 15(d) OF THE SECURITIES EXCHANGE ACT OF 1934\n",
"For the fiscal year ended December 31, 2021\n",
"OR\n",
"☐ TRANSITION REPORT PURSUANT TO SECTION 13 OR 15(d) OF THE SECURITIES EXCHANGE ACT OF 1934\n",
"For the transition period from_____ to _____\n",
"Commission File Number: 001-38902\n",
"\n",
"# UBER TECHNOLOGIES, INC.\n",
"(Exact name of registrant as specified in its charter)\n",
"\n",
"Delaware\n",
"...\n",
"\n",
"📋 Technical Spec Markdown (first 500 chars):\n",
"\n",
"\n",
"LM317\n",
"SLVS044Z SEPTEMBER 1997 REVISED APRIL 2025\n",
"\n",
"# LM317 3-Pin Adjustable Regulator\n",
"\n",
"## 1 Features\n",
"\n",
"• Output voltage range:\n",
" Adjustable: 1.25V to 37V\n",
"• Output current: 1.5A\n",
"• Line regulation: 0.01%/V (typ)\n",
"• Load regulation: 0.1% (typ)\n",
"• Internal short-circuit current limiting\n",
"• Thermal overload protection\n",
"• Output safe-area compensation (new chip)\n",
"• PSRR: 80dB at 120Hz for CADJ = 10μF (new chip)\n",
"• Packages:\n",
" 4-pin, SOT-223 (DCY)\n",
" 3-pin, TO-263 (KTT)\n",
" 3-pin, TO-220 (KCS, KCT),\n",
"...\n",
"\n",
"📏 Financial report markdown length: 1348671 characters\n",
"📏 Technical spec markdown length: 90971 characters\n"
]
}
],
"source": [
"# Get markdown content from parsed documents\n",
"financial_markdown = await financial_result.aget_markdown()\n",
"technical_markdown = await technical_result.aget_markdown()\n",
"\n",
"print(\"📋 Financial Report Markdown (first 500 chars):\")\n",
"print(financial_markdown[:500])\n",
"print(\"...\\n\")\n",
"\n",
"print(\"📋 Technical Spec Markdown (first 500 chars):\")\n",
"print(technical_markdown[:500])\n",
"print(\"...\\n\")\n",
"\n",
"print(f\"📏 Financial report markdown length: {len(financial_markdown)} characters\")\n",
"print(f\"📏 Technical spec markdown length: {len(technical_markdown)} characters\")\n",
"\n",
"document_texts = [financial_markdown, technical_markdown]"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Phase 2: Document Classification\n",
"\n",
"Next, let's classify our documents based on their content using the ClassifyClient."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"🏷️ Setting up document classification...\n",
"📝 Created 3 classification rules\n"
]
}
],
"source": [
"from llama_cloud_services.beta.classifier.client import ClassifyClient\n",
"from llama_cloud.types import ClassifierRule\n",
"from llama_cloud_services.files.client import FileClient\n",
"from llama_cloud.client import AsyncLlamaCloud\n",
"\n",
"# Initialize the classify client\n",
"api_key = os.environ[\"LLAMA_CLOUD_API_KEY\"]\n",
"classify_client = ClassifyClient.from_api_key(api_key)\n",
"\n",
"print(\"🏷️ Setting up document classification...\")\n",
"\n",
"# Define classification rules\n",
"classification_rules = [\n",
" ClassifierRule(\n",
" type=\"financial_document\",\n",
" description=\"Documents containing financial data, revenue, expenses, SEC filings, or financial statements\",\n",
" ),\n",
" ClassifierRule(\n",
" type=\"technical_specification\",\n",
" description=\"Technical datasheets, component specifications, engineering documents, or technical manuals\",\n",
" ),\n",
" ClassifierRule(\n",
" type=\"general_document\",\n",
" description=\"General business documents, contracts, or other unspecified document types\",\n",
" ),\n",
"]\n",
"\n",
"print(f\"📝 Created {len(classification_rules)} classification rules\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Phase 3: Structured Data Extraction using SourceText\n",
"\n",
"Now comes the key part - using the markdown content as input for structured data extraction via SourceText."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"⚙️ LlamaExtract initialized\n"
]
}
],
"source": [
"from llama_cloud_services.extract.extract import LlamaExtract, SourceText\n",
"from llama_cloud.types import ExtractConfig, ExtractMode\n",
"from pydantic import BaseModel, Field\n",
"from typing import List, Optional\n",
"\n",
"# Initialize LlamaExtract\n",
"llama_extract = LlamaExtract(api_key=api_key, verbose=True)\n",
"\n",
"print(\"⚙️ LlamaExtract initialized\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Define Extraction Schemas\n",
"\n",
"Let's define different schemas for different document types:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"📋 Extraction schemas defined\n"
]
}
],
"source": [
"# Schema for financial documents\n",
"class FinancialMetrics(BaseModel):\n",
" company_name: str = Field(description=\"Name of the company\")\n",
" document_type: str = Field(\n",
" description=\"Type of financial document (10-K, 10-Q, annual report, etc.)\"\n",
" )\n",
" fiscal_year: int = Field(description=\"Fiscal year of the report\")\n",
" revenue_2021: str = Field(description=\"Total revenue in 2021\")\n",
" net_income_2021: str = Field(description=\"Net income in 2021\")\n",
" key_business_segments: List[str] = Field(\n",
" default=[], description=\"Main business segments or divisions\"\n",
" )\n",
" risk_factors: List[str] = Field(\n",
" default=[], description=\"Key risk factors mentioned\"\n",
" )\n",
"\n",
"\n",
"# Schema for technical specifications\n",
"class VoltageRange(BaseModel):\n",
" min_voltage: Optional[float] = Field(description=\"Minimum voltage\")\n",
" max_voltage: Optional[float] = Field(description=\"Maximum voltage\")\n",
" unit: str = Field(default=\"V\", description=\"Voltage unit\")\n",
"\n",
"\n",
"class TechnicalSpec(BaseModel):\n",
" component_name: str = Field(description=\"Name of the technical component\")\n",
" manufacturer: Optional[str] = Field(description=\"Manufacturer name\")\n",
" part_number: Optional[str] = Field(description=\"Part or model number\")\n",
" description: str = Field(description=\"Brief description of the component\")\n",
" operating_voltage: Optional[VoltageRange] = Field(\n",
" description=\"Operating voltage range\"\n",
" )\n",
" maximum_current: Optional[float] = Field(\n",
" description=\"Maximum current rating in amperes\"\n",
" )\n",
" key_features: List[str] = Field(\n",
" default=[], description=\"Key features and capabilities\"\n",
" )\n",
" applications: List[str] = Field(default=[], description=\"Typical applications\")\n",
"\n",
"\n",
"print(\"📋 Extraction schemas defined\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Complete Workflow Summary\n",
"\n",
"Let's create a function that demonstrates the complete workflow:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"🔧 Workflow function defined!\n"
]
}
],
"source": [
"import tempfile\n",
"from pathlib import Path\n",
"from llama_cloud import ExtractConfig\n",
"\n",
"\n",
"async def complete_document_workflow(markdown_content: str):\n",
" \"\"\"\n",
" Complete workflow: Parse → Classify → Extract\n",
" \"\"\"\n",
" print(f\"🚀 Starting complete workflow\")\n",
" print(\"=\" * 60)\n",
"\n",
" # Step 1: Classify\n",
" print(\"🏷️ Step 2: Classifying document...\")\n",
"\n",
" with tempfile.NamedTemporaryFile(\n",
" mode=\"w\", suffix=\".md\", delete=False, encoding=\"utf-8\"\n",
" ) as tmp:\n",
" tmp.write(markdown_content)\n",
" temp_path = Path(tmp.name)\n",
"\n",
" print(temp_path)\n",
"\n",
" classification = await classify_client.aclassify_file_path(\n",
" rules=classification_rules, file_input_path=str(temp_path)\n",
" )\n",
" doc_type = classification.items[0].result.type\n",
" confidence = classification.items[0].result.confidence\n",
" print(f\" ✅ Classified as: {doc_type} (confidence: {confidence:.2f})\")\n",
"\n",
" # Step 2: Extract based on classification\n",
" print(\"🔍 Step 3: Extracting structured data using SourceText...\")\n",
" source_text = SourceText(\n",
" text_content=markdown_content,\n",
" filename=f\"{os.path.basename(temp_path)}_markdown.md\",\n",
" )\n",
"\n",
" # Choose schema based on classification\n",
" if \"financial\" in doc_type.lower():\n",
" schema = FinancialMetrics\n",
" print(\" 📊 Using FinancialMetrics schema\")\n",
" elif \"technical\" in doc_type.lower():\n",
" schema = TechnicalSpec\n",
" print(\" 🔧 Using TechnicalSpec schema\")\n",
" else:\n",
" schema = FinancialMetrics # Default fallback\n",
" print(\" 📊 Using default FinancialMetrics schema\")\n",
"\n",
" extract_config = ExtractConfig(\n",
" extraction_mode=\"BALANCED\",\n",
" )\n",
"\n",
" extraction_result = llama_extract.extract(\n",
" data_schema=schema, config=extract_config, files=source_text\n",
" )\n",
"\n",
" print(\" ✅ Extraction complete!\")\n",
"\n",
" return {\n",
" \"file_path\": temp_path,\n",
" \"markdown_length\": len(markdown_content),\n",
" \"classification\": doc_type,\n",
" \"confidence\": confidence,\n",
" \"extracted_data\": extraction_result.data,\n",
" \"markdown_sample\": markdown_content[:200] + \"...\"\n",
" if len(markdown_content) > 200\n",
" else markdown_content,\n",
" }\n",
"\n",
"\n",
"print(\"🔧 Workflow function defined!\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Run Complete Workflow on Both Documents"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"🚀 Starting complete workflow\n",
"============================================================\n",
"🏷️ Step 2: Classifying document...\n",
"/var/folders/g6/4b5lpp5974gcpr890ybhbw4r0000gn/T/tmpos3b62tm.md\n",
" ✅ Classified as: financial_document (confidence: 1.00)\n",
"🔍 Step 3: Extracting structured data using SourceText...\n",
" 📊 Using FinancialMetrics schema\n",
".. ✅ Extraction complete!\n",
"\n",
"============================================================\n",
"\n",
"🚀 Starting complete workflow\n",
"============================================================\n",
"🏷️ Step 2: Classifying document...\n",
"/var/folders/g6/4b5lpp5974gcpr890ybhbw4r0000gn/T/tmpppz9ub_m.md\n",
" ✅ Classified as: technical_specification (confidence: 1.00)\n",
"🔍 Step 3: Extracting structured data using SourceText...\n",
" 🔧 Using TechnicalSpec schema\n",
" ✅ Extraction complete!\n",
"\n",
"============================================================\n",
"\n",
"📋 Processed 2 documents successfully!\n"
]
}
],
"source": [
"# Process both documents through the complete workflow\n",
"results = []\n",
"\n",
"for doc_text in document_texts:\n",
" try:\n",
" result = await complete_document_workflow(doc_text)\n",
" results.append(result)\n",
" print(\"\\n\" + \"=\" * 60 + \"\\n\")\n",
" except Exception as e:\n",
" print(f\"❌ Error processing {doc_path}: {str(e)}\")\n",
" print(\"\\n\" + \"=\" * 60 + \"\\n\")\n",
"\n",
"print(f\"📋 Processed {len(results)} documents successfully!\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Final Results Summary"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"📈 COMPLETE WORKFLOW RESULTS SUMMARY\n",
"======================================================================\n",
"\n",
"📄 Document 1: tmpos3b62tm.md\n",
" 📊 Classification: financial_document (confidence: 1.00)\n",
" 📝 Markdown length: 1,348,671 characters\n",
" 📋 Markdown sample: \n",
"\n",
"# UNITED STATES\n",
"# SECURITIES AND EXCHANGE COMMISSION\n",
"Washington, D.C. 20549\n",
"\n",
"## FORM 10-K\n",
"\n",
"(Mark O...\n",
" 🎯 Extracted fields: 7 fields\n",
" • company_name: Uber Technologies, Inc.\n",
" • document_type: Annual Report on Form 10-K\n",
" • fiscal_year: 2021\n",
" • revenue_2021: $21,764\n",
" • net_income_2021: $(496)\n",
" • key_business_segments: ['Mobility', 'Delivery', 'Freight', 'All Other (including former New Mobility, e-bikes, e-scooters, Advanced Technologies Group and other technology programs)']\n",
" • risk_factors: [\"The company faces numerous risk factors across its business operations and environment. The COVID-19 pandemic and related mitigation measures have adversely affected parts of the business, including reduced demand for Mobility offerings and creating ongoing uncertainties. The company's operational and financial performance is influenced by competitive pressure in the mobility, delivery, and logistics industries, characterized by well-established alternatives, low barriers to entry, and low switching costs. Driver classification risks exist if Drivers are deemed employees, workers, or quasi-employees rather than independent contractors, exposing the company to legal actions and financial liabilities globally. Competition challenges require the company to sometimes lower fares, offer incentives, and promotions, which impacts profitability. There are significant operating losses historically with substantial future operating expense increases anticipated, and the ability to achieve or maintain profitability is uncertain. Network value depends on maintaining critical mass among Drivers, consumers, merchants, shippers, and carriers, and failures to do so diminish platform attractiveness. Brand and reputation maintenance is critical, with exposure to negative publicity, media coverage, and risks from associated companies' brands or licensed brands in joint ventures.\\n\\nOperational risks include historical workplace culture and compliance challenges, management complexity due to rapid growth, technological infrastructure issues potentially causing disruptions or poor user experience, and security or data privacy breaches that could impact revenue and reputation. Platform users may engage in or be subjected to criminal, violent, or dangerous activity leading to safety incidents and legal actions. New offerings and technologies investments are inherently risky without guaranteed benefits. Economic conditions, inflation, and increased costs (fuel, food, labor, energy) may negatively impact results. Regulatory risks are extensive and global, involving payment and financial services compliance, licensing, anti-money laundering laws, data privacy (GDPR, CCPA, LGPD), and labor laws. Legal and regulatory investigations and inquiries, including antitrust, FCPA, labor classification, data protection, and intellectual property matters, pose risks of fines, penalties, operational changes, and increased costs.\\n\\nGeopolitical and jurisdictional risks include operating limitations or bans in some locations, currency exchange risk, and complex evolving regulations with the potential for fines and loss of licenses or permits. Insurance risks include potential inadequacy of reserves, liability exposure from accidents or impersonation, and insurer insolvency. Driver qualification requirements and background checks may increase costs or fail to expose all relevant information, with associated insurance cost risks and potential for courtroom or regulatory challenges to pricing models.\\n\\nFinancial risks comprise significant accumulated deficits, requirement for additional capital with uncertain availability, debt obligations, tax exposure including uncertain positions and observed changes in tax laws, and volatility in common stock price with no expected cash dividends. Accounting judgments and estimates involve critical assumptions affecting reported financial metrics related to goodwill, revenue recognition, incentive accruals, and stock-based compensation. Cybersecurity risks include exposures to malware, ransomware, phishing, and other cyberattacks. Climate change presents physical and transitional risks that may impact operations and costs, and failure to meet climate commitments may have operational and reputational consequences.\\n\\nOther risks include potential liability under anti-corruption and anti-terrorism laws, adverse effects from defaults under debt agreements, limitations in takeover actions due to corporate governance provisions, and the impact of non-GAAP financial measure limitations. Overall, these diverse and interconnected risk factors contribute to significant uncertainty regarding the company's future business prospects, operating results, and financial condition.\"]\n",
"\n",
"📄 Document 2: tmpppz9ub_m.md\n",
" 📊 Classification: technical_specification (confidence: 1.00)\n",
" 📝 Markdown length: 90,971 characters\n",
" 📋 Markdown sample: \n",
"\n",
"LM317\n",
"SLVS044Z SEPTEMBER 1997 REVISED APRIL 2025\n",
"\n",
"# LM317 3-Pin Adjustable Regulator\n",
"\n",
"## 1 Fea...\n",
" 🎯 Extracted fields: 8 fields\n",
" • component_name: LM317\n",
" • manufacturer: Texas Instruments\n",
" • part_number: LM317\n",
" • description: The LM317 is an adjustable three-pin, positive-voltage regulator capable of supplying up to 1.5A over an output voltage range of 1.25V to 37V. It features line and load regulation, internal current limiting, thermal overload protection, and safe operating area compensation.\n",
" • operating_voltage: {'min_voltage': 1.25, 'max_voltage': 37.0, 'unit': 'V'}\n",
" • maximum_current: 1.5\n",
" • key_features: ['Adjustable output voltage: 1.25V to 37V', 'Output current up to 1.5A', 'Line regulation: 0.01%/V (typical)', 'Load regulation: 0.1% (typical)', 'Internal short-circuit current limiting', 'Thermal overload protection', 'Output safe-area compensation', 'High power supply rejection ratio (PSRR): 80dB at 120Hz (new chip)', 'Available in SOT-223, TO-263, and TO-220 packages']\n",
" • applications: ['Multifunction printers', 'AC drive power stage modules', 'Electricity meters', 'Servo drive control modules', 'Merchant network and server power supply units']\n",
"\n",
"✨ Workflow completed successfully!\n",
"\n",
"📚 Key Learnings:\n",
" • Parse: Converted documents to clean markdown format\n",
" • Classify: Automatically categorized document types\n",
" • Extract: Used SourceText with markdown for structured data extraction\n",
" • The markdown content provides much better context for extraction than raw PDFs\n"
]
}
],
"source": [
"print(\"📈 COMPLETE WORKFLOW RESULTS SUMMARY\")\n",
"print(\"=\" * 70)\n",
"\n",
"for i, result in enumerate(results, 1):\n",
" print(f\"\\n📄 Document {i}: {os.path.basename(result['file_path'])}\")\n",
" print(\n",
" f\" 📊 Classification: {result['classification']} (confidence: {result['confidence']:.2f})\"\n",
" )\n",
" print(f\" 📝 Markdown length: {result['markdown_length']:,} characters\")\n",
" print(f\" 📋 Markdown sample: {result['markdown_sample'][:100]}...\")\n",
" print(f\" 🎯 Extracted fields: {len(result['extracted_data'])} fields\")\n",
"\n",
" # Print all keyvalue pairs\n",
" extracted = result[\"extracted_data\"]\n",
" for key, value in extracted.items():\n",
" print(f\" • {key}: {value}\")\n",
"\n",
"print(\"\\n✨ Workflow completed successfully!\")\n",
"print(\"\\n📚 Key Learnings:\")\n",
"print(\" • Parse: Converted documents to clean markdown format\")\n",
"print(\" • Classify: Automatically categorized document types\")\n",
"print(\" • Extract: Used SourceText with markdown for structured data extraction\")\n",
"print(\n",
" \" • The markdown content provides much better context for extraction than raw PDFs\"\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Conclusion\n",
"\n",
"This notebook demonstrated the complete **Parse → Classify → Extract** workflow using LlamaCloud services:\n",
"\n",
"### Key Components:\n",
"\n",
"1. **LlamaParse** (`llama_cloud_services.parse.base.LlamaParse`):\n",
" - Converts documents to clean, structured markdown\n",
" - Preserves document structure and formatting\n",
" - Handles various file types (PDF, DOCX, etc.)\n",
"\n",
"2. **ClassifyClient** (`llama_cloud_services.beta.classifier.client.ClassifyClient`):\n",
" - Automatically categorizes documents based on content\n",
" - Uses customizable rules for classification\n",
" - Provides confidence scores for classifications\n",
"\n",
"3. **LlamaExtract with SourceText** (`llama_cloud_services.extract.extract.LlamaExtract`, `SourceText`):\n",
" - Extracts structured data using custom Pydantic schemas\n",
" - **SourceText** allows using markdown content as input instead of raw files\n",
" - Provides much better extraction accuracy when using processed markdown\n",
"\n",
"### Workflow Benefits:\n",
"\n",
"- **Better Accuracy**: Using markdown from parsing provides cleaner, more structured input for extraction\n",
"- **Automatic Routing**: Classification allows different processing logic for different document types\n",
"- **Structured Output**: Custom schemas ensure consistent, structured data extraction\n",
"- **Flexible Input**: SourceText supports text content, file paths, and bytes\n",
"\n",
"### Key Insights:\n",
"\n",
"1. **SourceText is the bridge**: It allows you to pass the clean markdown content from parsing directly to extraction\n",
"2. **Markdown improves extraction**: Pre-processed markdown provides much better context than raw PDFs\n",
"3. **Classification enables smart routing**: Different document types can use different extraction schemas\n",
"4. **End-to-end automation**: The entire workflow can be automated for production use\n",
"\n",
"This approach is ideal for production document processing pipelines where you need to:\n",
"- Process various document types automatically\n",
"- Extract structured data consistently\n",
"- Maintain high accuracy and reliability\n",
"- Handle documents at scale\n",
"\n",
"The combination of these three services provides a powerful, flexible document processing pipeline that can handle complex, real-world document processing requirements."
]
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
+8 -1
View File
@@ -5,9 +5,16 @@
"private": true,
"keywords": [],
"author": "",
"scripts": {
"pre-commit-version": "pnpm changeset",
"version": "./scripts/changeset-version.py version",
"publish": "./scripts/changeset-version.py publish"
},
"devDependencies": {
"prettier": "^3.6.2",
"lint-staged": "^15.4.2"
"lint-staged": "^15.4.2",
"@changesets/cli": "^2.29.5",
"changesets": "^1.0.2"
},
"lint-staged": {
"ts/llama_cloud_services/src/**/*.{ts,tsx,js,jsx}": [
+1 -1
View File
@@ -147,7 +147,7 @@ documents = SimpleDirectoryReader(
).load_data()
```
Full documentation for `SimpleDirectoryReader` can be found on the [LlamaIndex Documentation](https://docs.llamaindex.ai/en/stable/module_guides/loading/simpledirectoryreader.html).
Full documentation for `SimpleDirectoryReader` can be found on the [LlamaIndex Documentation](https://developers.llamaindex.ai/python/framework/module_guides/loading/simpledirectoryreader/).
## Examples
+575 -10
View File
File diff suppressed because it is too large Load Diff
+2 -1
View File
@@ -1,2 +1,3 @@
packages:
- "ts/**"
- "ts/*"
- "py"
+7
View File
@@ -0,0 +1,7 @@
{
"name": "llama-cloud-services-py",
"version": "0.6.55",
"private": "true",
"license": "MIT",
"scripts": {}
}
+244
View File
@@ -0,0 +1,244 @@
#!/usr/bin/env -S uv run --script
# /// script
# dependencies = ["click", "tomlkit", "packaging"]
# ///
"""
This is a script called by the changeset bot. Normally changeset can do the following things, but this is a mixed ts and python repo, so we need to do some extra things.
There's 2 things this does:
- Versioning: Makes changes that may be committed with the newest version.
- Releasing/Tagging: After versions are changed, we check each package to see if its released, and if not, we release it and tag it.
"""
import json
import os
import subprocess
import sys
from pathlib import Path
from typing import List
import urllib.request
import urllib.error
import click
import tomlkit
from packaging.version import Version
def _run_command(
cmd: List[str], check: bool = True, capture: bool = True, cwd: Path | None = None
) -> subprocess.CompletedProcess:
"""Run a command and return the result."""
return subprocess.run(
cmd, check=check, capture_output=capture, text=True, cwd=cwd or Path.cwd()
)
def update_python_versions(version: str) -> None:
"""llama-cloud-services and llama-parse share a version. llama-parse is just a silly sidecar that proxies to llama-cloud-services
for compatibility.
This function updates the version in both pyproject.toml files.
"""
# Update main pyproject.toml
main_path = Path("py/pyproject.toml")
main_content = main_path.read_text()
main_doc = tomlkit.parse(main_content)
if main_doc["project"]["version"] != version:
click.echo(f"Updating llama-cloud-services version to {version}")
main_doc["project"]["version"] = version
main_path.write_text(tomlkit.dumps(main_doc))
# Update llama_parse/pyproject.toml
parse_path = Path("py/llama_parse/pyproject.toml")
parse_content = parse_path.read_text()
parse_doc = tomlkit.parse(parse_content)
if parse_doc["project"]["version"] != version:
click.echo(f"Updating llama-parse version to {version}")
parse_doc["project"]["version"] = version
parse_path.write_text(tomlkit.dumps(parse_doc))
# Update the dependency reference
dependencies = parse_doc["project"]["dependencies"]
for i, dep in enumerate(dependencies):
if isinstance(dep, str) and dep.startswith("llama-cloud-services"):
dependencies[i] = f"llama-cloud-services>={version}"
break
parse_path.write_text(tomlkit.dumps(parse_doc))
click.echo(f"Updated Python packages to version {version}")
def lock_python_dependencies() -> None:
"""Lock Python dependencies."""
try:
_run_command(["uv", "lock"], capture=False)
click.echo("Locked Python dependencies")
except subprocess.CalledProcessError as e:
click.echo(f"Warning: Failed to lock Python dependencies: {e}", err=True)
@click.group()
def cli() -> None:
"""Changeset-based version management for llama-cloud-services."""
pass
@cli.command()
def version() -> None:
"""Apply changeset versions, and propagate them to Python packages."""
# First, run changeset version to update all package.json files (including py/package.json)
_run_command(["npx", "@changesets/cli", "version"], capture=False, check=True)
# Get the updated Python package version from py/package.json (updated by changesets)
py_package_path = Path("py/package.json")
if not py_package_path.exists():
click.echo("Python package.json not found", err=True)
sys.exit(1)
with open(py_package_path) as f:
py_package = json.load(f)
new_version = py_package["version"]
# Update Python pyproject.toml files based on the package.json version
update_python_versions(new_version)
click.echo(f"Successfully propagated version {new_version} to all Python packages")
@cli.command()
@click.option("--tag", is_flag=True, help="Tag the packages after publishing")
@click.option("--dry-run", is_flag=True, help="Dry run the publish")
def publish(tag: bool, dry_run: bool) -> None:
"""Publish all packages."""
# move to the root
os.chdir(Path(__file__).parent.parent)
if not os.getenv("NPM_TOKEN"):
click.echo("NPM_TOKEN is not set, skipping publish", err=True)
raise click.Abort("No token set")
if not os.getenv("UV_PUBLISH_TOKEN"):
click.echo("UV_PUBLISH_TOKEN is not set, skipping publish", err=True)
raise click.Abort("No token set")
if not os.getenv("LLAMA_PARSE_PYPI_TOKEN"):
click.echo("LLAMA_PARSE_PYPI_TOKEN is not set, skipping publish", err=True)
raise click.Abort("No token set")
# not general script. Just checks each of the 2 packages to see if they need to be published.
maybe_publish_ts_package(dry_run)
maybe_publish_py_packages(dry_run)
if tag:
if dry_run:
click.echo("Dry run, skipping tag. Would run:")
click.echo(" npx @changesets/cli tag")
click.echo(" git push --tags")
return
else:
_run_command(["npx", "@changesets/cli", "tag"], check=True, capture=True)
_run_command(["git", "push", "--tags"], check=True, capture=True)
def maybe_publish_ts_package(dry_run: bool) -> None:
"""Publish the ts package if it needs to be published."""
target_dir = Path("ts/llama_cloud_services")
ts_path_package = target_dir / "package.json"
package_json = json.loads(ts_path_package.read_text())
version = package_json["version"]
# Check if this version is already published on npm
result = _run_command(
["npm", "view", "llama-cloud-services", "versions", "--json"],
check=True,
capture=True,
cwd=target_dir,
)
published_versions = json.loads(result.stdout)
if version in published_versions:
click.echo(
f"npm package llama-cloud-services@{version} already published, skipping"
)
return
click.echo(f"Publishing llama-cloud-services@{version}")
# defer to the package.json publish script
if dry_run:
click.echo("Dry run, skipping publish. Would run:")
click.echo(" pnpm run publish")
return
else:
output = _run_command(
["pnpm", "runpublish"], check=True, capture=True, cwd=target_dir
)
click.echo(output.stdout)
def maybe_publish_py_packages(dry_run: bool) -> None:
"""Publish the py packages if they need to be published."""
for pyproject in list(Path("py").glob("*/pyproject.toml")) + [
Path("py/pyproject.toml")
]:
name, version = current_version(pyproject)
if is_published(name, version):
click.echo(f"PyPI package {name}@{version} already published, skipping")
continue
click.echo(f"Publishing {name}@{version}")
# Use different tokens for different packages
env = os.environ.copy()
if name == "llama-parse":
# llama-parse uses its own token
env["UV_PUBLISH_TOKEN"] = os.environ["LLAMA_PARSE_PYPI_TOKEN"]
else:
# llama-cloud-services uses the main PyPI token
env["UV_PUBLISH_TOKEN"] = os.environ["UV_PUBLISH_TOKEN"]
if dry_run:
token = env["UV_PUBLISH_TOKEN"]
summary = (token[:3] + "***") if len(token) <= 6 else token[:6] + "****"
click.echo(
f"Dry run, skipping publish. Would run with publish token {summary}:"
)
click.echo(" uv publish --dry-run")
return
else:
result = subprocess.run(
["uv", "publish"],
check=True,
capture_output=True,
text=True,
cwd=pyproject.parent,
env=env,
)
click.echo(result.stdout)
def current_version(pyproject: Path) -> tuple[str, str]:
"""Return (package_name, version_str) taken from the given pyproject.toml."""
doc = tomlkit.parse(pyproject.read_text())
name = doc["project"]["name"]
version = str(Version(doc["project"]["version"])) # normalise
return name, version
def is_published(
name: str, version: str, index_url: str = "https://pypi.org/pypi"
) -> bool:
"""
True → `<name>==<version>` exists on the given index
False → package missing *or* version missing
"""
url = f"{index_url.rstrip('/')}/{name}/json"
try:
data = json.load(urllib.request.urlopen(url))
except urllib.error.HTTPError as e: # 404 → package not published at all
if e.code == 404:
return False
raise # any other error should surface
return version in data["releases"] # keys are version strings
if __name__ == "__main__":
cli()
-226
View File
@@ -1,226 +0,0 @@
#!/usr/bin/env -S uv run --script
# /// script
# dependencies = ["click", "tomlkit"]
# ///
import click
import subprocess
import sys
import tomlkit
from pathlib import Path
import json
def get_current_versions() -> tuple[str, str, str, str | None]:
"""Get current versions from both pyproject.toml files and TS package.json."""
# Read main pyproject.toml
main_content = Path("py/pyproject.toml").read_text()
main_doc = tomlkit.parse(main_content)
main_version = main_doc["project"]["version"]
# Read llama_parse/pyproject.toml
llama_parse_content = Path("py/llama_parse/pyproject.toml").read_text()
llama_parse_doc = tomlkit.parse(llama_parse_content)
llama_parse_version = llama_parse_doc["project"]["version"]
# Find llama-cloud-services dependency in the dependencies list
dependency_version = None
for dep in llama_parse_doc["project"]["dependencies"]:
if isinstance(dep, str) and dep.startswith("llama-cloud-services"):
dependency_version = (
dep.split("==")[1]
if "==" in dep
else dep.split(">=")[1]
if ">=" in dep
else None
)
break
# Read TypeScript package.json version via helper
ts_version: str = get_ts_version()
return (
str(main_version),
str(llama_parse_version),
str(dependency_version),
str(ts_version) if ts_version is not None else None,
)
def validate_versions(
main_version: str,
llama_parse_version: str,
dependency_version: str,
) -> list[str]:
"""Validate that versions are consistent and return warnings."""
warnings = []
if main_version != llama_parse_version:
warnings.append(
f"Version mismatch: main={main_version}, llama_parse={llama_parse_version}"
)
# Extract version from dependency string (e.g., ">=0.6.51" -> "0.6.51")
if dependency_version and dependency_version.startswith(">="):
dep_ver = dependency_version[2:]
if dep_ver != main_version:
warnings.append(
f"Dependency version mismatch: dependency={dep_ver}, main={main_version}"
)
return warnings
def set_version(version: str) -> None:
"""Set version across Python projects (no TS change)."""
# Update main pyproject.toml
main_content = Path("py/pyproject.toml").read_text()
main_doc = tomlkit.parse(main_content)
main_doc["project"]["version"] = version
Path("py/pyproject.toml").write_text(tomlkit.dumps(main_doc))
# Update llama_parse/pyproject.toml
llama_parse_content = Path("py/llama_parse/pyproject.toml").read_text()
llama_parse_doc = tomlkit.parse(llama_parse_content)
llama_parse_doc["project"]["version"] = version
for dep_index, dep in enumerate(llama_parse_doc["project"]["dependencies"]):
if isinstance(dep, str) and dep.startswith("llama-cloud-services"):
llama_parse_doc["project"]["dependencies"][
dep_index
] = f"llama-cloud-services>={version}"
break
Path("py/llama_parse/pyproject.toml").write_text(tomlkit.dumps(llama_parse_doc))
click.echo(f"Updated Python versions to {version}")
def get_ts_version() -> str:
"""Read TypeScript package.json version (if present)."""
ts_package_path = Path("ts/llama_cloud_services/package.json")
package_data = json.loads(ts_package_path.read_text())
data = package_data.get("version")
if data is None:
raise RuntimeError("TypeScript package.json version not found")
return data
def set_ts_version(version: str) -> None:
"""Set TypeScript package.json version only."""
ts_package_path = Path("ts/llama_cloud_services/package.json")
package_data = json.loads(ts_package_path.read_text())
package_data["version"] = version
ts_package_path.write_text(json.dumps(package_data, indent=2) + "\n")
click.echo(f"Updated TypeScript package.json version to {version}")
def get_current_branch() -> str:
"""Get the current git branch."""
result = subprocess.run(
["git", "branch", "--show-current"], capture_output=True, text=True, check=True
)
return result.stdout.strip()
def create_if_not_exists(version: str) -> str:
"""Create a git tag and push it."""
current_branch = get_current_branch()
if current_branch != "main":
click.echo(
f"Error: Not on main branch (currently on {current_branch})", err=True
)
sys.exit(1)
tag_name = f"v{version}" if version[0].isdigit() else version
if not tag_exists(tag_name):
# Create tag
subprocess.run(["git", "tag", tag_name], check=True)
click.echo(f"Created tag {tag_name}")
else:
click.echo(f"Tag {tag_name} already exists")
return tag_name
def tag_exists(tag_name: str) -> bool:
"""Check if a git tag exists."""
result = subprocess.run(
["git", "tag", "-l", tag_name], capture_output=True, text=True, check=True
)
return tag_name in result.stdout.strip()
def push_tag(tag_name: str) -> None:
"""Push a git tag."""
subprocess.run(["git", "push", "origin", tag_name], check=True)
click.echo(f"Pushed tag {tag_name}")
@click.group()
def cli() -> None:
"""Version management for llama-cloud-services."""
pass
@cli.command()
def get() -> None:
"""Get current versions and show validation warnings."""
(
main_version,
llama_parse_version,
dependency_version,
ts_version,
) = get_current_versions()
click.echo("Current versions:")
click.echo(f" llama-cloud-services: {main_version}")
click.echo(f" llama-parse: {llama_parse_version}")
click.echo(f" dependency reference: {dependency_version}")
click.echo(f" typescript package: {ts_version}")
warnings = validate_versions(main_version, llama_parse_version, dependency_version)
if warnings:
click.echo("\nValidation warnings:")
for warning in warnings:
click.echo(f" ⚠️ {warning}")
else:
click.echo("\n✅ All versions are consistent")
@cli.command()
@click.argument("version")
@click.option("--js", is_flag=True, help="Update TypeScript package.json only")
def set(version: str, js: bool) -> None:
"""Set version for Python, TypeScript, or both (default: Python only)."""
if js:
set_ts_version(version)
return
else:
set_version(version)
@cli.command()
@click.option(
"--version", help="Version to tag (uses current version if not specified)"
)
@click.option(
"--push",
is_flag=True,
help="Push the tag to the remote repository",
)
@click.option(
"--js",
is_flag=True,
help="tag TypeScript package.json only",
)
def tag(version: str | None = None, push: bool = False, js: bool = False) -> None:
"""Create and push a git tag for the current version."""
if not version:
main_version, _, _, js_version = get_current_versions()
version = f"llama-cloud-services@{js_version}" if js else main_version
tag_name = create_if_not_exists(version)
if push:
push_tag(tag_name)
if __name__ == "__main__":
cli()
+2 -1
View File
@@ -13,7 +13,8 @@
"test": "vitest run --testTimeout=60000",
"test:watch": "vitest --watch",
"test:ui": "vitest --ui",
"test:coverage": "vitest --coverage"
"test:coverage": "vitest --coverage",
"release": "pnpm run build && pnpm publish"
},
"files": [
"openapi.json",