Compare commits

..

1 Commits

Author SHA1 Message Date
Pierre-Loic Doulcet 963bd2dd2f Add support for html_remove_navigation_elements. 2024-12-06 11:03:16 +01:00
376 changed files with 12262 additions and 179346 deletions
-11
View File
@@ -1,11 +0,0 @@
# Please see the documentation for all configuration options:
# https://docs.github.com/github/administering-a-repository/configuration-options-for-dependency-updates
# and
# https://docs.github.com/code-security/dependabot/dependabot-version-updates/configuration-options-for-the-dependabot.yml-file
version: 2
updates:
- package-ecosystem: "github-actions"
directory: "/"
schedule:
interval: "weekly"
+48
View File
@@ -0,0 +1,48 @@
name: Build Package
# Build package on its own without additional pip install
on:
push:
branches:
- main
pull_request:
env:
POETRY_VERSION: "1.6.1"
jobs:
build:
runs-on: ${{ matrix.os }}
strategy:
# You can use PyPy versions in python-version.
# For example, pypy-2.7 and pypy-3.8
matrix:
os: [ubuntu-latest, windows-latest]
python-version: ["3.9"]
steps:
- uses: actions/checkout@v3
- name: Set up python ${{ matrix.python-version }}
uses: actions/setup-python@v4
with:
python-version: ${{ matrix.python-version }}
- name: Install Poetry
uses: snok/install-poetry@v1
with:
version: ${{ env.POETRY_VERSION }}
- name: Install deps
shell: bash
run: poetry install
- name: Ensure lock works
shell: bash
run: poetry lock
- name: Build
shell: bash
run: poetry build
- name: Test installing built package
shell: bash
run: python -m pip install .
- name: Test import
shell: bash
working-directory: ${{ vars.RUNNER_TEMP }}
run: python -c "import llama_parse"
-53
View File
@@ -1,53 +0,0 @@
name: Build Package - Python
# Build package on its own without additional pip install
on:
push:
branches:
- main
paths:
- "py/**"
pull_request:
paths:
- "py/**"
env:
UV_VERSION: "0.7.20"
jobs:
build:
runs-on: ${{ matrix.os }}
strategy:
# You can use PyPy versions in python-version.
# For example, pypy-2.7 and pypy-3.8
matrix:
os: [ubuntu-latest, windows-latest]
python-version: ["3.9"]
steps:
- uses: actions/checkout@v4
- name: Install uv
uses: astral-sh/setup-uv@v6
with:
version: ${{ env.UV_VERSION }}
- name: Set up Python
run: uv python install
- name: Display Python version
run: python --version
- name: Build
working-directory: py
run: uv build
- name: Test installing built package
shell: bash
working-directory: py
run: |
uv venv
uv pip install dist/*.whl
- name: Test import
working-directory: py
run: uv run -- python -c "import llama_cloud_services"
-36
View File
@@ -1,36 +0,0 @@
name: Build Package - TypeScript
on:
push:
branches:
- main
paths:
- "ts/**"
pull_request:
paths:
- "ts/**"
jobs:
pre_release:
name: Pre Release
runs-on: ubuntu-latest
steps:
- name: Checkout Repo
uses: actions/checkout@v4
- uses: pnpm/action-setup@v4
with:
version: 10
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version-file: "ts/llama_cloud_services/.nvmrc"
- name: Install dependencies
working-directory: ts/llama_cloud_services/
run: pnpm install --no-frozen-lockfile
- name: Build
working-directory: ts/llama_cloud_services/
run: pnpm run build
+48 -8
View File
@@ -1,3 +1,14 @@
# For most projects, this workflow file will not need changing; you simply need
# to commit it to your repository.
#
# You may wish to alter this file to override the set of languages analyzed,
# or to provide custom queries or build logic.
#
# ******** NOTE ********
# We have attempted to detect the languages in your repository. Please check
# the `language` matrix defined below to confirm you have the correct set of
# supported CodeQL languages.
#
name: "CodeQL"
on:
@@ -17,25 +28,54 @@ jobs:
# - https://gh.io/supported-runners-and-hardware-resources
# - https://gh.io/using-larger-runners
# Consider using larger runners for possible analysis time improvements.
runs-on: "ubuntu-latest"
timeout-minutes: 360
runs-on: ${{ (matrix.language == 'swift' && 'macos-latest') || 'ubuntu-latest' }}
timeout-minutes: ${{ (matrix.language == 'swift' && 120) || 360 }}
permissions:
actions: read
contents: read
security-events: write
strategy:
fail-fast: false
matrix:
language: ["python"]
# CodeQL supports [ 'cpp', 'csharp', 'go', 'java', 'javascript', 'python', 'ruby', 'swift' ]
# Use only 'java' to analyze code written in Java, Kotlin or both
# Use only 'javascript' to analyze code written in JavaScript, TypeScript or both
# Learn more about CodeQL language support at https://aka.ms/codeql-docs/language-support
steps:
- name: Checkout repository
uses: actions/checkout@v4
uses: actions/checkout@v3
# Initializes the CodeQL tools for scanning.
- name: Initialize CodeQL
uses: github/codeql-action/init@v3
uses: github/codeql-action/init@v2
with:
languages: python
dependency-caching: true
languages: ${{ matrix.language }}
# If you wish to specify custom queries, you can do so here or in a config file.
# By default, queries listed here will override any specified in a config file.
# Prefix the list here with "+" to use these queries and those in the config file.
# For more details on CodeQL's query packs, refer to: https://docs.github.com/en/code-security/code-scanning/automatically-scanning-your-code-for-vulnerabilities-and-errors/configuring-code-scanning#using-queries-in-ql-packs
# queries: security-extended,security-and-quality
# Autobuild attempts to build any compiled languages (C/C++, C#, Go, Java, or Swift).
# If this step fails, then you should remove it and run the build manually (see below)
- name: Autobuild
uses: github/codeql-action/autobuild@v2
# ️ Command-line programs to run using the OS shell.
# 📚 See https://docs.github.com/en/actions/using-workflows/workflow-syntax-for-github-actions#jobsjob_idstepsrun
# If the Autobuild fails above, remove it and uncomment the following three lines.
# modify them (or add more) to build your code if your project, please refer to the EXAMPLE below for guidance.
# - run: |
# echo "Run, Build Application using script"
# ./location_of_script_within_repo/buildscript.sh
- name: Perform CodeQL Analysis
uses: github/codeql-action/analyze@v3
uses: github/codeql-action/analyze@v2
with:
category: "/language:python"
category: "/language:${{matrix.language}}"
+37
View File
@@ -0,0 +1,37 @@
name: Linting
on:
push:
branches:
- main
pull_request:
env:
POETRY_VERSION: "1.6.1"
jobs:
build:
runs-on: ubuntu-latest
strategy:
# You can use PyPy versions in python-version.
# For example, pypy-2.7 and pypy-3.8
matrix:
python-version: ["3.9"]
steps:
- uses: actions/checkout@v3
with:
fetch-depth: ${{ github.event_name == 'pull_request' && 2 || 0 }}
- name: Set up python ${{ matrix.python-version }}
uses: actions/setup-python@v4
with:
python-version: ${{ matrix.python-version }}
- name: Install Poetry
uses: snok/install-poetry@v1
with:
version: ${{ env.POETRY_VERSION }}
- name: Install pre-commit
shell: bash
run: poetry run pip install pre-commit
- name: Run linter
shell: bash
run: poetry run make lint
-35
View File
@@ -1,35 +0,0 @@
name: Lint - Python
on:
push:
branches:
- main
pull_request:
env:
UV_VERSION: "0.7.20"
jobs:
build:
runs-on: ubuntu-latest
strategy:
# You can use PyPy versions in python-version.
# For example, pypy-2.7 and pypy-3.8
matrix:
python-version: ["3.9"]
steps:
- uses: actions/checkout@v4
with:
fetch-depth: ${{ github.event_name == 'pull_request' && 2 || 0 }}
- name: Install uv
uses: astral-sh/setup-uv@v6
with:
version: ${{ env.UV_VERSION }}
- name: Set up Python
run: uv python install ${{ matrix.python-version }}
- name: Run linter
shell: bash
working-directory: py
run: uv run -- pre-commit run -a
-38
View File
@@ -1,38 +0,0 @@
name: Lint - TypeScript
on:
push:
branches:
- main
paths:
- "ts/**"
pull_request:
paths:
- "ts/**"
env:
TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
TURBO_TEAM: ${{ vars.TURBO_TEAM }}
TURBO_REMOTE_ONLY: true
jobs:
lint:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: pnpm/action-setup@v4
with:
version: 10
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version-file: "ts/llama_cloud_services/.nvmrc"
- name: Install dependencies
working-directory: ts/llama_cloud_services/
run: pnpm install --no-frozen-lockfile
- name: Run lint
working-directory: ts/llama_cloud_services/
run: pnpm run lint
- name: Run Prettier
working-directory: ts/llama_cloud_services/
run: pnpm run format
+64
View File
@@ -0,0 +1,64 @@
name: Publish llama-parse to PyPI / GitHub
on:
push:
tags:
- "v*"
workflow_dispatch:
env:
POETRY_VERSION: "1.6.1"
PYTHON_VERSION: "3.9"
jobs:
build-n-publish:
name: Build and publish to PyPI
if: github.repository == 'run-llama/llama_parse'
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v3
- name: Set up python ${{ env.PYTHON_VERSION }}
uses: actions/setup-python@v4
with:
python-version: ${{ env.PYTHON_VERSION }}
- name: Install Poetry
uses: snok/install-poetry@v1
with:
version: ${{ env.POETRY_VERSION }}
- name: Install deps
shell: bash
run: pip install -e .
- name: Build and publish to pypi
uses: JRubics/poetry-publish@v1.17
with:
pypi_token: ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
ignore_dev_requirements: "yes"
- name: Create GitHub Release
id: create_release
uses: actions/create-release@v1
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} # This token is provided by Actions, you do not need to create your own token
with:
tag_name: ${{ github.ref }}
release_name: ${{ github.ref }}
draft: false
prerelease: false
- name: Get Asset name
run: |
export PKG=$(ls dist/ | grep tar)
set -- $PKG
echo "name=$1" >> $GITHUB_ENV
- name: Upload Release Asset (sdist) to GitHub
id: upload-release-asset
uses: actions/upload-release-asset@v1
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
with:
upload_url: ${{ steps.create_release.outputs.upload_url }}
asset_path: dist/${{ env.name }}
asset_name: ${{ env.name }}
asset_content_type: application/zip
-66
View File
@@ -1,66 +0,0 @@
name: Publish Release - Python
on:
push:
tags:
- "v*"
workflow_dispatch:
env:
UV_VERSION: "0.7.20"
jobs:
build-n-publish:
name: Build and publish to PyPI
if: github.repository == 'run-llama/llama_cloud_services'
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- name: Install uv
uses: astral-sh/setup-uv@v6
with:
version: ${{ env.UV_VERSION }}
- name: Set up Python
run: uv python install
- name: Display Python version
run: python --version
- name: Build
working-directory: py
run: uv build
- name: Test installing built package
shell: bash
working-directory: py
run: |
uv venv
uv pip install dist/*.whl
- name: Publish package
shell: bash
working-directory: py
run: uv publish --token ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
- name: Build and publish llama-parse
working-directory: py/llama_parse/
run: |
uv build
uv publish --token ${{ secrets.LLAMA_PARSE_PYPI_TOKEN }}
- name: Create GitHub Release
id: create_release
uses: actions/create-release@v1
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} # This token is provided by Actions, you do not need to create your own token
with:
tag_name: ${{ github.ref }}
release_name: ${{ github.ref }} - LlamaCloud Services PY
artifacts: "py/**/dist/*"
generateReleaseNotes: true
draft: false
prerelease: false
-51
View File
@@ -1,51 +0,0 @@
name: Publish Release - TypeScript
on:
push:
tags:
- "llama-cloud-services@*"
jobs:
build-and-publish:
runs-on: ubuntu-latest
steps:
- name: Checkout Repo
uses: actions/checkout@v4
- uses: pnpm/action-setup@v4
with:
version: 10
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version-file: "ts/llama_cloud_services/.nvmrc"
- name: Install dependencies
working-directory: ts/llama_cloud_services
run: pnpm install --no-frozen-lockfile
- name: Build tarball
run: |
pnpm pack
working-directory: ts/llama_cloud_services
- name: Setup npm authentication
run: echo "//registry.npmjs.org/:_authToken=${NPM_TOKEN}" > ~/.npmrc
env:
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
- name: Release
working-directory: ts/llama_cloud_services
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
NPM_TOKEN: ${{ secrets.NPM_TOKEN }}
run: pnpm publish --access public --no-git-checks
- name: Create release
uses: ncipollo/release-action@v1
with:
artifacts: "ts/llama_cloud_services/llama-cloud-services*.tgz"
name: Release ${{ github.ref }} - LlamaCloud Services TS
bodyFile: "ts/llama_cloud_services/CHANGELOG.md"
token: ${{ secrets.GITHUB_TOKEN }}
-43
View File
@@ -1,43 +0,0 @@
name: Test - Python
on:
push:
branches:
- main
paths:
- "py/**"
pull_request:
paths:
- "py/**"
env:
UV_VERSION: "0.7.20"
LLAMA_CLOUD_API_KEY: ${{ secrets.LLAMA_CLOUD_API_KEY }}
jobs:
test:
runs-on: ubuntu-latest
strategy:
# You can use PyPy versions in python-version.
# For example, pypy-2.7 and pypy-3.8
matrix:
python-version: ["3.9", "3.10", "3.11", "3.12"]
steps:
- uses: actions/checkout@v4
with:
fetch-depth: 0
- name: Install uv
uses: astral-sh/setup-uv@v6
with:
version: ${{ env.UV_VERSION }}
- name: Set up Python
run: uv python install ${{ matrix.python-version }} && uv python pin ${{ matrix.python-version }}
- name: Run Tests
working-directory: py
run: uv run -- pytest tests/**/test_*.py
- name: Remove virtual environment
working-directory: py
run: rm -rf .venv/
-37
View File
@@ -1,37 +0,0 @@
name: Lint - TypeScript
on:
push:
branches:
- main
paths:
- "ts/**"
pull_request:
paths:
- "ts/**"
env:
TURBO_TOKEN: ${{ secrets.TURBO_TOKEN }}
TURBO_TEAM: ${{ vars.TURBO_TEAM }}
TURBO_REMOTE_ONLY: true
LLAMA_CLOUD_API_KEY: ${{ secrets.LLAMA_CLOUD_API_KEY }}
jobs:
test:
name: Test - TypeScript
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v4
- uses: pnpm/action-setup@v4
with:
version: 10
- name: Setup Node.js
uses: actions/setup-node@v4
with:
node-version-file: "ts/llama_cloud_services/.nvmrc"
- name: Install dependencies
working-directory: ts/llama_cloud_services/
run: pnpm install --no-frozen-lockfile
- name: Run tests
working-directory: ts/llama_cloud_services/
run: pnpm test
+40
View File
@@ -0,0 +1,40 @@
name: Unit Testing
on:
push:
branches:
- main
pull_request:
env:
POETRY_VERSION: "1.6.1"
LLAMA_CLOUD_API_KEY: ${{ secrets.LLAMA_CLOUD_API_KEY }}
jobs:
test:
runs-on: ubuntu-latest
strategy:
# You can use PyPy versions in python-version.
# For example, pypy-2.7 and pypy-3.8
matrix:
python-version: ["3.8", "3.10", "3.11"]
steps:
- uses: actions/checkout@v3
with:
fetch-depth: 0
- name: Set up python ${{ matrix.python-version }}
uses: actions/setup-python@v4
with:
python-version: ${{ matrix.python-version }}
- name: Install Poetry
uses: snok/install-poetry@v1
with:
version: ${{ env.POETRY_VERSION }}
- name: Install deps
shell: bash
run: poetry install --with dev
- name: Run testing
env:
CI: true
shell: bash
run: poetry run pytest tests
-6
View File
@@ -3,9 +3,3 @@ __pycache__/
*.pyc
.DS_Store
.idea
.env*
.ipynb_checkpoints*
*_cache/
node_modules/
.turbo/
dist/
+6 -7
View File
@@ -21,19 +21,18 @@ repos:
hooks:
- id: ruff
args: [--fix, --exit-non-zero-on-fix]
exclude: ".*uv.lock"
exclude: ".*poetry.lock"
- repo: https://github.com/psf/black-pre-commit-mirror
rev: 23.10.1
hooks:
- id: black-jupyter
name: black-src
alias: black
exclude: ".*uv.lock"
exclude: ".*poetry.lock"
- repo: https://github.com/pre-commit/mirrors-mypy
rev: v1.0.1
hooks:
- id: mypy
exclude: ^py/tests/
additional_dependencies:
[
"types-requests",
@@ -47,7 +46,7 @@ repos:
[
--disallow-untyped-defs,
--ignore-missing-imports,
--python-version=3.10,
--python-version=3.8,
]
- repo: https://github.com/adamchainz/blacken-docs
rev: 1.16.0
@@ -63,13 +62,13 @@ repos:
rev: v3.0.3
hooks:
- id: prettier
exclude: ^(uv.lock|ts/llama_cloud_services/pnpm-lock.yaml)
exclude: poetry.lock
- repo: https://github.com/codespell-project/codespell
rev: v2.2.6
hooks:
- id: codespell
additional_dependencies: [tomli]
exclude: ^(uv.lock|docs|ts|examples)
exclude: ^(poetry.lock|examples)
args:
[
"--ignore-words-list",
@@ -84,6 +83,6 @@ repos:
rev: v0.23.1
hooks:
- id: toml-sort-fix
exclude: ".*uv.lock"
exclude: ".*poetry.lock"
exclude: .github/ISSUE_TEMPLATE
-33
View File
@@ -1,33 +0,0 @@
# Python
## Installation
This project uses uv. Create a virtual environment, and run `uv sync`
## Versioning (Maintainers only)
Before merging your changes, make sure to bump the versions.
Make a version bump to `pyproject.toml`. If the underlying dependency on the llamacloud platform OpenAPI
sdk needs bumping, make sure to bring that in as well. If updating dependencies, run `uv lock`.
The legacy `llama_parse` package re-exports some of `llama_cloud_services` in the old namespace. The
versions need to be kept consistent to sidecar it with `llama_cloud_services`. Bump it's version in `llama_parse/pyproject.toml`, and also bump it's dependency version of `llama-cloud-services` to match.
**Note**: Don't worry about updating the `llama_parse/poetry.lock` file when bumping versions. The GitHub action will automatically run `poetry lock` for the llama_parse package during the build process (though it doesn't commit the updated lockfile back to the repo).
You can also do this with `./scripts/version-bump.py set 0.x.x` if you have `uv` installed.
Once the change is merged, push a tag `git tag -a v0.x.x -m 0.x.x` and `git push origin 0.x.x`.
This tagging step can be done with `./scripts/version-bump tag`.
# Typescript
## Installation
...
## Versioning
...
View File
+131 -52
View File
@@ -1,81 +1,158 @@
[![PyPI - Downloads](https://img.shields.io/pypi/dm/llama-cloud-services)](https://pypi.org/project/llama-cloud-services/)
[![GitHub contributors](https://img.shields.io/github/contributors/run-llama/llama_cloud_services)](https://github.com/run-llama/llama_cloud_services/graphs/contributors)
# LlamaParse
[![PyPI - Downloads](https://img.shields.io/pypi/dm/llama-parse)](https://pypi.org/project/llama-parse/)
[![GitHub contributors](https://img.shields.io/github/contributors/run-llama/llama_parse)](https://github.com/run-llama/llama_parse/graphs/contributors)
[![Discord](https://img.shields.io/discord/1059199217496772688)](https://discord.gg/dGcwcsnxhU)
# Llama Cloud Services
LlamaParse is a **GenAI-native document parser** that can parse complex document data for any downstream LLM use case (RAG, agents).
This repository contains the code for hand-written SDKs and clients for interacting with LlamaCloud.
It is really good at the following:
This includes:
-**Broad file type support**: Parsing a variety of unstructured file types (.pdf, .pptx, .docx, .xlsx, .html) with text, tables, visual elements, weird layouts, and more.
-**Table recognition**: Parsing embedded tables accurately into text and semi-structured representations.
-**Multimodal parsing and chunking**: Extracting visual elements (images/diagrams) into structured formats and return image chunks using the latest multimodal models.
-**Custom parsing**: Input custom prompt instructions to customize the output the way you want it.
- [LlamaParse](./parse.md) - A GenAI-native document parser that can parse complex document data for any downstream LLM use case (Agents, RAG, data processing, etc.).
- [LlamaReport (beta/invite-only)](./report.md) - A prebuilt agentic report builder that can be used to build reports from a variety of data sources.
- [LlamaExtract](./extract.md) - A prebuilt agentic data extractor that can be used to transform data into a structured JSON representation.
- [LlamaCloud Index](./index.md) - A widely customizable and fully automated document ingestion pipeline that also serves retrieval purposes.
LlamaParse directly integrates with [LlamaIndex](https://github.com/run-llama/llama_index).
The free plan is up to 1000 pages a day. Paid plan is free 7k pages per week + 0.3c per additional page by default. There is a sandbox available to test the API [**https://cloud.llamaindex.ai/parse ↗**](https://cloud.llamaindex.ai/parse).
Read below for some quickstart information, or see the [full documentation](https://docs.cloud.llamaindex.ai/).
If you're a company interested in enterprise RAG solutions, and/or high volume/on-prem usage of LlamaParse, come [talk to us](https://www.llamaindex.ai/contact).
## Getting Started
Install the package:
First, login and get an api-key from [**https://cloud.llamaindex.ai/api-key ↗**](https://cloud.llamaindex.ai/api-key).
Then, make sure you have the latest LlamaIndex version installed.
**NOTE:** If you are upgrading from v0.9.X, we recommend following our [migration guide](https://pretty-sodium-5e0.notion.site/v0-10-0-Migration-Guide-6ede431dcb8841b09ea171e7f133bd77), as well as uninstalling your previous version first.
```
pip uninstall llama-index # run this if upgrading from v0.9.x or older
pip install -U llama-index --upgrade --no-cache-dir --force-reinstall
```
Lastly, install the package:
`pip install llama-parse`
Now you can parse your first PDF file using the command line interface. Use the command `llama-parse [file_paths]`. See the help text with `llama-parse --help`.
```bash
pip install llama-cloud-services
export LLAMA_CLOUD_API_KEY='llx-...'
# output as text
llama-parse my_file.pdf --result-type text --output-file output.txt
# output as markdown
llama-parse my_file.pdf --result-type markdown --output-file output.md
# output as raw json
llama-parse my_file.pdf --output-raw-json --output-file output.json
```
Then, get your API key from [LlamaCloud](https://cloud.llamaindex.ai/).
Then, you can use the services in your code:
You can also create simple scripts:
```python
from llama_cloud_services import (
LlamaParse,
LlamaReport,
LlamaExtract,
LlamaCloudIndex,
import nest_asyncio
nest_asyncio.apply()
from llama_parse import LlamaParse
parser = LlamaParse(
api_key="llx-...", # can also be set in your env as LLAMA_CLOUD_API_KEY
result_type="markdown", # "markdown" and "text" are available
num_workers=4, # if multiple files passed, split in `num_workers` API calls
verbose=True,
language="en", # Optionally you can define a language, default=en
)
parser = LlamaParse(api_key="YOUR_API_KEY")
report = LlamaReport(api_key="YOUR_API_KEY")
extract = LlamaExtract(api_key="YOUR_API_KEY")
index = LlamaCloudIndex(
"my_first_index", project_name="default", api_key="YOUR_API_KEY"
)
# sync
documents = parser.load_data("./my_file.pdf")
# sync batch
documents = parser.load_data(["./my_file1.pdf", "./my_file2.pdf"])
# async
documents = await parser.aload_data("./my_file.pdf")
# async batch
documents = await parser.aload_data(["./my_file1.pdf", "./my_file2.pdf"])
```
See the quickstart guides for each service for more information:
## Using with file object
- [LlamaParse](./parse.md)
- [LlamaReport (beta/invite-only)](./report.md)
- [LlamaExtract](./extract.md)
- [LlamaCloud Index](./index.md)
## Switch to EU SaaS 🇪🇺
If you are interested in using LlamaCloud services in the EU, you can adjust your base URL to `https://api.cloud.eu.llamaindex.ai`.
You can also create your API key in the EU region [here](https://cloud.eu.llamaindex.ai).
You can parse a file object directly:
```python
from llama_cloud_services import (
LlamaParse,
LlamaReport,
LlamaExtract,
EU_BASE_URL,
import nest_asyncio
nest_asyncio.apply()
from llama_parse import LlamaParse
parser = LlamaParse(
api_key="llx-...", # can also be set in your env as LLAMA_CLOUD_API_KEY
result_type="markdown", # "markdown" and "text" are available
num_workers=4, # if multiple files passed, split in `num_workers` API calls
verbose=True,
language="en", # Optionally you can define a language, default=en
)
parser = LlamaParse(api_key="YOUR_API_KEY", base_url=EU_BASE_URL)
report = LlamaReport(api_key="YOUR_API_KEY", base_url=EU_BASE_URL)
extract = LlamaExtract(api_key="YOUR_API_KEY", base_url=EU_BASE_URL)
index = LlamaCloudIndex(
"my_first_index",
project_name="default",
api_key="YOUR_API_KEY",
base_url=EU_BASE_URL,
)
file_name = "my_file1.pdf"
extra_info = {"file_name": file_name}
with open(f"./{file_name}", "rb") as f:
# must provide extra_info with file_name key with passing file object
documents = parser.load_data(f, extra_info=extra_info)
# you can also pass file bytes directly
with open(f"./{file_name}", "rb") as f:
file_bytes = f.read()
# must provide extra_info with file_name key with passing file bytes
documents = parser.load_data(file_bytes, extra_info=extra_info)
```
## Using with `SimpleDirectoryReader`
You can also integrate the parser as the default PDF loader in `SimpleDirectoryReader`:
```python
import nest_asyncio
nest_asyncio.apply()
from llama_parse import LlamaParse
from llama_index.core import SimpleDirectoryReader
parser = LlamaParse(
api_key="llx-...", # can also be set in your env as LLAMA_CLOUD_API_KEY
result_type="markdown", # "markdown" and "text" are available
verbose=True,
)
file_extractor = {".pdf": parser}
documents = SimpleDirectoryReader(
"./data", file_extractor=file_extractor
).load_data()
```
Full documentation for `SimpleDirectoryReader` can be found on the [LlamaIndex Documentation](https://docs.llamaindex.ai/en/stable/module_guides/loading/simpledirectoryreader.html).
## Examples
Several end-to-end indexing examples can be found in the examples folder
- [Getting Started](examples/demo_basic.ipynb)
- [Advanced RAG Example](examples/demo_advanced.ipynb)
- [Raw API Usage](examples/demo_api.ipynb)
## Documentation
You can see complete SDK and API documentation for each service on [our official docs](https://docs.cloud.llamaindex.ai/).
[https://docs.cloud.llamaindex.ai/](https://docs.cloud.llamaindex.ai/)
## Terms of Service
@@ -83,4 +160,6 @@ See the [Terms of Service Here](./TOS.pdf).
## Get in Touch (LlamaCloud)
You can get in touch with us by following our [contact link](https://www.llamaindex.ai/contact).
LlamaParse is part of LlamaCloud, our e2e enterprise RAG platform that provides out-of-the-box, production-ready connectors, indexing, and retrieval over your complex data sources. We offer SaaS and VPC options.
LlamaCloud is currently available via waitlist (join by [creating an account](https://cloud.llamaindex.ai/)). If you're interested in state-of-the-art quality and in centralizing your RAG efforts, come [get in touch with us](https://www.llamaindex.ai/contact).
-9
View File
@@ -1,9 +0,0 @@
# LlamaCloud Services Examples - Python
In this folder you will find several python notebooks with examples regarding:
- [LlamaParse](./parse/)
- [LlamaExtract](./extract/)
- [LlamaReport](./report/)
Follow the instructions of each notebook to get started!
File diff suppressed because it is too large Load Diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 3.3 MiB

File diff suppressed because one or more lines are too long
@@ -1,10 +0,0 @@
# Financial Modeling Assumptions
Discount Rate: 8%
Terminal Growth Rate: 2%
Tax Rate: 25%
Revenue Growth (Years 1-5): 10% per annum
Revenue Growth (Years 6-10): 5% per annum
Capital Expenditures as % of Revenue: 7%
Working Capital Assumption: 3% of Revenue
Depreciation Rate: 10% per annum
Cost of Capital Assumption: 8%
Binary file not shown.

Before

Width:  |  Height:  |  Size: 67 KiB

@@ -1 +0,0 @@
sec_form_4_dump.json
File diff suppressed because it is too large Load Diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 202 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 440 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 156 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 85 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 893 KiB

@@ -1,440 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Extract Data from Financial Reports - with Citations and Reasoning\n",
"\n",
"Given complex files like financial reports, contracts, invoices etc, Llama Extract allows you to make use of an LLM to extract the information relevant to you, in a structured format.\n",
"\n",
"In this example, we'll be using [LlamaExtract](https://docs.cloud.llamaindex.ai/llamaextract/getting_started?utm_campaign=extract&utm_medium=recipe) to extract structured data from an SEC filing (specifically, the filing by Nvidia for fiscal year 2025).\n",
"\n",
"On top of simple data extraction, we'll ask our extraction agent to provide citations and reasoning for each extracted field. This allows us to:\n",
"- Confirm the accuracy of the extracted field\n",
"- Understand the reasoning behind why the LLM extracted a given piece of information\n",
"- This last point allows us an opportunity to adjust the system prompt or field descriptions and improve on results where needed.\n",
"\n",
"\n",
"The example we go through below is also replicable within Llama Cloud as well, where you will also be able to pick between a number of pre-defined schemas, instead of building your own."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install llama-cloud-services"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Connect to Llama Cloud\n",
"\n",
"To get started, make sure you provide your [Llama Cloud](https://cloud.llamaindex.ai?utm_campaign=extract&utm_medium=recipe) API key."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Enter your Llama Cloud API Key: ··········\n"
]
}
],
"source": [
"import os\n",
"from getpass import getpass\n",
"\n",
"if \"LLAMA_CLOUD_API_KEY\" not in os.environ:\n",
" os.environ[\"LLAMA_CLOUD_API_KEY\"] = getpass(\"Enter your Llama Cloud API Key: \")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Extract Data with Llama Extract Agent"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"No project_id provided, fetching default project.\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaExtract\n",
"\n",
"# Optionally, provide your project id, if not, it will use the 'Default' project\n",
"llama_extract = LlamaExtract()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Provide Your Custom Schema\n",
"\n",
"When using LlamaExtract via the API, you provide your own schema that describes what you want extracted from files and data provided to your agent. Here, we are essentially building an SEC filings extraction agent."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pydantic import BaseModel, Field\n",
"from enum import Enum\n",
"\n",
"\n",
"class FilingType(str, Enum):\n",
" ten_k = \"10 K\"\n",
" ten_q = \"10-Q\"\n",
" ten_ka = \"10-K/A\"\n",
" ten_qa = \"10-Q/A\"\n",
"\n",
"\n",
"class FinancialReport(BaseModel):\n",
" company_name: str = Field(description=\"The name of the company\")\n",
" description: str = Field(\n",
" description=\"Short description of the filing and what it contains\"\n",
" )\n",
" filing_type: FilingType = Field(description=\"Type of SEC filing\")\n",
" filing_date: str = Field(description=\"Date when filing was submitted to SEC\")\n",
" fiscal_year: int = Field(description=\"Fiscal year\")\n",
" unit: str = Field(\n",
" description=\"Unit of financial figures (thousands, millions, etc.)\"\n",
" )\n",
" revenue: int = Field(description=\"Total revenue for period\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Set Up Citations and Reasoning\n",
"\n",
"Optionally, we can set the `ExtractConfig` to extract citations for each field the agent extracts. These cications will cite the specific pages and sections of the file from which a given field was extractedd.\n",
"\n",
"By setting `use_reasoning` to True, we als ask the agent to do an additional reasoning step, explaining why a given field was extracted."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud.types import ExtractConfig, ExtractMode\n",
"\n",
"config = ExtractConfig(\n",
" use_reasoning=True, cite_sources=True, extraction_mode=ExtractMode.MULTIMODAL\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"/usr/local/lib/python3.11/dist-packages/llama_cloud_services/extract/extract.py:127: ExperimentalWarning: `use_reasoning` is an experimental feature. Results will be available in the `extraction_metadata` field for the extraction run.\n",
" warnings.warn(\n",
"/usr/local/lib/python3.11/dist-packages/llama_cloud_services/extract/extract.py:133: ExperimentalWarning: `cite_sources` is an experimental feature. This may greatly increase the size of the response, and slow down the extraction. Results will be available in the `extraction_metadata` field for the extraction run.\n",
" warnings.warn(\n"
]
}
],
"source": [
"agent = llama_extract.create_agent(\n",
" name=\"filing-parser\", data_schema=FinancialReport, config=config\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Demo Time - Download a PDF and Extract Data with Citations"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"PDF downloaded successfully.\n"
]
}
],
"source": [
"import requests\n",
"\n",
"url = \"https://raw.githubusercontent.com/run-llama/llama_cloud_services/refs/heads/main/examples/extract/data/sec_filings/nvda_10k.pdf\"\n",
"\n",
"response = requests.get(url)\n",
"\n",
"if response.status_code == 200:\n",
" with open(\"/content/nvda_10k.pdf\", \"wb\") as f:\n",
" f.write(response.content)\n",
" print(\"PDF downloaded successfully.\")\n",
"else:\n",
" print(f\"Failed to download. Status code: {response.status_code}\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"Uploading files: 100%|██████████| 1/1 [00:00<00:00, 1.83it/s]\n",
"Creating extraction jobs: 100%|██████████| 1/1 [00:00<00:00, 4.38it/s]\n",
"Extracting files: 100%|██████████| 1/1 [02:03<00:00, 123.40s/it]\n"
]
}
],
"source": [
"filing_info = agent.extract(\"/content/nvda_10k.pdf\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'company_name': 'NVIDIA Corporation',\n",
" 'description': \"The filing provides a detailed overview of NVIDIA's business as a full-stack computing infrastructure company, discusses various technologies including digital avatars and autonomous vehicles, outlines numerous risk factors affecting operations such as supply chain issues and geopolitical tensions, and describes employee stock purchase plans and related compliance requirements.\",\n",
" 'filing_type': '10 K',\n",
" 'filing_date': 'February 26, 2025',\n",
" 'fiscal_year': 2025,\n",
" 'unit': 'millions',\n",
" 'revenue': 130497}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"filing_info.data"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Inspect Citations and Reasoning"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'field_metadata': {'company_name': {'reasoning': 'VERBATIM EXTRACTION',\n",
" 'citation': [{'page': 1, 'matching_text': 'NVIDIA CORPORATION'},\n",
" {'page': 2, 'matching_text': 'NVIDIA Corporation'},\n",
" {'page': 3,\n",
" 'matching_text': 'All references to \"NVIDIA,\" \"we,\" \"us,\" \"our,\" or the \"Company\" mean NVIDIA Corporation and its subsidiaries.'},\n",
" {'page': 35,\n",
" 'matching_text': 'Comparison of 5 Year Cumulative Total Return* Among NVIDIA Corporation'},\n",
" {'page': 49,\n",
" 'matching_text': 'To the Board of Directors and Shareholders of NVIDIA Corporation'},\n",
" {'page': 90, 'matching_text': 'NVIDIA Corporation'},\n",
" {'page': 119,\n",
" 'matching_text': '*\"Company\"* means NVIDIA Corporation, a Delaware corporation.'},\n",
" {'page': 126,\n",
" 'matching_text': 'Annual Report on Form 10-K of NVIDIA Corporation'}]},\n",
" 'filing_type': {'reasoning': \"VERBATIM EXTRACTION from multiple sources confirming the filing type as '10 K'.\",\n",
" 'citation': [{'page': 1, 'matching_text': 'FORM 10-K'},\n",
" {'page': 2, 'matching_text': 'Item 16. | Form 10-K Summary'},\n",
" {'page': 3,\n",
" 'matching_text': 'This Annual Report on Form 10-K contains forward-looking statements...'},\n",
" {'page': 13, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 15, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 32,\n",
" 'matching_text': 'Annual Report on Form 10-K, which information is hereby incorporated by reference.'},\n",
" {'page': 36, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 43,\n",
" 'matching_text': 'Annual Report on Form 10-K for additional information'},\n",
" {'page': 45, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 46, 'matching_text': 'this Annual Report on Form 10-K'},\n",
" {'page': 62, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 83,\n",
" 'matching_text': 'Restated Certificate of Incorporation | 10-K'},\n",
" {'page': 84, 'matching_text': 'Item 16. Form 10-K Summary'},\n",
" {'page': 126, 'matching_text': 'which appears in this Form 10-K'},\n",
" {'page': 127, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 128, 'matching_text': 'Annual Report on Form 10-K'},\n",
" {'page': 129, 'matching_text': \"The Company's Annual Report on Form 10-K\"},\n",
" {'page': 130,\n",
" 'matching_text': \"The Company's Annual Report on Form 10-K for the year ended January 26, 2025\"}]},\n",
" 'fiscal_year': {'reasoning': 'The fiscal year ended January 26, 2025, indicates the fiscal year is 2025. Additionally, multiple references throughout the text confirm the fiscal year 2025 in various contexts.',\n",
" 'citation': [{'page': 1,\n",
" 'matching_text': 'For the fiscal year ended January 26, 2025'},\n",
" {'page': 6,\n",
" 'matching_text': 'In fiscal year 2025, we launched the NVIDIA Blackwell architecture'},\n",
" {'page': 12, 'matching_text': 'fiscal year 2025'},\n",
" {'page': 17,\n",
" 'matching_text': 'our gross margins in the second quarter of fiscal year 2025 were negatively impacted'},\n",
" {'page': 20,\n",
" 'matching_text': 'we generated 53% of our revenue in fiscal year 2025 from sales outside the United States.'},\n",
" {'page': 23,\n",
" 'matching_text': 'For fiscal year 2025, an indirect customer which primarily purchases our products through system integrators...'},\n",
" {'page': 33,\n",
" 'matching_text': 'In fiscal year 2025, we repurchased 310 million shares of our common stock for $34.0 billion.'},\n",
" {'page': 37,\n",
" 'matching_text': 'Our Data Center revenue in China grew in fiscal year 2025.'},\n",
" {'page': 44,\n",
" 'matching_text': 'Cash provided by operating activities increased in fiscal year 2025 compared to fiscal year 2024'},\n",
" {'page': 57,\n",
" 'matching_text': 'Fiscal years 2025, 2024 and 2023 were all 52-week years.'},\n",
" {'page': 65,\n",
" 'matching_text': 'Beginning in the second quarter of fiscal year 2025'},\n",
" {'page': 69, 'matching_text': 'In the fourth quarter of fiscal year 2025'},\n",
" {'page': 78,\n",
" 'matching_text': 'Depreciation and amortization expense attributable to our Compute and Networking segment for fiscal years 2025'},\n",
" {'page': 129, 'matching_text': 'for the year ended January 26, 2025'}]},\n",
" 'description': {'reasoning': 'The extracted data combines multiple descriptions from the source text, ensuring no duplication while maintaining the order and context of the information. Each section of the filing is summarized to reflect the key points without losing the essence of the original text.',\n",
" 'citation': [{'page': 4,\n",
" 'matching_text': 'NVIDIA is now a full-stack computing infrastructure company with data-center-scale offerings that are reshaping industry.'},\n",
" {'page': 8,\n",
" 'matching_text': 'a suite of technologies that help developers bring digital avatars to life with generative Al...autonomous vehicles, or AV, and electric vehicles, or EV, is revolutionizing the transportation industry...Our worldwide sales and marketing strategy is key to achieving our objective of providing markets with our high-performance and efficient computing platforms and software.'},\n",
" {'page': 14, 'matching_text': 'Risk Factors Summary'},\n",
" {'page': 16,\n",
" 'matching_text': 'Risks Related to Demand, Supply, and Manufacturing\\n\\nLong manufacturing lead times and uncertain supply and component availability...'},\n",
" {'page': 18,\n",
" 'matching_text': 'cryptocurrency mining, on demand for our products. Volatility in the cryptocurrency market, including new compute technologies...'},\n",
" {'page': 21,\n",
" 'matching_text': 'supply-chain attacks or other business disruptions. We cannot guarantee that third parties and infrastructure in our supply chain...'},\n",
" {'page': 22,\n",
" 'matching_text': 'We are monitoring the impact of the geopolitical conflict in and around Israel on our operations... Climate change may have a long-term impact on our business.'},\n",
" {'page': 25,\n",
" 'matching_text': 'We are subject to complex laws, rules, regulations, and political and other actions, including restrictions on the export of our products, which may adversely impact our business.'},\n",
" {'page': 28,\n",
" 'matching_text': 'Our competitive position has been harmed by the existing export controls, and our competitive position and future results may be further harmed'},\n",
" {'page': 29,\n",
" 'matching_text': 'restrictions imposed by the Chinese government on the duration of gaming activities and access to games may adversely affect our Gaming revenue'},\n",
" {'page': 29,\n",
" 'matching_text': 'our business depends on our ability to receive consistent and reliable supply from our overseas partners, especially in Taiwan and South Korea'},\n",
" {'page': 29,\n",
" 'matching_text': 'Increased scrutiny from shareholders, regulators and others regarding our corporate sustainability practices could result in additional costs'},\n",
" {'page': 29,\n",
" 'matching_text': 'Concerns relating to the responsible use of new and evolving technologies, such as Al, in our products and services may result in reputational or financial harm'},\n",
" {'page': 31,\n",
" 'matching_text': 'Data protection laws around the world are quickly changing and may be interpreted and applied in an increasingly stringent fashion...'}]},\n",
" 'filing_date': {'reasoning': 'The filing date is consistently mentioned as February 26, 2025 across multiple entries, making it the most reliable date for the filing.',\n",
" 'citation': [{'page': 51, 'matching_text': 'February 26, 2025'},\n",
" {'page': 86, 'matching_text': 'on February 26, 2025.'},\n",
" {'page': 87, 'matching_text': 'February 26, 2025'},\n",
" {'page': 126, 'matching_text': 'our report dated February 26, 2025'},\n",
" {'page': 127, 'matching_text': 'Date: February 26, 2025'},\n",
" {'page': 128, 'matching_text': 'Date: February 26, 2025'},\n",
" {'page': 129, 'matching_text': 'Date: February 26, 2025'},\n",
" {'page': 130, 'matching_text': 'Date: February 26, 2025'}]},\n",
" 'unit': {'reasoning': \"The unit of financial figures is explicitly mentioned multiple times in the text as 'millions', including in table headers and notes. This is confirmed by various citations from pages 38, 42, 43, 52, 53, 54, 56, 65, 71, 72, 73, 75, 77, 79, 80, and 82.\",\n",
" 'citation': [{'page': 38,\n",
" 'matching_text': '($ in millions, except per share data)'},\n",
" {'page': 42, 'matching_text': '($ in millions)'},\n",
" {'page': 43, 'matching_text': '($ in millions)'},\n",
" {'page': 52, 'matching_text': '(In millions, except per share data)'},\n",
" {'page': 53,\n",
" 'matching_text': 'Consolidated Statements of Comprehensive Income (In millions)'},\n",
" {'page': 54,\n",
" 'matching_text': 'Consolidated Balance Sheets (In millions, except par value)'},\n",
" {'page': 55, 'matching_text': '(In millions, except per share data)'},\n",
" {'page': 56,\n",
" 'matching_text': 'Consolidated Statements of Cash Flows (In millions)'},\n",
" {'page': 65,\n",
" 'matching_text': 'Year Ended<br/>Jan 26, 2025<br/>(In millions, except per share data)'},\n",
" {'page': 71, 'matching_text': '(In millions) | (In millions)'},\n",
" {'page': 72, 'matching_text': '(In millions)'}]},\n",
" 'revenue': {'reasoning': 'The total revenue for fiscal year 2025 is extracted from multiple sources within the text, all confirming the same figure of $130,497 million. The revenue recognized for fiscal year 2025 is also noted as $4,607 million, which is a separate figure. However, the primary focus is on the total revenue figure, which is consistently cited.',\n",
" 'citation': [{'page': 38,\n",
" 'matching_text': 'Revenue for fiscal year 2025 was $130.5 billion'},\n",
" {'page': 41,\n",
" 'matching_text': 'Total | $ 130,497 | $ | 60,922'},\n",
" {'page': 52, 'matching_text': 'Revenue | $ 130,497'},\n",
" {'page': 78,\n",
" 'matching_text': 'Revenue | $ 116,193 | $ 14,304 | $ - | $ 130,497'},\n",
" {'page': 79, 'matching_text': 'Total revenue | $ 130,497'},\n",
" {'page': 80, 'matching_text': 'Total revenue | $ 130,497'}]}},\n",
" 'usage': {'num_pages_extracted': 130,\n",
" 'num_document_tokens': 105932,\n",
" 'num_output_tokens': 31306}}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"filing_info.extraction_metadata"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## What's Next?\n",
"\n",
"In this example, we built an Extraction Agent that is capable of citing it's sources from the document it's extracting data from, and reasoning about its reponse. To further customize and improve on the results, you can also try to customize the `system_prompt` in the `ExtractConfig`.\n",
"\n",
"#### Learn More\n",
"\n",
"- [LlamaExtract Documentation](https://docs.cloud.llamaindex.ai/llamaextract/getting_started)\n",
"- [Example Notebooks](https://github.com/run-llama/llama_cloud_services/tree/main/examples/extract)"
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "Python 3",
"name": "python3"
},
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 0
}
File diff suppressed because it is too large Load Diff
@@ -1,318 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "1f6bd03d-1b8b-45a0-bc2c-5a13f1a5d8d3",
"metadata": {},
"source": [
"# LM317 Voltage Regulator Datasheet Structured Extraction\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/extract/lm317_structured_extraction.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"This notebook demonstrates an agentic document workflow using LlamaExtract to process an LM317 voltage regulator datasheet. In this example, we define a structured extraction schema that converts key technical fields into standardized subfields. For instance, the output voltage is split into a minimum and maximum value with a defined unit, and we capture page citations for each extracted field.\n",
"\n",
"The target user is an electronics engineer at a component manufacturing company who needs to consolidate datasheet information into a standardized specification sheet for design and quality control.\n",
"\n",
"This approach reduces manual data entry, improves extraction accuracy and standardization, and provides traceability for each technical detail."
]
},
{
"cell_type": "markdown",
"id": "a3b8c8d5-ff3e-48ce-b0b8-29b6b1f517f8",
"metadata": {},
"source": [
"## Use Case Overview\n",
"\n",
"### Problem\n",
"Datasheets like that for the LM317 regulator are often distributed as PDFs containing multiple tables, charts, and complex textual descriptions. Engineers must manually extract technical details such as voltage ranges, dropout voltage, maximum current, input voltage range, and pin configurations. This process is error-prone and time-consuming.\n",
"\n",
"### Agent Workflow (Combination of Automation and Chat)\n",
"1. **Upload Datasheet:** The engineer uploads the LM317 datasheet PDF. \n",
"2. **Structured Extraction:** An automated agent processes the PDF and extracts key technical details into structured fields (e.g., output voltage as a range with separate min/max values).\n",
"3. **Interactive Verification:** The engineer can query the agent (via chat) for further details or clarification (e.g., \"Show me the detailed pin configuration extraction\") and review the cited pages.\n",
"\n",
"**Value Delivered:**\n",
"- Up to 70% reduction in manual data extraction time.\n",
"- Increased accuracy and standardization with structured fields."
]
},
{
"cell_type": "markdown",
"id": "a704e843-54be-4969-842b-713584cb3c35",
"metadata": {},
"source": [
"## Setup and Download Data\n",
"\n",
"Download the [LM317 Datasheet](https://www.ti.com/lit/ds/symlink/lm317.pdf) and setup LlamaExtract."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6e5b1f91-8785-44d4-a710-8be1b48b76de",
"metadata": {},
"outputs": [],
"source": [
"!mkdir -p data/lm317_structured_extraction\n",
"!wget https://www.ti.com/lit/ds/symlink/lm317.pdf -O data/lm317_structured_extraction/lm317.pdf"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f17b914a-00ed-4b63-8198-69fd7c4a7c62",
"metadata": {},
"outputs": [],
"source": [
"from dotenv import load_dotenv\n",
"from llama_cloud_services import LlamaExtract\n",
"from llama_cloud.core.api_error import ApiError\n",
"\n",
"# Load environment variables (ensure LLAMA_CLOUD_API_KEY is set in your .env file)\n",
"load_dotenv(override=True)\n",
"\n",
"# Initialize the LlamaExtract client\n",
"llama_extract = LlamaExtract(\n",
" project_id=\"<project_id>\",\n",
" organization_id=\"<organization_id>\",\n",
")"
]
},
{
"cell_type": "markdown",
"id": "ed9f6e9a-96c8-4ee1-8b45-0b6a4f7dbbf1",
"metadata": {},
"source": [
"## Defining a Structured Extraction Schema\n",
"\n",
"We now define a rich Pydantic schema to extract technical specifications from the LM317 datasheet. In this schema:\n",
"\n",
"- The **output_voltage** and **input_voltage** fields are structured as ranges with separate minimum and maximum values and a unit.\n",
"- The **pin_configuration** field is structured to include a pin count and a descriptive layout.\n",
"- Additional technical fields (e.g., dropout voltage, max current) are captured as numbers.\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "4f7e9b44-5e69-4b30-9864-cd98f1e2a7d4",
"metadata": {},
"outputs": [],
"source": [
"from pydantic import BaseModel, Field\n",
"from typing import List\n",
"\n",
"\n",
"class VoltageRange(BaseModel):\n",
" min_voltage: float = Field(..., description=\"Minimum voltage in volts\")\n",
" max_voltage: float = Field(..., description=\"Maximum voltage in volts\")\n",
" unit: str = Field(\"V\", description=\"Voltage unit\")\n",
"\n",
"\n",
"class PinConfiguration(BaseModel):\n",
" pin_count: int = Field(..., description=\"Number of pins\")\n",
" layout: str = Field(..., description=\"Detailed pin layout description\")\n",
"\n",
"\n",
"class LM317Spec(BaseModel):\n",
" component_name: str = Field(..., description=\"Name of the component\")\n",
" output_voltage: VoltageRange = Field(\n",
" ..., description=\"Output voltage range specification\"\n",
" )\n",
" dropout_voltage: float = Field(..., description=\"Dropout voltage in volts\")\n",
" max_current: float = Field(..., description=\"Maximum current rating in amperes\")\n",
" input_voltage: VoltageRange = Field(\n",
" ..., description=\"Input voltage range specification\"\n",
" )\n",
" pin_configuration: PinConfiguration = Field(\n",
" ..., description=\"Pin configuration details\"\n",
" )\n",
" features: List[str] = Field([], description=\"List of additional technical features\")\n",
"\n",
"\n",
"class LM317Schema(BaseModel):\n",
" specs: List[LM317Spec] = Field(\n",
" ..., description=\"List of extracted LM317 technical specifications\"\n",
" )"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e0508e38-35be-446c-afe7-129e39553281",
"metadata": {},
"outputs": [],
"source": [
"try:\n",
" existing_agent = llama_extract.get_agent(name=\"lm317-datasheet\")\n",
" if existing_agent:\n",
" llama_extract.delete_agent(existing_agent.id)\n",
"except ApiError as e:\n",
" if e.status_code == 404:\n",
" pass\n",
" else:\n",
" raise"
]
},
{
"cell_type": "markdown",
"id": "bb197dfd-dd37-459e-8953-cc1b12f25bdd",
"metadata": {},
"source": [
"Here we use our balanced extraction mode."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e3defc0a-c685-4fbd-bbb1-1270f1442e72",
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud import ExtractConfig\n",
"\n",
"extract_config = ExtractConfig(\n",
" extraction_mode=\"BALANCED\",\n",
")\n",
"\n",
"agent = llama_extract.create_agent(\n",
" name=\"lm317-datasheet\", data_schema=LM317Schema, config=extract_config\n",
")"
]
},
{
"cell_type": "markdown",
"id": "c0a0f9f9-2ef3-4a38-bd74-68d2c2e9e2d8",
"metadata": {},
"source": [
"## Extracting Information from the LM317 Datasheet\n",
"\n",
"For this demonstration, please download a publicly available LM317 voltage regulator datasheet (for example, from Texas Instruments) and save it as `lm317.pdf` in the `./data` directory. Then run the cell below to extract the structured technical specifications."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c58e8b7a-8f9b-46f3-8f72-3c2f96b49e8f",
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"Uploading files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:01<00:00, 1.08s/it]\n",
"Creating extraction jobs: 100%|████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:00<00:00, 1.96it/s]\n",
"Extracting files: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [01:27<00:00, 87.38s/it]\n"
]
}
],
"source": [
"# Path to the LM317 datasheet PDF\n",
"lm317_pdf = \"./data/lm317_structured_extraction/lm317.pdf\"\n",
"\n",
"# Extract structured technical specifications from the datasheet\n",
"lm317_extract = agent.extract(lm317_pdf)"
]
},
{
"cell_type": "markdown",
"id": "1a2e2e44-6c48-4a38-a6de-5f2f3c7d4d8b",
"metadata": {},
"source": [
"## Assessing the Extraction Results\n",
"\n",
"The output will be a consolidated list of LM317 technical specifications. For each entry, you should see structured fields including:\n",
"\n",
"- **component_name**\n",
"- **output_voltage** as a range (with separate `min_voltage` and `max_voltage` plus `unit`)\n",
"- **dropout_voltage** and **max_current** as numbers\n",
"- **input_voltage** as a structured range\n",
"- **pin_configuration** with a `pin_count` and `layout`\n",
"- **features** (if available)\n",
"\n",
"This structured approach makes it easier to standardize the information for downstream integration and verification. Engineers can click on the cited page numbers (in a UI that supports it) to validate the extraction."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "fb2abc44-7c9b-4b19-958e-d0d7b390ae57",
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'specs': [{'component_name': 'LM317',\n",
" 'output_voltage': {'min_voltage': 1.25, 'max_voltage': 37.0, 'unit': 'V'},\n",
" 'dropout_voltage': 0.0,\n",
" 'max_current': 1.5,\n",
" 'input_voltage': {'min_voltage': 4.25, 'max_voltage': 40.0, 'unit': 'V'},\n",
" 'pin_configuration': {'pin_count': 3,\n",
" 'layout': '1: ADJUST, 2: OUTPUT, 3: INPUT'},\n",
" 'features': ['Output voltage range adjustable from 1.25 V to 37 V',\n",
" 'Output current greater than 1.5 A',\n",
" 'Internal short-circuit current limiting',\n",
" 'Thermal overload protection',\n",
" 'Output safe-area compensation']}]}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"# Display the extraction results\n",
"lm317_extract.data"
]
},
{
"cell_type": "markdown",
"id": "c7a2a523-095e-40bf-b713-f509c13a7747",
"metadata": {},
"source": [
"You can also see the output result in the UI."
]
},
{
"cell_type": "markdown",
"id": "dc22dfa5-b667-4fb0-8dbe-24e401b12389",
"metadata": {},
"source": [
"![](data/lm317_structured_extraction/lm317_extraction.png)"
]
},
{
"cell_type": "markdown",
"id": "e0e0c12a-9f89-4bb3-b40d-3e9f7c6d2fef",
"metadata": {},
"source": [
"## Conclusion\n",
"\n",
"This notebook demonstrated how to use LlamaExtract with a structured extraction schema for the LM317 voltage regulator datasheet. By defining detailed subfields (such as splitting voltage ranges into minimum and maximum values, and structuring the pin configuration), we ensure that the extracted data is standardized and traceable through page citations. This approach minimizes manual effort and improves accuracy, providing a robust example of an agentic document workflow for technical documentation processing.\n",
"\n",
"Feel free to modify or extend the schema to capture additional technical details or to suit your own use cases."
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama_parse",
"language": "python",
"name": "llama_parse"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
@@ -1,834 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Extracting data from Resumes\n",
"\n",
"Let us assume that we are running a hiring process for a company and we have received a list of resumes from candidates. We want to extract structured data from the resumes so that we can run a screening process and shortlist candidates. \n",
"\n",
"Take a look at one of the resumes in the `data/resumes` directory. "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/html": [
"\n",
" <iframe\n",
" width=\"600\"\n",
" height=\"400\"\n",
" src=\"./data/resumes/ai_researcher.pdf\"\n",
" frameborder=\"0\"\n",
" allowfullscreen\n",
" \n",
" ></iframe>\n",
" "
],
"text/plain": [
"<IPython.lib.display.IFrame at 0x109a7dcd0>"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"from IPython.display import IFrame\n",
"\n",
"IFrame(src=\"./data/resumes/ai_researcher.pdf\", width=600, height=400)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"You will notice that all the resumes have different layouts but contain common information like name, email, experience, education, etc. \n",
"\n",
"With LlamaExtract, we will show you how to:\n",
"- *Define* a data schema to extract the information of interest. \n",
"- *Iterate* over the data schema to generalize the schema for multiple resumes.\n",
"- *Finalize* the schema and schedule extractions for multiple resumes.\n",
"\n",
"We will start by defining a `LlamaExtract` client which provides a Python interface to the LlamaExtract API. "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from dotenv import load_dotenv\n",
"from llama_cloud_services import LlamaExtract\n",
"\n",
"\n",
"# Load environment variables (put LLAMA_CLOUD_API_KEY in your .env file)\n",
"load_dotenv(override=True)\n",
"\n",
"# Optionally, add your project id/organization id\n",
"llama_extract = LlamaExtract()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Defining the data schema\n",
"\n",
"Next, let us try to extract two fields from the resume: `name` and `email`. We can either use a Python dictionary structure to define the `data_schema` as a JSON or use a Pydantic model instead, for brevity and convenience. In either case, our output is guaranteed to validate against this schema."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from pydantic import BaseModel, Field\n",
"\n",
"\n",
"class Resume(BaseModel):\n",
" name: str = Field(description=\"The name of the candidate\")\n",
" email: str = Field(description=\"The email address of the candidate\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"Uploading files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:02<00:00, 2.20s/it]\n",
"Creating extraction jobs: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:02<00:00, 2.93s/it]\n",
"Extracting files: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:02<00:00, 2.94s/it]\n",
"Uploading files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:00<00:00, 1.13it/s]\n",
"Creating extraction jobs: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:00<00:00, 1.80it/s]\n",
"Extracting files: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:15<00:00, 15.18s/it]\n",
"Uploading files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:00<00:00, 1.16it/s]\n",
"Creating extraction jobs: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:00<00:00, 2.33it/s]\n",
"Extracting files: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 1/1 [00:32<00:00, 32.86s/it]\n"
]
}
],
"source": [
"from llama_cloud.core.api_error import ApiError\n",
"\n",
"try:\n",
" existing_agent = llama_extract.get_agent(name=\"resume-screening\")\n",
" if existing_agent:\n",
" llama_extract.delete_agent(existing_agent.id)\n",
"except ApiError as e:\n",
" if e.status_code == 404:\n",
" pass\n",
" else:\n",
" raise\n",
"\n",
"agent = llama_extract.create_agent(name=\"resume-screening\", data_schema=Resume)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"[ExtractionAgent(id=1fef43b5-8230-43b4-9e80-c1cddf53889c, name=resume-screening),\n",
" ExtractionAgent(id=93f8508b-3570-46f0-ae62-6315b40043bd, name=receipt/noisebridge_receipt.pdf_56db3d92),\n",
" ExtractionAgent(id=08315f0e-7146-430b-99b8-9701cb3ace6a, name=receipt/noisebridge_receipt.pdf_5c4730a7),\n",
" ExtractionAgent(id=cfcd7756-015d-4dbd-b142-a3eefcb16cd3, name=resume/software_architect_resume.html_4a11cf15),\n",
" ExtractionAgent(id=17cb83d9-601e-4f5c-a7aa-286e3045bcb4, name=resume/software_architect_resume.html_0b7d84a8),\n",
" ExtractionAgent(id=adc8e88c-44d3-4613-a5aa-d666ef007494, name=slide/saas_slide.pdf_bcc627a5),\n",
" ExtractionAgent(id=189f14cd-6370-4476-a6ad-36eafbc62618, name=slide/saas_slide.pdf_065aa22b),\n",
" ExtractionAgent(id=b9938ca5-6225-43cb-89ea-b0065237792f, name=test2),\n",
" ExtractionAgent(id=574d37b8-59dc-41e9-bde0-5c506a8eb670, name=test)]"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"llama_extract.list_agents()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'name': 'Dr. Rachel Zhang', 'email': 'rachel.zhang@email.com'}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"resume = agent.extract(\"./data/resumes/ai_researcher.pdf\")\n",
"resume.data"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Iterating over the data schema\n",
"\n",
"Now that we have created a data schema, let us add more fields to the schema. We will add `experience` and `education` fields to the schema. \n",
"- We can create a new Pydantic model for each of these fields and represent `experience` and `education` as lists of these models. Doing this will allow us to extract multiple entities from the resume without having to pre-define how many experiences or education the candidate has. \n",
"- We have added a `description` parameter to provide more context for extraction. We can use `description` to provide example inputs/outputs for the extraction. \n",
"- Note that we have annotated the `start_date` and `end_date` fields with `Optional[str]` to indicate that these fields are optional. This is *important* because the schema will be used to extract data from multiple resumes and not all resumes will have the same format. A field must only be required if it is guaranteed to be present in all the resumes. \n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from typing import List, Optional\n",
"\n",
"\n",
"class Education(BaseModel):\n",
" institution: str = Field(description=\"The institution of the candidate\")\n",
" degree: str = Field(description=\"The degree of the candidate\")\n",
" start_date: Optional[str] = Field(\n",
" default=None, description=\"The start date of the candidate's education\"\n",
" )\n",
" end_date: Optional[str] = Field(\n",
" default=None, description=\"The end date of the candidate's education\"\n",
" )\n",
"\n",
"\n",
"class Experience(BaseModel):\n",
" company: str = Field(description=\"The name of the company\")\n",
" title: str = Field(description=\"The title of the candidate\")\n",
" description: Optional[str] = Field(\n",
" default=None, description=\"The description of the candidate's experience\"\n",
" )\n",
" start_date: Optional[str] = Field(\n",
" default=None, description=\"The start date of the candidate's experience\"\n",
" )\n",
" end_date: Optional[str] = Field(\n",
" default=None, description=\"The end date of the candidate's experience\"\n",
" )\n",
"\n",
"\n",
"class Resume(BaseModel):\n",
" name: str = Field(description=\"The name of the candidate\")\n",
" email: str = Field(description=\"The email address of the candidate\")\n",
" links: List[str] = Field(\n",
" description=\"The links to the candidate's social media profiles\"\n",
" )\n",
" experience: List[Experience] = Field(description=\"The candidate's experience\")\n",
" education: List[Education] = Field(description=\"The candidate's education\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Next, we will update the `data_schema` for the `resume-screening` agent to use the new `Resume` model. "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'name': 'Dr. Rachel Zhang',\n",
" 'email': 'rachel.zhang@email.com',\n",
" 'links': ['linkedin.com/in/rachelzhang',\n",
" 'github.com/rzhang-ai',\n",
" 'scholar.google.com/rachelzhang'],\n",
" 'experience': [{'company': 'DeepMind',\n",
" 'title': 'Senior Research Scientist',\n",
" 'description': '- Lead researcher on large-scale multi-task learning systems, developing novel architectures that improve cross-task generalization by 40%\\n- Pioneered new approach to zero-shot learning using contrastive training, published in NeurIPS 2023\\n- Built and led team of 6 researchers working on foundational ML models\\n- Developed novel regularization techniques for large language models, reducing catastrophic forgetting by 35%',\n",
" 'start_date': '2019',\n",
" 'end_date': 'Present'},\n",
" {'company': 'Google Research',\n",
" 'title': 'Research Scientist',\n",
" 'description': '- Developed probabilistic frameworks for robust ML, published in ICML 2018\\n- Created novel attention mechanisms for computer vision models, improving accuracy by 25%\\n- Led collaboration with Google Brain team on efficient training methods for transformer models\\n- Mentored 4 PhD interns and collaborated with academic institutions',\n",
" 'start_date': '2015',\n",
" 'end_date': '2019'},\n",
" {'company': 'Columbia University',\n",
" 'title': 'Research Assistant Professor',\n",
" 'description': '- Published seminal work on Bayesian optimization methods (cited 1000+ times)\\n- Taught graduate-level courses in Machine Learning and Statistical Learning Theory\\n- Supervised 5 PhD students and 3 MSc students\\n- Secured $500K in research grants for probabilistic ML research',\n",
" 'start_date': '2011',\n",
" 'end_date': '2015'}],\n",
" 'education': [{'institution': 'Columbia University',\n",
" 'degree': 'Ph.D. in Computer Science',\n",
" 'start_date': '2007',\n",
" 'end_date': '2011'},\n",
" {'institution': 'Stanford University',\n",
" 'degree': 'M.S. in Computer Science',\n",
" 'start_date': '2005',\n",
" 'end_date': '2007'}]}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"agent.data_schema = Resume\n",
"resume = agent.extract(\"./data/resumes/ai_researcher.pdf\")\n",
"resume.data"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"This is a good start. Let us add a few more fields to the schema and re-run the extraction. "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"class TechnicalSkills(BaseModel):\n",
" programming_languages: List[str] = Field(\n",
" description=\"The programming languages the candidate is proficient in.\"\n",
" )\n",
" frameworks: List[str] = Field(\n",
" description=\"The tools/frameworks the candidate is proficient in, e.g. React, Django, PyTorch, etc.\"\n",
" )\n",
" skills: List[str] = Field(\n",
" description=\"Other general skills the candidate is proficient in, e.g. Data Engineering, Machine Learning, etc.\"\n",
" )\n",
"\n",
"\n",
"class Resume(BaseModel):\n",
" name: str = Field(description=\"The name of the candidate\")\n",
" email: str = Field(description=\"The email address of the candidate\")\n",
" links: List[str] = Field(\n",
" description=\"The links to the candidate's social media profiles\"\n",
" )\n",
" experience: List[Experience] = Field(description=\"The candidate's experience\")\n",
" education: List[Education] = Field(description=\"The candidate's education\")\n",
" technical_skills: TechnicalSkills = Field(\n",
" description=\"The candidate's technical skills\"\n",
" )\n",
" key_accomplishments: str = Field(\n",
" description=\"Summarize the candidates highest achievements.\"\n",
" )"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'name': 'Dr. Rachel Zhang, Ph.D.',\n",
" 'email': 'rachel.zhang@email.com',\n",
" 'links': ['linkedin.com/in/rachelzhang',\n",
" 'github.com/rzhang-ai',\n",
" 'scholar.google.com/rachelzhang'],\n",
" 'experience': [{'company': 'DeepMind',\n",
" 'title': 'Senior Research Scientist',\n",
" 'description': 'Lead researcher on large-scale multi-task learning systems, developing novel architectures that improve cross-task generalization by 40%\\nPioneered new approach to zero-shot learning using contrastive training, published in NeurIPS 2023\\nBuilt and led team of 6 researchers working on foundational ML models\\nDeveloped novel regularization techniques for large language models, reducing catastrophic forgetting by 35%',\n",
" 'start_date': '2019',\n",
" 'end_date': 'Present'},\n",
" {'company': 'Google Research',\n",
" 'title': 'Research Scientist',\n",
" 'description': 'Developed probabilistic frameworks for robust ML, published in ICML 2018\\nCreated novel attention mechanisms for computer vision models, improving accuracy by 25%\\nLed collaboration with Google Brain team on efficient training methods for transformer models\\nMentored 4 PhD interns and collaborated with academic institutions',\n",
" 'start_date': '2015',\n",
" 'end_date': '2019'},\n",
" {'company': 'Columbia University',\n",
" 'title': 'Research Assistant Professor',\n",
" 'description': 'Published seminal work on Bayesian optimization methods (cited 1000+ times)\\nTaught graduate-level courses in Machine Learning and Statistical Learning Theory\\nSupervised 5 PhD students and 3 MSc students\\nSecured $500K in research grants for probabilistic ML research',\n",
" 'start_date': '2011',\n",
" 'end_date': '2015'}],\n",
" 'education': [{'institution': 'Columbia University',\n",
" 'degree': 'Ph.D. in Computer Science',\n",
" 'start_date': '2007',\n",
" 'end_date': '2011'},\n",
" {'institution': 'Stanford University',\n",
" 'degree': 'M.S. in Computer Science',\n",
" 'start_date': '2005',\n",
" 'end_date': '2007'}],\n",
" 'technical_skills': {'programming_languages': ['Python',\n",
" 'C++',\n",
" 'Julia',\n",
" 'CUDA'],\n",
" 'frameworks': ['PyTorch', 'TensorFlow', 'JAX', 'Ray'],\n",
" 'skills': ['Deep Learning',\n",
" 'Reinforcement Learning',\n",
" 'Probabilistic Models',\n",
" 'Multi-Task Learning',\n",
" 'Zero-Shot Learning',\n",
" 'Neural Architecture Search']},\n",
" 'key_accomplishments': 'AI researcher with 12+ years of experience spanning classical machine learning, deep learning, and probabilistic modeling. Led groundbreaking research in reinforcement learning, generative models, and multi-task learning. Published 25+ papers in top-tier conferences (NeurIPS, ICML, ICLR). Strong track record of transitioning theoretical advances into practical applications in both academic and industrial settings.'}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"agent.data_schema = Resume\n",
"resume = agent.extract(\"./data/resumes/ai_researcher.pdf\")\n",
"resume.data"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Finalizing the schema\n",
"\n",
"This is great! We have extracted a lot of key information from the resume that is well-typed and can be used downstream for further processing. Until now, this data is ephemeral and will be lost if we close the session. Let us save the state of our extraction and use it to extract data from multiple resumes. "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"agent.save()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'type': 'object',\n",
" 'required': ['name',\n",
" 'email',\n",
" 'links',\n",
" 'experience',\n",
" 'education',\n",
" 'technical_skills',\n",
" 'key_accomplishments'],\n",
" 'properties': {'name': {'type': 'string',\n",
" 'description': 'The name of the candidate'},\n",
" 'email': {'type': 'string',\n",
" 'description': 'The email address of the candidate'},\n",
" 'links': {'type': 'array',\n",
" 'items': {'type': 'string'},\n",
" 'description': \"The links to the candidate's social media profiles\"},\n",
" 'education': {'type': 'array',\n",
" 'items': {'type': 'object',\n",
" 'required': ['institution', 'degree', 'start_date', 'end_date'],\n",
" 'properties': {'degree': {'type': 'string',\n",
" 'description': 'The degree of the candidate'},\n",
" 'end_date': {'anyOf': [{'type': 'string'}, {'type': 'null'}],\n",
" 'description': \"The end date of the candidate's education\"},\n",
" 'start_date': {'anyOf': [{'type': 'string'}, {'type': 'null'}],\n",
" 'description': \"The start date of the candidate's education\"},\n",
" 'institution': {'type': 'string',\n",
" 'description': 'The institution of the candidate'}},\n",
" 'additionalProperties': False},\n",
" 'description': \"The candidate's education\"},\n",
" 'experience': {'type': 'array',\n",
" 'items': {'type': 'object',\n",
" 'required': ['company', 'title', 'description', 'start_date', 'end_date'],\n",
" 'properties': {'title': {'type': 'string',\n",
" 'description': 'The title of the candidate'},\n",
" 'company': {'type': 'string', 'description': 'The name of the company'},\n",
" 'end_date': {'anyOf': [{'type': 'string'}, {'type': 'null'}],\n",
" 'description': \"The end date of the candidate's experience\"},\n",
" 'start_date': {'anyOf': [{'type': 'string'}, {'type': 'null'}],\n",
" 'description': \"The start date of the candidate's experience\"},\n",
" 'description': {'anyOf': [{'type': 'string'}, {'type': 'null'}],\n",
" 'description': \"The description of the candidate's experience\"}},\n",
" 'additionalProperties': False},\n",
" 'description': \"The candidate's experience\"},\n",
" 'technical_skills': {'type': 'object',\n",
" 'required': ['programming_languages', 'frameworks', 'skills'],\n",
" 'properties': {'skills': {'type': 'array',\n",
" 'items': {'type': 'string'},\n",
" 'description': 'Other general skills the candidate is proficient in, e.g. Data Engineering, Machine Learning, etc.'},\n",
" 'frameworks': {'type': 'array',\n",
" 'items': {'type': 'string'},\n",
" 'description': 'The tools/frameworks the candidate is proficient in, e.g. React, Django, PyTorch, etc.'},\n",
" 'programming_languages': {'type': 'array',\n",
" 'items': {'type': 'string'},\n",
" 'description': 'The programming languages the candidate is proficient in.'}},\n",
" 'description': \"The candidate's technical skills\",\n",
" 'additionalProperties': False},\n",
" 'key_accomplishments': {'type': 'string',\n",
" 'description': 'Summarize the candidates highest achievements.'}},\n",
" 'additionalProperties': False}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"agent = llama_extract.get_agent(\"resume-screening\")\n",
"agent.data_schema # Latest schema should be returned"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"#### Queueing extractions"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"For multiple resumes, we can use the `queue_extraction` method to run extractions asynchronously. This is ideal for processing batch extraction jobs."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"Uploading files: 100%|█████████████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 3/3 [00:01<00:00, 2.13it/s]\n",
"Creating extraction jobs: 100%|████████████████████████████████████████████████████████████████████████████████████████████████████████████████| 3/3 [00:00<00:00, 5.83it/s]\n"
]
}
],
"source": [
"import os\n",
"\n",
"# All resumes in the data/resumes directory\n",
"resumes = []\n",
"\n",
"with os.scandir(\"./data/resumes\") as entries:\n",
" for entry in entries:\n",
" if entry.is_file():\n",
" resumes.append(entry.path)\n",
"\n",
"jobs = await agent.queue_extraction(resumes)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"To get the latest status of the extractions for any `job_id`, we can use the `get_extraction_job` method. \n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"[<StatusEnum.PENDING: 'PENDING'>,\n",
" <StatusEnum.PENDING: 'PENDING'>,\n",
" <StatusEnum.PENDING: 'PENDING'>]"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"[agent.get_extraction_job(job_id=job.id).status for job in jobs]"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"We notice that all extraction runs are in a PENDING state. We can check back again to see if the extractions have completed. "
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"[<StatusEnum.SUCCESS: 'SUCCESS'>,\n",
" <StatusEnum.SUCCESS: 'SUCCESS'>,\n",
" <StatusEnum.SUCCESS: 'SUCCESS'>]"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"[agent.get_extraction_job(job_id=job.id).status for job in jobs]"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"#### Retrieving results\n",
"\n",
"Let us now retrieve the results of the extractions. If the status of the extraction is `SUCCESS`, we can retrieve the data from the `data` field. In case there are errors (status = `ERROR`), we can retrieve the error message from the `error` field. \n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"results = []\n",
"for job in jobs:\n",
" extract_run = agent.get_extraction_run_for_job(job.id)\n",
" if extract_run.status == \"SUCCESS\":\n",
" results.append(extract_run.data)\n",
" else:\n",
" print(f\"Extraction status for job {job.id}: {extract_run.status}\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'name': 'Dr. Rachel Zhang, Ph.D.',\n",
" 'email': 'rachel.zhang@email.com',\n",
" 'links': ['linkedin.com/in/rachelzhang',\n",
" 'github.com/rzhang-ai',\n",
" 'scholar.google.com/rachelzhang'],\n",
" 'education': [{'degree': 'Ph.D. in Computer Science',\n",
" 'end_date': '2011',\n",
" 'start_date': '2007',\n",
" 'institution': 'Columbia University'},\n",
" {'degree': 'M.S. in Computer Science',\n",
" 'end_date': '2007',\n",
" 'start_date': '2005',\n",
" 'institution': 'Stanford University'}],\n",
" 'experience': [{'title': 'Senior Research Scientist',\n",
" 'company': 'DeepMind',\n",
" 'end_date': None,\n",
" 'start_date': '2019',\n",
" 'description': '- Lead researcher on large-scale multi-task learning systems, developing novel architectures that improve cross-task generalization by 40%\\n- Pioneered new approach to zero-shot learning using contrastive training, published in NeurIPS 2023\\n- Built and led team of 6 researchers working on foundational ML models\\n- Developed novel regularization techniques for large language models, reducing catastrophic forgetting by 35%'},\n",
" {'title': 'Research Scientist',\n",
" 'company': 'Google Research',\n",
" 'end_date': '2019',\n",
" 'start_date': '2015',\n",
" 'description': '- Developed probabilistic frameworks for robust ML, published in ICML 2018\\n- Created novel attention mechanisms for computer vision models, improving accuracy by 25%\\n- Led collaboration with Google Brain team on efficient training methods for transformer models\\n- Mentored 4 PhD interns and collaborated with academic institutions'},\n",
" {'title': 'Research Assistant Professor',\n",
" 'company': 'Columbia University',\n",
" 'end_date': '2015',\n",
" 'start_date': '2011',\n",
" 'description': '- Published seminal work on Bayesian optimization methods (cited 1000+ times)\\n- Taught graduate-level courses in Machine Learning and Statistical Learning Theory\\n- Supervised 5 PhD students and 3 MSc students\\n- Secured $500K in research grants for probabilistic ML research'}],\n",
" 'technical_skills': {'skills': ['Deep Learning',\n",
" 'Reinforcement Learning',\n",
" 'Probabilistic Models',\n",
" 'Multi-Task Learning',\n",
" 'Zero-Shot Learning',\n",
" 'Neural Architecture Search'],\n",
" 'frameworks': ['PyTorch', 'TensorFlow', 'JAX', 'Ray'],\n",
" 'programming_languages': ['Python', 'C++', 'Julia', 'CUDA']},\n",
" 'key_accomplishments': 'AI researcher with 12+ years of experience spanning classical machine learning, deep learning, and probabilistic modeling. Led groundbreaking research in reinforcement learning, generative models, and multi-task learning. Published 25+ papers in top-tier conferences (NeurIPS, ICML, ICLR). Strong track record of transitioning theoretical advances into practical applications in both academic and industrial settings.'}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"results[0]"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'name': 'Alex Park',\n",
" 'email': 'alex park@email.com',\n",
" 'links': ['linkedin.com/in/alexpark'],\n",
" 'education': [{'degree': 'M.S. Computer Science',\n",
" 'end_date': None,\n",
" 'start_date': None,\n",
" 'institution': 'University of California, Berkeley'},\n",
" {'degree': 'B.S. Computer Science',\n",
" 'end_date': None,\n",
" 'start_date': None,\n",
" 'institution': 'University of California, Berkeley'}],\n",
" 'experience': [{'title': 'Senior Machine Learning Engineer',\n",
" 'company': 'SearchTech AI',\n",
" 'end_date': None,\n",
" 'start_date': None,\n",
" 'description': 'Led development of next-generation learning-to-rank system using BER\\nArchitected and deployed real-time personalization system processing 10\\nIncreasing CTR by 15%\\nImproving search relevance by 24% (NDCG@10)'},\n",
" {'title': '',\n",
" 'company': 'Commerce Corp',\n",
" 'end_date': None,\n",
" 'start_date': None,\n",
" 'description': 'Developed semantic search system using transformer models and approximate nearest neighbors, reducing null search results by 35%'},\n",
" {'title': 'Machine Learning Engineer',\n",
" 'company': 'Tech Solutions Inc',\n",
" 'end_date': None,\n",
" 'start_date': None,\n",
" 'description': 'Implemented query understanding pipeline'},\n",
" {'title': 'Software Engineer',\n",
" 'company': '',\n",
" 'end_date': None,\n",
" 'start_date': None,\n",
" 'description': 'Built data pipelines and Flasticsearch'}],\n",
" 'technical_skills': {'skills': ['Elasticsearch',\n",
" 'Solr',\n",
" 'Lucene',\n",
" 'Python',\n",
" 'SQL',\n",
" 'Java',\n",
" 'Scala',\n",
" 'Shell Scripting'],\n",
" 'frameworks': ['PyTorch',\n",
" 'TensorFlow',\n",
" 'Scikit-learn',\n",
" 'BERT',\n",
" 'Word2Vec',\n",
" 'FastAI',\n",
" 'BM25',\n",
" 'FAISS',\n",
" 'Docker',\n",
" 'Kubernetes'],\n",
" 'programming_languages': []},\n",
" 'key_accomplishments': 'Machine Learning Engineer with 5 years of experience building and deploying large-scale search and relevance systems: Specialized in developing personalized search algorithms, learning-to-rank models; and recommendation systems. Strong track record of improving search relevance metrics and user engagement through ML-driven solutions:'}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"results[1]"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"{'name': 'Sarah Chen',\n",
" 'email': 'sarah.chen@email.com',\n",
" 'links': [],\n",
" 'education': [{'degree': 'Master of Science in Computer Science',\n",
" 'end_date': '2013',\n",
" 'start_date': None,\n",
" 'institution': 'Stanford University'},\n",
" {'degree': 'Bachelor of Science in Computer Engineering',\n",
" 'end_date': '2011',\n",
" 'start_date': None,\n",
" 'institution': 'University of California, Berkeley'}],\n",
" 'experience': [{'title': 'Senior Software Architect',\n",
" 'company': 'TechCorp Solutions',\n",
" 'end_date': None,\n",
" 'start_date': '2020',\n",
" 'description': '- Led architectural design and implementation of a cloud-native platform serving 2M+ users\\n- Established architectural guidelines and best practices adopted across 12 development teams\\n- Reduced system latency by 40% through implementation of event-driven architecture\\n- Mentored 15+ senior developers in cloud-native development practices'},\n",
" {'title': 'Lead Software Engineer',\n",
" 'company': 'DataFlow Systems',\n",
" 'end_date': '2020',\n",
" 'start_date': '2016',\n",
" 'description': '- Architected and led development of distributed data processing platform handling 5TB daily\\n- Designed microservices architecture reducing deployment time by 65%\\n- Led migration of legacy monolith to cloud-native architecture\\n- Managed team of 8 engineers across 3 international locations'},\n",
" {'title': 'Senior Software Engineer',\n",
" 'company': 'InnovateTech',\n",
" 'end_date': '2016',\n",
" 'start_date': '2013',\n",
" 'description': '- Developed high-performance trading platform processing 100K transactions per second\\n- Implemented real-time analytics engine reducing processing latency by 75%\\n- Led adoption of container orchestration reducing deployment costs by 35%'}],\n",
" 'technical_skills': {'skills': ['Architecture & Design',\n",
" 'Microservices',\n",
" 'Event-Driven Architecture',\n",
" 'Domain-Driven Design',\n",
" 'REST APIs',\n",
" 'Cloud Platforms'],\n",
" 'frameworks': ['AWS (Advanced)', 'Azure', 'Google Cloud Platform'],\n",
" 'programming_languages': ['Java', 'Python', 'Go', 'JavaScript/TypeScript']},\n",
" 'key_accomplishments': '- Co-inventor on three patents for distributed systems architecture\\n- Published paper on \"Scalable Microservices Architecture\" at IEEE Cloud Computing Conference 2022\\n- Keynote Speaker, CloudCon 2023: \"Future of Cloud-Native Architecture\"\\n- Regular presenter at local tech meetups and conferences'}"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"results[2]"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Congratulations! You now have an agent that can extract structured data from resumes. \n",
"- You can now use this agent to extract data from more resumes and use the extracted data for further processing. \n",
"- To update the schema, you can simply update the `data_schema` attribute of the agent and re-run the extraction. \n",
"- You can also use the `save` method to save the state of the agent and persist changes to the schema for future use. \n",
"\n"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
File diff suppressed because it is too large Load Diff
@@ -1,450 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "00f6713b-2a32-4f8f-80e5-9a7d9b6e3b90",
"metadata": {},
"source": [
"# Solar Panel Datasheet Comparison Workflow\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/extract/solar_panel_e2e_comparison.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"\n",
"This notebook demonstrates an endtoend agentic workflow using LlamaExtract and the LlamaIndex eventdriven workflow framework. In this workflow, we:\n",
"\n",
"1. **Extract** structured technical specifications from a solar panel datasheet (e.g. a PDF downloaded from a vendor).\n",
"2. **Load** design requirements (provided as a text blob) for a labgrade solar panel.\n",
"3. **Generate** a detailed comparison report by triggering an event that injects both the extracted data and the requirements into an LLM prompt.\n",
"\n",
"The workflow is designed for renewable energy engineers who need to quickly validate that a solar panel meets specific design criteria.\n",
"\n",
"The following notebook uses the eventdriven syntax (with custom events, steps, and a workflow class) adapted from the technical datasheet and contract review examples."
]
},
{
"cell_type": "markdown",
"id": "36d8e34e-ed98-46ac-b744-1642f6e253d5",
"metadata": {},
"source": [
"## Setup and Load Data\n",
"\n",
"We download the [Honey M TSM-DE08M.08(II) datasheet](https://static.trinasolar.com/sites/default/files/EU_Datasheet_HoneyM_DE08M.08%28II%29_2021_A.pdf) as a PDF.\n",
"\n",
"**NOTE**: The design requirements are already stored in `data/solar_panel_e2e_comparison/design_reqs.txt`."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "1de7b1b3-c285-492c-8b2e-b37974b4fc63",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"--2025-04-01 14:47:56-- https://static.trinasolar.com/sites/default/files/EU_Datasheet_HoneyM_DE08M.08%28II%29_2021_A.pdf\n",
"Resolving static.trinasolar.com (static.trinasolar.com)... 47.246.23.232, 47.246.23.234, 47.246.23.227, ...\n",
"Connecting to static.trinasolar.com (static.trinasolar.com)|47.246.23.232|:443... connected.\n",
"WARNING: cannot verify static.trinasolar.com's certificate, issued by CN=DigiCert Global G2 TLS RSA SHA256 2020 CA1,O=DigiCert Inc,C=US:\n",
" Unable to locally verify the issuer's authority.\n",
"HTTP request sent, awaiting response... 200 OK\n",
"Length: 1888183 (1.8M) [application/pdf]\n",
"Saving to: data/solar_panel_e2e_comparison/datasheet.pdf\n",
"\n",
"data/solar_panel_e2 100%[===================>] 1.80M 7.47MB/s in 0.2s \n",
"\n",
"2025-04-01 14:47:56 (7.47 MB/s) - data/solar_panel_e2e_comparison/datasheet.pdf saved [1888183/1888183]\n",
"\n"
]
}
],
"source": [
"!wget https://static.trinasolar.com/sites/default/files/EU_Datasheet_HoneyM_DE08M.08%28II%29_2021_A.pdf -O data/solar_panel_e2e_comparison/datasheet.pdf --no-check-certificate"
]
},
{
"cell_type": "markdown",
"id": "89d2f4c9-f785-424d-a409-3381796c457c",
"metadata": {},
"source": [
"## Define the Structured Extraction Schema\n",
"\n",
"We define a new, rich schema called `SolarPanelSchema` to capture key technical details from the datasheet. This schema includes:\n",
"\n",
"- **PowerRange:** Structured as minimum and maximum power output (in Watts).\n",
"- **SolarPanelSpec:** Includes module name, power output range, maximum efficiency, certifications, and a mapping of page citations.\n",
"\n",
"This schema replaces the earlier LM317 schema and will be used when creating our extraction agent."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "bfb40d48-36e0-4b1c-97a1-32a1704c582b",
"metadata": {},
"outputs": [],
"source": [
"from pydantic import BaseModel, Field\n",
"from typing import List\n",
"\n",
"\n",
"class PowerRange(BaseModel):\n",
" min_power: float = Field(..., description=\"Minimum power output in Watts\")\n",
" max_power: float = Field(..., description=\"Maximum power output in Watts\")\n",
" unit: str = Field(\"W\", description=\"Power unit\")\n",
"\n",
"\n",
"class SolarPanelSpec(BaseModel):\n",
" module_name: str = Field(..., description=\"Name or model of the solar panel module\")\n",
" power_output: PowerRange = Field(..., description=\"Power output range\")\n",
" maximum_efficiency: float = Field(\n",
" ..., description=\"Maximum module efficiency in percentage\"\n",
" )\n",
" temperature_coefficient: float = Field(\n",
" ..., description=\"Temperature coefficient in %/°C\"\n",
" )\n",
" certifications: List[str] = Field([], description=\"List of certifications\")\n",
" page_citations: dict = Field(\n",
" ..., description=\"Mapping of each extracted field to its page numbers\"\n",
" )\n",
"\n",
"\n",
"class SolarPanelSchema(BaseModel):\n",
" specs: List[SolarPanelSpec] = Field(\n",
" ..., description=\"List of extracted solar panel specifications\"\n",
" )"
]
},
{
"cell_type": "markdown",
"id": "19dc309e-7cec-43c1-8f6c-72e14df58f8f",
"metadata": {},
"source": [
"## Initialize Extraction Agent\n",
"\n",
"Here we initialize our extraction agent that will be responsible for extracting the schema from the solar panel datasheet."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c9d9f4a2-2e14-493d-8a7e-d01159d38b8f",
"metadata": {},
"outputs": [],
"source": [
"from dotenv import load_dotenv\n",
"from llama_cloud_services import LlamaExtract\n",
"from llama_cloud.core.api_error import ApiError\n",
"from llama_cloud import ExtractConfig\n",
"\n",
"# Initialize the LlamaExtract client\n",
"llama_extract = LlamaExtract(\n",
" project_id=\"2fef999e-1073-40e6-aeb3-1f3c0e64d99b\",\n",
" organization_id=\"43b88c8f-e488-46f6-9013-698e3d2e374a\",\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ec0eb2a7-6e02-45da-a6af-227e2f7c81f2",
"metadata": {},
"outputs": [],
"source": [
"try:\n",
" existing_agent = llama_extract.get_agent(name=\"solar-panel-datasheet\")\n",
" if existing_agent:\n",
" llama_extract.delete_agent(existing_agent.id)\n",
"except ApiError as e:\n",
" if e.status_code == 404:\n",
" pass\n",
" else:\n",
" raise\n",
"\n",
"extract_config = ExtractConfig(\n",
" extraction_mode=\"BALANCED\",\n",
")\n",
"\n",
"agent = llama_extract.create_agent(\n",
" name=\"solar-panel-datasheet\", data_schema=SolarPanelSchema, config=extract_config\n",
")"
]
},
{
"cell_type": "markdown",
"id": "b4d7bb60-0456-4a2d-8d48-14f9bb3e71d2",
"metadata": {},
"source": [
"## Workflow Overview\n",
"\n",
"The workflow consists of four main steps:\n",
"\n",
"1. **parse_datasheet:** Reads the solar panel datasheet (PDF) and converts its content into text (with page citations).\n",
"2. **load_requirements:** Loads the design requirements (as a text blob) that will be injected into the prompt.\n",
"3. **generate_comparison_report:** Constructs a prompt using the extracted datasheet content and design requirements and triggers the LLM to generate a comparison report.\n",
"4. **output_result:** Logs and returns the final report as the workflows result.\n",
"\n",
"Each step is implemented as an asynchronous function decorated with `@step`, and the workflow is built by subclassing `Workflow`."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "7c482e3a-66b4-4e1b-8d2d-9a9c6b3967f3",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.workflow import (\n",
" Event,\n",
" StartEvent,\n",
" StopEvent,\n",
" Context,\n",
" Workflow,\n",
" step,\n",
")\n",
"from llama_index.llms.openai import OpenAI\n",
"from llama_index.core.prompts import ChatPromptTemplate\n",
"from llama_cloud_services import LlamaExtract\n",
"from llama_cloud.core.api_error import ApiError\n",
"from pydantic import BaseModel, Field\n",
"from typing import List\n",
"\n",
"\n",
"# Define output schema for the comparison report (for reference)\n",
"class ComparisonReportOutput(BaseModel):\n",
" component_name: str = Field(\n",
" ..., description=\"The name of the component being evaluated.\"\n",
" )\n",
" meets_requirements: bool = Field(\n",
" ...,\n",
" description=\"Overall indicator of whether the component meets the design criteria.\",\n",
" )\n",
" summary: str = Field(..., description=\"A brief summary of the evaluation results.\")\n",
" details: dict = Field(\n",
" ..., description=\"Detailed comparisons for each key parameter.\"\n",
" )\n",
"\n",
"\n",
"# Define custom events\n",
"\n",
"\n",
"class DatasheetParseEvent(Event):\n",
" datasheet_content: dict\n",
"\n",
"\n",
"class RequirementsLoadEvent(Event):\n",
" requirements_text: str\n",
"\n",
"\n",
"class ComparisonReportEvent(Event):\n",
" report: ComparisonReportOutput\n",
"\n",
"\n",
"class LogEvent(Event):\n",
" msg: str\n",
" delta: bool = False\n",
"\n",
"\n",
"# For our demonstration, we assume that LlamaExtract is used to parse the datasheet into text.\n",
"# We'll also use OpenAI (via LlamaIndex) as our LLM for generating the report.\n",
"\n",
"llm = OpenAI(model=\"gpt-4o\") # or your preferred model"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "67a0c391-c7f5-4b93-8d6b-9e31b2d7a817",
"metadata": {},
"outputs": [],
"source": [
"class SolarPanelComparisonWorkflow(Workflow):\n",
" \"\"\"\n",
" Workflow to extract data from a solar panel datasheet and generate a comparison report\n",
" against provided design requirements.\n",
" \"\"\"\n",
"\n",
" def __init__(self, agent: LlamaExtract, requirements_path: str, **kwargs):\n",
" super().__init__(**kwargs)\n",
" self.agent = agent\n",
" # Load design requirements from file as a text blob\n",
" with open(requirements_path, \"r\") as f:\n",
" self.requirements_text = f.read()\n",
"\n",
" @step\n",
" async def parse_datasheet(\n",
" self, ctx: Context, ev: StartEvent\n",
" ) -> DatasheetParseEvent:\n",
" # datasheet_path is provided in the StartEvent\n",
" datasheet_path = (\n",
" ev.datasheet_path\n",
" ) # e.g., \"./data/solar_panel_comparison/datasheet.pdf\"\n",
" extraction_result = await self.agent.aextract(datasheet_path)\n",
" datasheet_dict = (\n",
" extraction_result.data\n",
" ) # assumed to be a string with page citations\n",
" await ctx.set(\"datasheet_content\", datasheet_dict)\n",
" ctx.write_event_to_stream(LogEvent(msg=\"Datasheet parsed successfully.\"))\n",
" return DatasheetParseEvent(datasheet_content=datasheet_dict)\n",
"\n",
" @step\n",
" async def load_requirements(\n",
" self, ctx: Context, ev: DatasheetParseEvent\n",
" ) -> RequirementsLoadEvent:\n",
" # Use the pre-loaded requirements text from __init__\n",
" req_text = self.requirements_text\n",
" ctx.write_event_to_stream(LogEvent(msg=\"Design requirements loaded.\"))\n",
" return RequirementsLoadEvent(requirements_text=req_text)\n",
"\n",
" @step\n",
" async def generate_comparison_report(\n",
" self, ctx: Context, ev: RequirementsLoadEvent\n",
" ) -> StopEvent:\n",
" # Build a prompt that injects both the extracted datasheet content and the design requirements\n",
" datasheet_content = await ctx.get(\"datasheet_content\")\n",
" prompt_str = \"\"\"\n",
"You are an expert renewable energy engineer.\n",
"\n",
"Compare the following solar panel datasheet information with the design requirements.\n",
"\n",
"Design Requirements:\n",
"{requirements_text}\n",
"\n",
"Extracted Datasheet Information:\n",
"{datasheet_content}\n",
"\n",
"Generate a detailed comparison report in JSON format with the following schema:\n",
" - component_name: string\n",
" - meets_requirements: boolean\n",
" - summary: string\n",
" - details: dictionary of comparisons for each parameter\n",
"\n",
"For each parameter (Maximum Power, Open-Circuit Voltage, Short-Circuit Current, Efficiency, Temperature Coefficient),\n",
"indicate PASS or FAIL and provide brief explanations and recommendations.\n",
"\"\"\"\n",
"\n",
" # extract from contract\n",
" prompt = ChatPromptTemplate.from_messages([(\"user\", prompt_str)])\n",
"\n",
" # Call the LLM to generate the report using the prompt\n",
" report_output = await llm.astructured_predict(\n",
" ComparisonReportOutput,\n",
" prompt,\n",
" requirements_text=ev.requirements_text,\n",
" datasheet_content=str(datasheet_content),\n",
" )\n",
" ctx.write_event_to_stream(LogEvent(msg=\"Comparison report generated.\"))\n",
" return StopEvent(\n",
" result={\"report\": report_output, \"datasheet_content\": datasheet_content}\n",
" )"
]
},
{
"cell_type": "markdown",
"id": "d205f532-1a11-4a48-b5a8-87a7f85e9ce7",
"metadata": {},
"source": [
"## Running the Workflow\n",
"\n",
"Below, we instantiate and run the workflow. We inject the design requirements as a text blob (no custom code to load) and pass the path to the solar panel datasheet (the HoneyM datasheet from Trina).\n",
"\n",
"The design requirements are:\n",
"\n",
"```\n",
"Solar Panel Design Requirements:\n",
"- Power Output Range: ≥ 350 W\n",
"- Maximum Efficiency: ≥ 18%\n",
"- Certifications: Must include IEC61215 and UL1703\n",
"```\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6b24fa61-a2f5-4ebb-84eb-1c9b48683b1b",
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "be3ebad5-1f70-4671-a2ec-17bf9e4d788f",
"metadata": {},
"outputs": [],
"source": [
"# Path to design requirements file (e.g., a text file with design criteria for solar panels)\n",
"requirements_path = \"./data/solar_panel_e2e_comparison/design_reqs.txt\"\n",
"\n",
"# Instantiate the workflow\n",
"workflow = SolarPanelComparisonWorkflow(\n",
" agent=agent, requirements_path=requirements_path, verbose=True, timeout=120\n",
")\n",
"\n",
"# Run the workflow; pass the datasheet path in the StartEvent\n",
"result = await workflow.run(\n",
" datasheet_path=\"./data/solar_panel_e2e_comparison/datasheet.pdf\"\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e1e61f1e-8701-4acc-8f99-cc89d8aae535",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"\n",
"********Final Comparison Report:********\n",
"\n",
"{\n",
" \"component_name\": \"TSM-DE08M.08(II)\",\n",
" \"meets_requirements\": true,\n",
" \"summary\": \"The solar panel TSM-DE08M.08(II) meets all the design requirements, making it a suitable choice for the intended application.\",\n",
" \"details\": {\n",
" \"Maximum Power Output\": \"PASS - The panel's power output ranges from 360 W to 385 W, exceeding the minimum requirement of 350 W.\",\n",
" \"Open-Circuit Voltage\": \"PASS - The datasheet does not specify Voc, but the panel meets other critical requirements. Verification of Voc is recommended.\",\n",
" \"Short-Circuit Current\": \"PASS - The datasheet does not specify Isc, but the panel meets other critical requirements. Verification of Isc is recommended.\",\n",
" \"Efficiency\": \"PASS - The panel's efficiency is 21.0%, which is above the required 18%.\",\n",
" \"Temperature Coefficient\": \"PASS - The temperature coefficient is -0.34%/°C, which is better than the maximum allowable -0.5%/°C.\"\n",
" }\n",
"}\n"
]
}
],
"source": [
"print(\"\\n********Final Comparison Report:********\\n\")\n",
"print(result[\"report\"].model_dump_json(indent=4))\n",
"# print(\"\\n********Datasheet Content:********\\n\", result[\"datasheet_content\"])"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama_parse",
"language": "python",
"name": "llama_parse"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
File diff suppressed because it is too large Load Diff
@@ -1,302 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# LlamaParse Agent\n",
"\n",
"This demo walks through using an OpenAI Agent with [LlamaParse](https://cloud.llamaindex.ai)."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Setup"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install llama-cloud-services llama-index llama-index-postprocessor-sbert-rerank"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"llx-...\"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-...\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import Settings\n",
"from llama_index.embeddings.openai import OpenAIEmbedding\n",
"from llama_index.llms.openai import OpenAI\n",
"\n",
"Settings.embed_model = OpenAIEmbedding(model=\"text-embedding-3-small\")\n",
"Settings.llm = OpenAI(model=\"gpt-3.5-turbo\", temperature=0.2)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Parsing \n",
"\n",
"For parsing, lets use a [recent paper](https://huggingface.co/papers/2403.09611) on Multi-Modal pretraining"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!wget https://arxiv.org/pdf/2403.09611.pdf -O paper.pdf"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Below, we can tell the parser to skip content we don't want. In this case, the references section will just add noise to a RAG system."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(\n",
" result_type=\"markdown\",\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 81251f39-01be-434e-99e8-1c1b83b82098\n"
]
}
],
"source": [
"documents = await parser.aload_data(\"paper.pdf\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Embeddings have been explicitly disabled. Using MockEmbedding.\n"
]
},
{
"name": "stderr",
"output_type": "stream",
"text": [
"41it [00:00, 26765.21it/s]\n",
"100%|██████████| 41/41 [00:13<00:00, 2.98it/s]\n"
]
}
],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()\n",
"\n",
"from llama_index.core.node_parser import (\n",
" MarkdownElementNodeParser,\n",
" SentenceSplitter,\n",
")\n",
"\n",
"# explicitly extract tables with the MarkdownElementNodeParser\n",
"node_parser = MarkdownElementNodeParser(num_workers=8)\n",
"nodes = node_parser.get_nodes_from_documents(documents)\n",
"nodes, objects = node_parser.get_nodes_and_objects(nodes)\n",
"\n",
"# Chain splitters to ensure chunk size requirements are met\n",
"nodes = SentenceSplitter(chunk_size=512, chunk_overlap=20).get_nodes_from_documents(\n",
" nodes\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Chat over the paper, lets find out what it is about!"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import VectorStoreIndex, SummaryIndex\n",
"\n",
"vector_index = VectorStoreIndex(nodes=nodes)\n",
"summary_index = SummaryIndex(nodes=nodes)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.agent.openai import OpenAIAgent\n",
"from llama_index.core.tools import QueryEngineTool, ToolMetadata\n",
"from llama_index.postprocessor.colbert_rerank import ColbertRerank\n",
"\n",
"tools = [\n",
" QueryEngineTool(\n",
" vector_index.as_query_engine(\n",
" similarity_top_k=8, node_postprocessors=[ColbertRerank(top_n=3)]\n",
" ),\n",
" metadata=ToolMetadata(\n",
" name=\"search\",\n",
" description=\"Search the document, pass the entire user message in the query\",\n",
" ),\n",
" ),\n",
" QueryEngineTool(\n",
" summary_index.as_query_engine(),\n",
" metadata=ToolMetadata(\n",
" name=\"summarize\",\n",
" description=\"Summarize the document using the user message\",\n",
" ),\n",
" ),\n",
"]\n",
"\n",
"agent = OpenAIAgent.from_tools(tools=tools, verbose=True)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Added user message to memory: What is the summary of the paper?\n",
"=== Calling Function ===\n",
"Calling function: summarize with args: {\"input\":\"summary\"}\n",
"Got output: The research focuses on developing Multimodal Large Language Models (MLLMs) by incorporating image-caption, interleaved image-text, and text-only data for pre-training. It highlights the importance of factors like the image encoder, resolution, and token count, while downplaying the design of the vision-language connector. With models scaling up to 30B parameters, the MM1 family demonstrates impressive performance in pre-training metrics and competitive outcomes on diverse multimodal benchmarks. It demonstrates abilities such as in-context learning and multi-image reasoning, aiming to provide valuable insights for creating MLLMs that benefit the research community.\n",
"========================\n",
"\n"
]
}
],
"source": [
"# note -- this will take a while with local LLMs, its sending every node in the document to the LLM\n",
"resp = agent.chat(\"What is the summary of the paper?\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The summary of the paper highlights the development of Multimodal Large Language Models (MLLMs) by incorporating image-caption, interleaved image-text, and text-only data for pre-training. The research emphasizes factors like the image encoder, resolution, and token count, while de-emphasizing the design of the vision-language connector. The MM1 family of models, scaling up to 30B parameters, shows impressive performance in pre-training metrics and competitive outcomes on various multimodal benchmarks. These models demonstrate capabilities such as in-context learning and multi-image reasoning, aiming to provide valuable insights for creating MLLMs that benefit the research community.\n"
]
}
],
"source": [
"print(str(resp))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Added user message to memory: How do the authors evaluate their work?\n",
"=== Calling Function ===\n",
"Calling function: search with args: {\"input\":\"evaluation methods\"}\n",
"Got output: The evaluation methods involve synthesizing all benchmark results into a single meta-average number to simplify comparisons. This is achieved by normalizing the evaluation metrics with respect to a baseline configuration, standardizing the results for each task, adjusting every metric by dividing it by its respective baseline, and then averaging across all metrics.\n",
"========================\n",
"\n"
]
}
],
"source": [
"resp = agent.chat(\"How do the authors evaluate their work?\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The authors evaluate their work by synthesizing all benchmark results into a single meta-average number to simplify comparisons. They normalize the evaluation metrics with respect to a baseline configuration, standardize the results for each task, adjust every metric by dividing it by its respective baseline, and then average across all metrics for evaluation.\n"
]
}
],
"source": [
"print(str(resp))"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama-parse-aNC435Vv-py3.10",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
-618
View File
@@ -1,618 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Advanced RAG with LlamaParse\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_parse/blob/main/examples/demo_advanced.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"This notebook is a complete walkthrough for using LlamaParse with advanced indexing/retrieval techniques in LlamaIndex over the Apple 10K Filing. \n",
"\n",
"This allows us to ask sophisticated questions that aren't possible with \"naive\" parsing/indexing techniques with existing models."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-index llama-cloud-services"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!wget \"https://s2.q4cdn.com/470004039/files/doc_financials/2021/q4/_10-K-2021-(As-Filed).pdf\" -O apple_2021_10k.pdf"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Some OpenAI and LlamaParse details"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"# API access to llama-cloud\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"llx-...\"\n",
"\n",
"# Using OpenAI API for embeddings/llms\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-proj-...\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.llms.openai import OpenAI\n",
"from llama_index.embeddings.openai import OpenAIEmbedding\n",
"from llama_index.core import Settings\n",
"\n",
"embed_model = OpenAIEmbedding(model_name=\"text-embedding-3-small\")\n",
"llm = OpenAI(model=\"gpt-4o-mini\")\n",
"\n",
"Settings.llm = llm\n",
"Settings.embed_model = embed_model"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Using brand new `LlamaParse` PDF reader for PDF Parsing\n",
"\n",
"We also compare three different retrieval/query engine strategies:\n",
"1. Baseline using default parsing from `SimpleDirectoryReader`\n",
"2. Using raw markdown text as nodes for building index and apply simple query engine for generating the results;\n",
"3. Using markdown + page screenshots to help retrieve the proper nodes."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id e403a457-1721-4093-82bf-4a316d2d637a\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"result = await LlamaParse(take_screenshot=True).aparse(\"./apple_2021_10k.pdf\")\n",
"\n",
"markdown_nodes = await result.aget_markdown_nodes(split_by_page=True)\n",
"screenshot_image_nodes = await result.aget_image_nodes(\n",
" include_screenshot_images=True,\n",
" include_object_images=False,\n",
" image_download_dir=\"./images\",\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import SimpleDirectoryReader\n",
"\n",
"baseline_documents = SimpleDirectoryReader(\n",
" input_files=[\"apple_2021_10k.pdf\"]\n",
").load_data()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Setup Baseline Index\n",
"\n",
"For comparison, we setup a naive RAG pipeline with default parsing and standard chunking, indexing, retrieval."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import VectorStoreIndex\n",
"\n",
"baseline_index = VectorStoreIndex.from_documents(baseline_documents)\n",
"baseline_query_engine = baseline_index.as_query_engine(similarity_top_k=3)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Setup our LlamaParse Indexes\n",
"\n",
"Using both the markdown and screenshot images, we can build two different indexes.\n",
"\n",
"1. An index over just the markdown documents\n",
"2. A custom index that uses the markdown + screenshot images to help with response quality."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import VectorStoreIndex\n",
"\n",
"markdown_index = VectorStoreIndex(nodes=markdown_nodes)\n",
"markdown_query_engine = markdown_index.as_query_engine(similarity_top_k=3)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.indices import MultiModalVectorStoreIndex\n",
"from llama_index.embeddings.huggingface import HuggingFaceEmbedding\n",
"from llama_index.core import Settings\n",
"\n",
"# could also use other API-based multimodal models like voyageai or jinaai\n",
"# Note: this may take quite a while if running on CPU!\n",
"image_embed_model = HuggingFaceEmbedding(\n",
" model_name=\"llamaindex/vdr-2b-multi-v1\",\n",
" embed_batch_size=2,\n",
" trust_remote_code=True,\n",
" cache_folder=\"./hf_cache_2\",\n",
" device=\"cpu\", # set to \"cuda\" if you have a GPU or remove to auto-detect\n",
")\n",
"\n",
"multi_modal_index = MultiModalVectorStoreIndex(\n",
" nodes=[*markdown_nodes, *screenshot_image_nodes],\n",
" embed_model=Settings.embed_model,\n",
" image_embed_model=image_embed_model,\n",
" show_progress=True,\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Below, we will create a custom query engine that does a few things\n",
"1. Retrieves both image nodes and text nodes\n",
"2. Combines them into two lists -- one where images and texts come from the same page, and one where we have texts alone\n",
"3. Use a Jinja-based `RichPromptTemplate` to format the retrieved content automatically into a list of multimodal chat messages\n",
"4. Send our messages to the LLM and return a result\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.async_utils import asyncio_run\n",
"from llama_index.core.llms import LLM\n",
"from llama_index.core.query_engine import CustomQueryEngine\n",
"from llama_index.core.prompts import RichPromptTemplate\n",
"from llama_index.core.response import Response\n",
"from llama_index.core.schema import NodeWithScore\n",
"from llama_index.core import Settings\n",
"\n",
"TEXT_IMAGE_PROMPT_TEMPLATE = RichPromptTemplate(\n",
" \"\"\"\n",
"<context>\n",
"Here is some retrieved content from a knowledge base:\n",
"{% for image_path, text in images_and_texts %}\n",
"<page>\n",
"<text>{{ text }}</text>\n",
"<image>{{ image_path | image }}</image>\n",
"</page>\n",
"{% endfor %}\n",
"{% for text in texts %}\n",
"<page>\n",
"<text>{{ text }}</text>\n",
"</page>\n",
"{% endfor %}\n",
"</context>\n",
"\n",
"Using the context, answer the following question:\n",
"<query>{{ query_str }}</query>\n",
"\"\"\"\n",
")\n",
"\n",
"\n",
"class SimpleMultiModalQueryEngine(CustomQueryEngine):\n",
" def __init__(\n",
" self,\n",
" index: MultiModalVectorStoreIndex,\n",
" image_top_k: int = 4,\n",
" text_top_k: int = 4,\n",
" llm: LLM | None = None,\n",
" **kwargs\n",
" ):\n",
" super().__init__(**kwargs)\n",
" self._retriever = index.as_retriever(\n",
" similarity_top_k=text_top_k, image_similarity_top_k=image_top_k\n",
" )\n",
" self._llm = llm or Settings.llm\n",
"\n",
" def _match_images_and_texts(\n",
" self, text_results: list[NodeWithScore], image_results: list[NodeWithScore]\n",
" ) -> tuple[list[NodeWithScore], list[NodeWithScore]]:\n",
" # combine results, prioritize images and texts\n",
" # if both an image and matching text was retrieved, that is a strong indicator\n",
" images_and_texts = []\n",
" text_keys = {\n",
" (x.metadata[\"page_number\"], x.metadata[\"file_name\"]): x\n",
" for x in text_results\n",
" }\n",
" for image_result in image_results:\n",
" key = (\n",
" image_result.metadata[\"page_number\"],\n",
" image_result.metadata[\"file_name\"],\n",
" )\n",
" # add matching text to results if available\n",
" if key in text_keys:\n",
" text_result = text_keys[key]\n",
" images_and_texts.append(\n",
" (image_result.node.image_path, text_result.node.text)\n",
" )\n",
"\n",
" # remove from list\n",
" text_keys.pop(key)\n",
"\n",
" # get the remaining texts as a fallback\n",
" texts = [result.node.text for result in text_keys.values()]\n",
"\n",
" return images_and_texts, texts\n",
"\n",
" def custom_query(self, query_str: str) -> Response:\n",
" # wrap the async method to avoid code duplication\n",
" # asyncio_run is a slightly safer asyncio.run() call\n",
" return asyncio_run(self.acustom_query(query_str))\n",
"\n",
" async def acustom_query(self, query_str: str) -> Response:\n",
" text_results = await self._retriever.atext_retrieve(query_str)\n",
" image_results = await self._retriever.atext_to_image_retrieve(query_str)\n",
"\n",
" images_and_texts, texts = self._match_images_and_texts(\n",
" text_results, image_results\n",
" )\n",
" messages = TEXT_IMAGE_PROMPT_TEMPLATE.format_messages(\n",
" images_and_texts=images_and_texts, texts=texts, query_str=str(query_str)\n",
" )\n",
"\n",
" response = await self._llm.achat(messages)\n",
"\n",
" return Response(\n",
" response.message.content, source_nodes=[*text_results, *image_results]\n",
" )"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"multimodal_query_engine = SimpleMultiModalQueryEngine(\n",
" index=multi_modal_index,\n",
" image_top_k=3,\n",
" text_top_k=3,\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Try out the Query Engines and Compare!\n",
"\n",
"Now with our three query engines assembled, we can compare each approach with a rough \"vibes-based\" evaluation."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"\n",
"***********Baseline Query Engine***********\n",
"The total fair value of marketable securities in 2020 was $190,516 million.\n",
"\n",
"***********Markdown Query Engine***********\n",
"The total fair value of marketable securities in 2020 was $191,830 million.\n",
"\n",
"***********MultiModal Query Engine***********\n",
"The total fair value of marketable securities in 2020 was $191,830 million.\n"
]
}
],
"source": [
"query = \"What were the total fair value of marketable securities in 2020\"\n",
"\n",
"response_1 = await baseline_query_engine.aquery(query)\n",
"print(\"\\n***********Baseline Query Engine***********\")\n",
"print(response_1)\n",
"\n",
"response_2 = await markdown_query_engine.aquery(query)\n",
"print(\"\\n***********Markdown Query Engine***********\")\n",
"print(response_2)\n",
"\n",
"response_3 = await multimodal_query_engine.aquery(query)\n",
"print(\"\\n***********MultiModal Query Engine***********\")\n",
"print(response_3)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"As we can see, the multimodal and markdown query engines are able to retrieve the correct content, while the default query engine struggles to find the correct total value."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"We can also inspect the source nodes, and see the pages that were retrieved. Here is the correct page for the total fair value of marketable securities in 2020:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"'images/page_41.jpg'"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"response_3.source_nodes[4].node.image_path"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Lets try a few more queries to see how the query engines perform."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"\n",
"***********Baseline Query Engine***********\n",
"The effective interest rates for the debt issuances in 2021 were as follows:\n",
"\n",
"- Floating-rate notes: 0.48% 0.63%\n",
"- Fixed-rate notes: 0.03% 4.78% for maturities from 2022 to 2060\n",
"- Fixed-rate notes issued in the second quarter: 0.75% 2.81% for maturities from 2026 to 2061\n",
"- Fixed-rate notes issued in the fourth quarter: 1.43% 2.86% for maturities from 2028 to 2061\n",
"\n",
"***********Markdown Query Engine***********\n",
"The effective interest rates for the debt issuances in 2021 were as follows:\n",
"\n",
"- Floating-rate notes: 0.48% 0.63%\n",
"- Fixed-rate notes: 0.03% 4.78% for the 0.000% 4.650% notes, 0.75% 2.81% for the 0.700% 2.800% notes, and 1.43% 2.86% for the 1.400% 2.850% notes.\n",
"\n",
"***********MultiModal Query Engine***********\n",
"The effective interest rates of all debt issuances in 2021 were as follows:\n",
"\n",
"1. **Floating-rate notes**: 0.48% 0.63%\n",
"2. **Fixed-rate 0.000% 4.650% notes**: 0.03% 4.78%\n",
"3. **Fixed-rate 0.700% 2.800% notes**: 0.75% 2.81%\n",
"4. **Fixed-rate 1.400% 2.850% notes**: 1.43% 2.86%\n"
]
}
],
"source": [
"query = \"What were the effective interest rates of all debt issuances in 2021\"\n",
"\n",
"response_1 = await baseline_query_engine.aquery(query)\n",
"print(\"\\n***********Baseline Query Engine***********\")\n",
"print(response_1)\n",
"\n",
"response_2 = await markdown_query_engine.aquery(query)\n",
"print(\"\\n***********Markdown Query Engine***********\")\n",
"print(response_2)\n",
"\n",
"response_3 = await multimodal_query_engine.aquery(query)\n",
"print(\"\\n***********MultiModal Query Engine***********\")\n",
"print(response_3)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"\n",
"***********Baseline Query Engine***********\n",
"The federal deferred tax amounts for the years 2019 to 2021 are as follows (in millions):\n",
"\n",
"- **2019**: $(2,939)\n",
"- **2020**: $(3,619)\n",
"- **2021**: $(7,176)\n",
"\n",
"These figures represent the deferred tax expense for each respective year.\n",
"\n",
"***********Markdown Query Engine***********\n",
"As of September 25, 2021, the total deferred tax assets and liabilities for the years 2021 and 2020 are as follows:\n",
"\n",
"**Deferred Tax Assets:**\n",
"- 2021: $25,176 million\n",
"- 2020: $19,336 million\n",
"\n",
"**Deferred Tax Liabilities:**\n",
"- 2021: $7,200 million\n",
"- 2020: $10,138 million\n",
"\n",
"**Net Deferred Tax Assets:**\n",
"- 2021: $13,073 million\n",
"- 2020: $8,157 million\n",
"\n",
"The information for 2019 is not provided in the context.\n",
"\n",
"***********MultiModal Query Engine***********\n",
"The federal deferred tax assets and liabilities for the years 2019 to 2021 are as follows:\n",
"\n",
"### Deferred Tax Assets (in millions):\n",
"- **2021**: $25,176\n",
"- **2020**: $19,336\n",
"- **2019**: Not specified in the provided content.\n",
"\n",
"### Deferred Tax Liabilities (in millions):\n",
"- **2021**: $7,200\n",
"- **2020**: $10,138\n",
"- **2019**: Not specified in the provided content.\n",
"\n",
"### Net Deferred Tax Assets (in millions):\n",
"- **2021**: $13,073\n",
"- **2020**: $8,157\n",
"- **2019**: Not specified in the provided content.\n",
"\n",
"The significant components of deferred tax assets and liabilities reflect the effects of tax credits and temporary differences between financial statement carrying amounts and their respective tax bases.\n"
]
}
],
"source": [
"query = \"federal deferred tax in 2019-2021\"\n",
"\n",
"response_1 = await baseline_query_engine.aquery(query)\n",
"print(\"\\n***********Baseline Query Engine***********\")\n",
"print(response_1)\n",
"\n",
"response_2 = await markdown_query_engine.aquery(query)\n",
"print(\"\\n***********Markdown Query Engine***********\")\n",
"print(response_2)\n",
"\n",
"response_3 = await multimodal_query_engine.aquery(query)\n",
"print(\"\\n***********MultiModal Query Engine***********\")\n",
"print(response_3)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"\n",
"***********Baseline Query Engine***********\n",
"The current state taxes for the years 2019 to 2021 are as follows (in millions):\n",
"\n",
"- 2021: $1,620\n",
"- 2020: $455\n",
"- 2019: $475\n",
"\n",
"This indicates an increase of $1,165 million from 2020 to 2021, a decrease of $20 million from 2018 to 2019, and an increase of $80 million from 2019 to 2020.\n",
"\n",
"***********Markdown Query Engine***********\n",
"The current state taxes for the years 2019 to 2021 are as follows (in millions):\n",
"\n",
"- **2021**: $1,620\n",
"- **2020**: $455\n",
"- **2019**: $475\n",
"\n",
"The changes in current state taxes from year to year are:\n",
"\n",
"- From 2019 to 2020: Decrease of $20 million\n",
"- From 2020 to 2021: Increase of $1,165 million\n",
"\n",
"***********MultiModal Query Engine***********\n",
"The current state taxes for the years 2019 to 2021 are as follows (in millions):\n",
"\n",
"- **2021**: $1,620\n",
"- **2020**: $455\n",
"- **2019**: $475\n",
"\n",
"So, the changes are:\n",
"- From 2019 to 2020: Decrease of $20 million\n",
"- From 2020 to 2021: Increase of $1,165 million\n"
]
}
],
"source": [
"query = \"current state taxes per year in 2019-2021 (include +/-)\"\n",
"\n",
"response_1 = await baseline_query_engine.aquery(query)\n",
"print(\"\\n***********Baseline Query Engine***********\")\n",
"print(response_1)\n",
"\n",
"response_2 = await markdown_query_engine.aquery(query)\n",
"print(\"\\n***********Markdown Query Engine***********\")\n",
"print(response_2)\n",
"\n",
"response_3 = await multimodal_query_engine.aquery(query)\n",
"print(\"\\n***********MultiModal Query Engine***********\")\n",
"print(response_3)"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama-parse-aNC435Vv-py3.10",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
-196
View File
@@ -1,196 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# LlamaParse Usage"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-index llama-cloud-services"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!wget \"https://arxiv.org/pdf/1706.03762.pdf\" -O \"./attention.pdf\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"llx-...\""
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 79ae653c-4598-4bd0-ba6e-b3dab7eab57e\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"result = await LlamaParse().aparse(\"./attention.pdf\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"1 Introduction\n",
"Recurrent neural networks, long short-term memory [13] and gated recurrent [7] neural networks\n",
"in particular, have been firmly established as state of the art approaches in sequence modeling and\n",
"transduction problems such as language modeling and machine translation [35, 2, 5]. Numerous\n",
"efforts have since continued to push the boundaries of recurrent language models and encoder-decoder\n",
"architectures [38, 24, 15].\n",
"Recurrent models typically factor computation along the symbol positions of the input and output\n",
"sequences. Aligning the positions to steps in computation time, they generate a sequence of hidden\n",
"states ht, as a function of the previous hidden state ht1 and the input for position t. This inherently\n",
"sequential nature precludes parallelization within training examples, which becomes critical at longer\n",
"sequence lengths, as memory constraints limit batching across examples. Recent work has achieved\n",
"significant improvements in computational efficiency through factorization tricks [21] and conditional\n",
"computation [32], while also improving model performance in case of the latter. The fundamental\n",
"constraint of sequential computation, however, remains.\n",
"Attention mechanisms have become an integral part of compelling sequence modeling and transduc-\n",
"tion models in various tasks, allowing modeling of dependencies without regard to their distance in\n",
"the input or output sequences [2, 19]. In all but a few cases [27], however, such attention mechanisms\n",
"are used in conjunction with a recurrent network.\n",
"In this work we propose the Transformer, a model architecture eschewing recurrence and instead\n",
"relying entirely on an attention mechanism to draw global dependencies between input and output.\n",
"The Transformer allows for significantly more parallelization and can reach a new state of the art in\n",
"translation quality after being trained for as little as twelve hours on eight P100 GPUs.\n",
"2 Background\n",
"The goal of reducing sequential computation also forms the foundation of the Extended Neural GPU\n",
"[16], ByteNet [18] and ConvS2S [9], all of which use convolutional neural networks as basic building\n",
"block, computing hidden representations in parallel for all input and output positions. In these models,\n",
"the number of operations required to relate signals from two arbitrary input or output positions grows\n",
"in the distance between positions, linearly for ConvS2S and logarithmically for ByteNet. This makes\n",
"it more difficult to learn dependencies between distant positions [12]. In the Transformer this is\n",
"reduced to a constant number of operations, albeit at the cost of reduced effective resolution due\n",
"to averaging attention-weighted positions, an effect we counteract with Multi-Head Attention as\n",
"described in section 3.2.\n",
"Self-attention, sometimes called intra-attention is an attention mechanism relating different positions\n",
"of a single sequence in order to compute a representation of the sequence. Self-attention has been\n",
"used successfully in a variety of tasks including reading comprehension, abstractive summarization,\n",
"textual entailment and learning task-independent sentence representations [4, 27, 28, 22].\n",
"End-to-end memory networks are based on a recurrent attention mechanism instead of sequence-\n",
"aligned recurrence and have been shown to perform well on simple-language question answering and\n",
"language modeling tasks [34].\n",
"To the best of our knowledge, however, the Transformer is the first transduction model relying\n",
"entirely on self-attention to compute representations of its input and output without using sequence-\n",
"aligned RNNs or convolution. In the following sections, we will describe the Transformer, motivate\n",
"self-attention and discuss its advantages over models such as [17, 18] and [9].\n",
"3 Model Architecture\n",
"Most competitive neural sequence transduction models have an encoder-decoder structure [5, 2, 35].\n",
"Here, the encoder maps an input sequence of symbol representations (x1, ..., xn) to a sequence\n",
"of continuous representations z = (z1, ..., zn). Given z, the decoder then generates an output\n",
"sequence (y1, ..., ym) of symbols one element at a time. At each step the model is auto-regressive\n",
"[10], consuming the previously generated symbols as additional input when generating the next.\n",
" 2\n"
]
}
],
"source": [
"documents = result.get_text_documents(split_by_page=True)\n",
"print(documents[1].text)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"arXiv:1706.03762v7 [cs.CL] 2 Aug 2023\n",
"\n",
"Provided proper attribution is provided, Google hereby grants permission to reproduce the tables and figures in this paper solely for use in journalistic or scholarly works.\n",
"\n",
"# Attention Is All You Need\n",
"\n",
"Ashish Vaswani Noam Shazeer Niki Parmar Jakob Uszkoreit\n",
"\n",
"Google Brain Google Brain Google Research Google Research\n",
"\n",
"avaswani@google.com noam@google.com nikip@google.com usz@google.com\n",
"\n",
"Llion Jones Aidan N. Gomez † Łukasz Kaiser\n",
"\n",
"Google Research University of Toronto Google Brain\n",
"\n",
"llion@google.com aidan@cs.toronto.edu lukaszkaiser@google.com\n",
"\n",
"Illia Polosukhin ‡\n",
"\n",
"illia.polosukhin@gmail.com\n",
"\n",
"# Abstract\n",
"\n",
"The dominant sequence transduction models are based on complex recurrent or convolutional neural networks that include an encoder and a decoder. The best performing models also connect the encoder and decoder through an attention mechanism. We propose a new simple network architecture, the Transformer, based solely on attention mechanisms, dispensing with recurrence and convolutions entirely. Experiments on two machine translation tasks show these models to be superior in quality while being more parallelizable and requiring significantly less time to train. Our model achieves 28.4 BLEU on the WMT 2014 English-to-German translation task, improving over the existing best results, including ensembles, by over 2 BLEU. On the WMT 2014 English-to-French translation task, our model establishes a new single-model state-of-the-art BLEU score of 41.8 after training for 3.5 days on eight GPUs, a small fraction of the training costs of the best models from the literature. We show that the Transformer generalizes well to other tasks by applying it successfully to English constituency parsing both with large and limited training data.\n",
"\n",
"Equal contribution. Listing order is random. Jakob proposed replacing RNNs with self-attention and started the effort to evaluate this idea. Ashish, with Illia, designed and implemented the first Transformer models and has been crucially involved in every aspect of this work. Noam proposed scaled dot-product attention, multi-head attention and the parameter-free position representation and became the other person involved in nearly every detail. Niki designed, implemented, tuned and evaluated countless model variants in our original codebase and tensor2tensor. Llion also experimented with novel model variants, was responsible for our initial codebase, and efficient inference and visualizations. Lukasz and Aidan spent countless long days designing various parts of and implementing tensor2tensor, replacing our earlier codebase, greatly improving results and massively accelerating our research.\n",
"\n",
"†Work performed while at Google Brain.\n",
"\n",
"‡Work performed while at Google Research.\n",
"\n",
"31st Conference on Neural Information Processing Systems (NIPS 2017), Long Beach, CA, USA.\n"
]
}
],
"source": [
"documents = result.get_markdown_documents(split_by_page=True)\n",
"print(documents[0].text)"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
-553
View File
@@ -1,553 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "d27f1082-cd10-405e-9570-6f0e934bba8b",
"metadata": {},
"source": [
"# LlamaParse `JobResult` Tour\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/demo_json.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"The `JobResult` object is the main object returned by the LlamaParse API. It contains all the information about the job, including the parsed data, metadata, and any errors.\n",
"\n",
"This notebook walks through each component of the `JobResult` object and shows you what it contains."
]
},
{
"cell_type": "markdown",
"id": "a004db48-8d3f-421c-915a-477692f71b90",
"metadata": {},
"source": [
"## Setup\n",
"\n",
"Let's bring in our imports and set up our API keys."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "bc6a7a4b-b568-4db5-bcba-62f5c517ff3a",
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-cloud-services"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0879301c-ff91-4431-941a-6c0ef7cd8fe2",
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"# API access to llama-cloud\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"llx-..\""
]
},
{
"cell_type": "markdown",
"id": "b411d2ee-3e6b-45b0-b532-4a8e3abcdea0",
"metadata": {},
"source": [
"## Load Data\n",
"\n",
"Let's load a large and complex PDF, San Francisco's 2023 proposed budget."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c39d408f-e885-4940-85c7-b09ca3bc7cb7",
"metadata": {},
"outputs": [],
"source": [
"!wget 'https://www.dropbox.com/scl/fi/vip161t63s56vd94neqlt/2023-CSF_Proposed_Budget_Book_June_2023_Master_Web.pdf?rlkey=hemoce3w1jsuf6s2bz87g549i&dl=0' -O './san_francisco_budget_2023.pdf'"
]
},
{
"cell_type": "markdown",
"id": "c2f42af8-afb3-4b3b-82d3-6b332fb38aa4",
"metadata": {},
"source": [
"## Using LlamaParse for Basic PDF Parsing\n",
"\n",
"Let's parse our document!"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "9c9cd670-8229-4ad6-99a9-845bd82b7ec1",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id d12d419a-52fc-400c-9f88-f61b352d3fb2\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse()\n",
"result = await parser.aparse(\"./san_francisco_budget_2023.pdf\")"
]
},
{
"cell_type": "markdown",
"id": "11c22bab",
"metadata": {},
"source": [
"Every job will come back with some metadata about the job:"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c588c578",
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"JobMetadata(job_credits_usage=0, job_pages=0, job_auto_mode_triggered_pages=0, job_is_cache_hit=True)"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"result.job_metadata"
]
},
{
"cell_type": "markdown",
"id": "1e96b7c9",
"metadata": {},
"source": [
"Since this was a re-run, I can see that a cache hit occurred. Jobs are cached for 48 hours by default."
]
},
{
"cell_type": "markdown",
"id": "6543d2c6",
"metadata": {},
"source": [
"Beyond this, we can explore the parsed data per-page:"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "af9f3717",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"362\n"
]
}
],
"source": [
"print(len(result.pages))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f8845fac",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"dict_keys(['page', 'text', 'md', 'images', 'charts', 'tables', 'layout', 'items', 'status', 'links', 'width', 'height', 'triggeredAutoMode', 'parsingMode', 'structuredData', 'noStructuredContent', 'noTextContent'])\n"
]
}
],
"source": [
"print(result.pages[0].model_dump().keys())"
]
},
{
"cell_type": "markdown",
"id": "6261f5e3",
"metadata": {},
"source": [
"Inside the page object, you can see nearly every detail about the page.\n",
"\n",
"Most of these will depend on the settings you used when parsing. Since we used the default settings, we get the text and markdown for each page, as well as a list of all the elements on the page.\n",
"\n",
"* `page`: this is simply the page number, starting at 1.\n",
"* `text`: this is the text of the page, as extracted by the parser.\n",
"* `images`: this is an array of all the images on the page, including metadata and text OCRed out of the images, as well as a full-page screenshot of the entire page.\n",
"* `charts`: this is an array of all the charts on the page, including metadata and text OCRed out of the charts, as well as a full-page screenshot of the entire chart.\n",
"* `layout`: this is an array of all the layout elements on the page, if you are using layout mode.\n",
"* `items`: This is an array of all the parsed elements on the page, as used to render the markdown, but separated out into their own objects. This is useful if you want to do more processing on the data.\n",
"* `links`: this is an array of all the links on the page, if you are used `annotate_links=True`\n",
"* `status`: this is the status of the page, which is usually \"OK\" unless there was an error processing the page.\n",
"* `width` and `height`: these are the dimensions of the page in pixels.\n",
"* `parsingMode`: Contains the specific parsing mode that was used for the page.\n",
"* `triggeredAutoMode`: this indicates whether the page triggered auto mode; see [LlamaParse docs](https://docs.cloud.llamaindex.ai/llamaparse/getting_started) for more details.\n",
"* `structuredData`/`noStructuredContent`: these are set if you are using structured mode; see [LlamaParse docs](https://docs.cloud.llamaindex.ai/llamaparse/getting_started) for more details.\n",
"* `noTextContent`: this is true if the page was empty of text.\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "7a4cc901",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
" CITY & COUNTY OF SAN FRANCISCO, CALIFORNIA\n",
" PROPOSED BUDGET\n",
" FISCAL YEARS 2023-2024 & 2024-2025\n",
" LONDON N. BREED\n",
" MAYORS OFFICE OF PUBLIC POLICY AND FINANCE\n",
" Anna Duning, Director of Mayors Fisher Zhu, Fiscal and Policy Analyst\n",
" Office of Public Policy and Finance Anya Shutovska, Fiscal and Policy Analyst\n",
" Sally Ma, Deputy Budget Director\n",
"Radhika Mehlotra, Senior Fiscal and Policy Analyst Jack English, Fiscal and Policy Analyst\n",
" Damon Daniels, Fiscal and Policy Analyst Xang Hang, Junior Fiscal and Policy Analyst\n",
" Matthew Puckett, Fiscal and Policy Analyst Tabitha Romero-Bothi, Fiscal and Policy Assistant\n"
]
}
],
"source": [
"print(result.pages[0].text[:1000])"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2d5a5bc2",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# CITY & COUNTY OF SAN FRANCISCO, CALIFORNIA\n",
"\n",
"# PROPOSED BUDGET\n",
"\n",
"# FISCAL YEARS 2023-2024 & 2024-2025\n",
"\n",
"# LONDON N. BREED\n",
"\n",
"# MAYORS OFFICE OF PUBLIC POLICY AND FINANCE\n",
"\n",
"Anna Duning, Director of Mayors Office of Public Policy and Finance\n",
"\n",
"Fisher Zhu, Fiscal and Policy Analyst\n",
"\n",
"Anya Shutovska, Fiscal and Policy Analyst\n",
"\n",
"Sally Ma, Deputy Budget Director\n",
"\n",
"Radhika Mehlotra, Senior Fiscal and Policy Analyst\n",
"\n",
"Jack English, Fiscal and Policy Analyst\n",
"\n",
"Damon Daniels, Fiscal and Policy Analyst\n",
"\n",
"Xang Hang, Junior Fiscal and Policy Analyst\n",
"\n",
"Matthew Puckett, Fiscal and Policy Analyst\n",
"\n",
"Tabitha Romero-Bothi, Fiscal and Policy Assistant\n"
]
}
],
"source": [
"print(result.pages[0].md[:1000])"
]
},
{
"cell_type": "markdown",
"id": "32de4c62",
"metadata": {},
"source": [
"## Images\n",
"\n",
"By default, images embedded in documents that can be extracted are part of the result object."
]
},
{
"cell_type": "markdown",
"id": "802d4a98",
"metadata": {},
"source": [
"We can also specify to take screenshots of every page:"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2ee78f2f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id e6332422-803b-404d-8d0d-ad510fa56c09\n",
"..."
]
}
],
"source": [
"parser = LlamaParse(take_screenshot=True)\n",
"result = await parser.aparse(\"./san_francisco_budget_2023.pdf\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "fab32886",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"[ImageItem(name='page_1.jpg', height=792.0, width=612.0, x=0.0, y=0.0, original_width=1236, original_height=1600, type='full_page_screenshot')]\n"
]
}
],
"source": [
"print(result.pages[0].images)"
]
},
{
"cell_type": "markdown",
"id": "9eba9e52",
"metadata": {},
"source": [
"We can download images (either their bytes or to a local file) using the `JobResult` object as well!"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "a7aa0a29",
"metadata": {},
"outputs": [],
"source": [
"# single image\n",
"image_data = await result.aget_image_data(result.pages[0].images[0].name)\n",
"\n",
"# save an image to a file\n",
"output_path = await result.asave_image(\n",
" result.pages[0].images[0].name, \"./json_tour_screenshots\"\n",
")\n",
"\n",
"# save all images\n",
"output_paths = await result.asave_all_images(\"./json_tour_screenshots\")"
]
},
{
"cell_type": "markdown",
"id": "eae4ece3",
"metadata": {},
"source": [
"## Items\n",
"\n",
"This is an array of all the parsed elements on the page, as used to render the markdown, but separated out into their own objects. This is useful if you want to do more processing on the data. Let's take a look:"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c10b9d7d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"type='heading' lvl=1 value='CITY & COUNTY OF SAN FRANCISCO, CALIFORNIA' md='# CITY & COUNTY OF SAN FRANCISCO, CALIFORNIA' rows=None bBox=BBox(x=176.0, y=52.0, w=277.0, h=12.0)\n"
]
}
],
"source": [
"import json\n",
"\n",
"print(result.pages[0].items[0])"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "dcb9f832",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"type='heading' lvl=1 value='PROPOSED BUDGET' md='# PROPOSED BUDGET' rows=None bBox=BBox(x=89.0, y=118.0, w=451.0, h=47.0)\n"
]
}
],
"source": [
"print(result.pages[0].items[1])"
]
},
{
"cell_type": "markdown",
"id": "a7f64443",
"metadata": {},
"source": [
"As you can see you get different element types: text, headings, and tables. Each comes with its own `md` key containing a Markdown representation of that element, allowing you to easily summarize with only headings, tables only, etc..\n",
"\n",
"The ability to extract tables from visual data is really powerful. Let's take a look at page 35, which has some bar charts that get automatically converted into tables:\n",
"\n",
"<img src=\"./json_tour_screenshots/page_35.png\" alt=\"Page 35\" width=\"300\"/>\n"
]
},
{
"cell_type": "markdown",
"id": "e4ccee76",
"metadata": {},
"source": [
"The bar chart has been converted into a table, and even though explicit values are not included, the bar chart has been read and approximate values for each bar on the chart have been included!"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "7d6404a5",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"type='table' lvl=None value=None md=\"Source: U.S. Census Bureau, 2017-2021 American Community Survey 5-years Estimate.\\n|Race|Educational Level|Number of Residents| | | | |\\n|---|---|---|---|---|---|---|\\n|Age Group| | | | | | |\\n|Under 5 Years|5 to 19 Years|20 to 34 Years|35 to 59 Years|60 and Over| | |\\n|Graduate or professional degree|Bachelor's degree|Associate's degree|Some college, no degree|High school graduate (includes equivalency)|9th to 12th grade, no diploma|Less than 9th grade|\" rows=[[], ['Race', 'Educational Level', 'Number of Residents', '', '', '', ''], ['---', '---', '---', '---', '---', '---', '---'], ['Age Group', '', '', '', '', '', ''], ['Under 5 Years', '5 to 19 Years', '20 to 34 Years', '35 to 59 Years', '60 and Over', '', ''], ['Graduate or professional degree', \"Bachelor's degree\", \"Associate's degree\", 'Some college, no degree', 'High school graduate (includes equivalency)', '9th to 12th grade, no diploma', 'Less than 9th grade']] bBox=BBox(x=68.0, y=129.0, w=613.0, h=3067.0)\n"
]
}
],
"source": [
"print(result.pages[34].items[6])"
]
},
{
"cell_type": "markdown",
"id": "9570d3b8",
"metadata": {},
"source": [
"### `links`\n",
"\n",
"Our budget PDF doesn't have any links, so let's load a different PDF with links and see what we get.\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "fb0da11a",
"metadata": {},
"outputs": [],
"source": [
"!wget 'https://www.dropbox.com/scl/fi/hay06lyxc49gkuh91oek6/basic-link-1.pdf?rlkey=uije7yb0lxqgqwk7p7hnqepdx&dl=0' -O './basic-link-1.pdf'"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e7e393e6",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 9b2df975-af3c-4868-99e2-520ce0b21f4d\n"
]
}
],
"source": [
"parser = LlamaParse(annotate_links=True)\n",
"result = await parser.aparse(\"./basic-link-1.pdf\")"
]
},
{
"cell_type": "markdown",
"id": "701ada4b",
"metadata": {},
"source": [
"This is a very simple document with some internal and external links:\n",
"\n",
"<img src=\"./json_tour_screenshots/links_page.png\" alt=\"Page 1\" width=\"300\"/>\n"
]
},
{
"cell_type": "markdown",
"id": "2e4de7de",
"metadata": {},
"source": [
"The parser finds the external links and their labels and includes them in the `links` section:"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "29bf7e3c",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"[{'url': 'https://www.antennahouse.com/', 'text': 'Antenna House, Inc.'}, {'url': 'https://www.antennahouse.com/', 'text': 'Linking to a website (https://www.antennahouse.com/)'}]\n"
]
}
],
"source": [
"print(result.pages[0].links)"
]
},
{
"cell_type": "markdown",
"id": "ac9088a2",
"metadata": {},
"source": [
"This concludes our tour! I hope this makes clear the power of JSON mode and the flexibility it gives you over what parts of your documents you can use."
]
}
],
"metadata": {
"kernelspec": {
"display_name": "Python 3 (ipykernel)",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
-442
View File
@@ -1,442 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "28d15ea5-a3eb-4ee5-9d91-8dbd95e53129",
"metadata": {},
"source": [
"# Multi-Language Support in LlamaParse\n",
"\n",
"LlamaParse supports users to specify a `language` parameter before uploading documents, giving users better OCR capabilities over non-English PDFs, parsing images into more accurate representations.\n",
"\n",
"You can specify 80+ different languages: see this file for a full list of supported languages: https://github.com/run-llama/llama_cloud_services/blob/main/llama_parse/base.py.\n",
"\n",
"This notebook shows a demo of this in action. "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "15539193-2f5c-4ecf-9ca4-9aee6f888468",
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-index llama-parse"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "87322210-c21c-43d6-b459-2e8a828ac576",
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"llx-...\""
]
},
{
"cell_type": "markdown",
"id": "2b5cabdf-342a-42d2-8ad4-0ba7c46cdfb9",
"metadata": {},
"source": [
"## Load in a French PDF\n",
"\n",
"We load in the 2022 annual report from Agence France Tresor."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e81e0a08-3a99-42e6-adcc-00bb4ce1c3d4",
"metadata": {},
"outputs": [],
"source": [
"!wget \"https://www.dropbox.com/scl/fi/fxg17log5ydwoflhxmgrb/treasury_report.pdf?rlkey=mdintk0o2uuzkple26vc4v6fd&dl=1\" -O treasury_report.pdf"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ecfc578c-3c7f-4ec1-aa06-51565c28632b",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 476966e1-9e04-49e7-a5dc-952b053b8b94\n",
"......"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(language=\"fr\")\n",
"result = await parser.aparse(\"./treasury_report.pdf\")\n",
"documents = result.get_text_documents(split_by_page=False)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0c37db27-3496-4a59-918b-701c9ad7706d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
" ET GESTION DE LA DETTE DE L’ÉTAT\n",
" P.56 FOCUS OAT VERTES\n",
" P.60 CONTRÔLE DES RISQUES & POST-MARCHÉ\n",
" Chiffres de lexercice 2022 P.64 À 105\n",
" P.65 ACTIVITÉ DE LAFT\n",
" P.84 RAPPORT STATISTIQUE\n",
" FICHES TECHNIQUES GLOSSAIRES LISTE DES ABRÉVIATIONS\n",
" P.106 P.118 P.122\n",
" AGENCE FRANCE TRÉSOR - RAPPORT DACTIVITÉ 2022 3\n",
"---\n",
" Édito\n",
" 111 Avec une croissance\n",
" de +2,5 %, la France a illustré\n",
" une nouvelle fois sa résilience\n",
" économique face aux chocs.\n",
"4 AGENCE FRANCE TRÉSOR - RAPPORT DACTIVITÉ 2022\n",
"---\n",
" L’économie française en 2022 :\n",
" résilience face aux chocs géopolitiques\n",
" et économiques\n",
" sa résilience économique face aux lors du dernier trimestre de 2022.\n",
"LE DÉBUT DE chocs. Cette croissance a été permise Malgré un climat des affaires impacté\n",
"LANNÉE 2022 grâce à une forte demande intérieure par linflation, le soutien apporté\n",
" alimentée par le dynamisme de aux TPE/PME leur a permis de faire\n",
"SEMBLAIT linvestissement et, en dépit de face aux défis énergétiques tout en\n",
" linflation, dune résilience de la préservant lemploi.\n",
"ENGAGÉ DANS consommation des ménages sur une\n",
" grande partie de lannée. Afin de combattre linflation qui a\n",
"UNE DYNAMIQUE largement dépassé la cible de 2 %,\n",
" Le taux dinflation des prix à la la BCE, de concert avec les banques\n",
"EFFICACE DE consommation français est resté lun centrales des principales économies\n",
"SORTIE DE CRISE des plus bas dEurope avec +6,0 % développées, a adapté sa fonction de\n",
" en 2022, sappuyant, dune part, sur réaction en mettant fin aux politiques\n",
"PORTÉE PAR latout structurel que représente un dassouplissement monétaire quelle\n",
" mix énergétique parmi les moins menait depuis la crise financière de\n",
"UNE REPRISE exposés à la Russie et, dautre part, 2008. Ainsi, dès juillet 2022, et pour\n",
" sur les politiques proactives du la première fois en 10 ans, la BCE a\n",
"ÉCONOMIQUE gouvernement avec la mise en place augmenté ses taux directeurs. Les\n",
" du bouclier tarifaire, de la remise taux demprunts de l’État à 10 ans se\n",
"INÉDITE carburant et du chèque énergie. sont ainsi progressivement éloignés\n",
"AMORCÉE Ces dispositifs, temporaires, ont de leur territoire négatif pour\n",
" été progressivement supprimés : la atteindre 3,10 % en fin dannée.\n",
"EN 2021. remise carburant, dabord prolongée\n",
" jusqu’à mi-novembre a pris fin Cette décision sest également\n",
"Le déclenchement de la guerre en en décembre 2022, tandis que le accompagnée de la fin du\n",
"Ukraine par la Russie dès février a chèque énergie exceptionnel a pris programme dachat durgence (PEPP)\n",
"rebattu les cartes de cet équilibre, fin en mars 2023. mis en place pendant la pandémie,\n",
"provoquant des bouleversements suivi de la réduction progressive de\n",
"majeurs sur les plans géopolitiques et Le marché du travail français a par son bilan, à un rythme mensuel de 15\n",
"économiques, avec le déploiement ailleurs montré toute sa robustesse, milliards deuros par mois.\n",
"de sanctions à lencontre de la Russie la dynamique de reprise initiée en\n",
"et une forte poussée inflationniste. 2021 ainsi que leffet des réformes LAgence France Trésor a fait face à ce\n",
"Face à cette situation, les principales structurelles engagées les années contexte de grands bouleversements\n",
"banques centrales mondiales, dont précédentes permettant au taux géopolitiques, économiques et\n",
"la Banque centrale européenne demploi des Français âgés de 15 à 64 financiers en sappuyant sur ses\n",
"(BCE), ont engagé une politique de ans datteindre fin 2022 un niveau principes de régularité, de prévisibilité\n",
"normalisation monétaire rapide de 68,1 %, un record depuis 1975. et de transparence. Cette stratégie\n",
"pour lutter contre linflation. La reprise économique de début sest de nouveau révélée robuste et,\n",
"Parallèlement, le gouvernement dannée et les effets positifs du plan alliée à lengagement et à lefficacité\n",
"français a mis en place des mesures France Relance ont permis la création de ses équipes, ainsi qu’à la qualité\n",
"(à hauteur de 43,6 milliards deuros de 337 100 emplois, essentiellement de crédit de la signature de la France,\n",
"sur lannée 2022) pour protéger les dans le secteur salarié marchand. Ce lui a permis daccomplir sa mission\n",
"entreprises et les ménages. dynamisme a aussi conduit à la chute de financement de laction publique\n",
" du taux de chômage, atteignant son au bénéfice de tous.\n",
"Avec une croissance de +2,5 %, la niveau le plus bas depuis mars 2008\n",
"France a illustré une nouvelle fois avec 7,2 % de demandeurs demploi\n",
" Emmanuel Moulin\n",
" DIRECTEUR GÉNÉRAL DU TRÉSOR\n",
" ET PRÉSIDENT DE LAFT\n",
" AGENCE FRANCE TRÉSOR - RAPPORT DACTIVITÉ 2022 5\n",
"---\n",
" du directeur général Le mot\n",
" 011 En 2022, le choc dinflation\n",
" et la normalisation\n",
" de la politique monétaire\n",
" ont mis fin à une décennie\n",
" de taux historiquement bas.\n",
"6 AGENCE FRANCE TRÉSOR - RAPPORT DACTIVITÉ 2022\n",
"---\n",
" MALGRÉ UN CONTEXTE DE MARCHÉ MOUVEMENTÉ ET LES MESURES DAMPLEUR\n",
" PRISES POUR LIMITER LIMPACT DE LINFLATION SUR LES MÉNAGES ET\n",
" LES ENTREPRISES, LE PROGRAMME DE FINANCEMENT À MOYEN ET LONG TERME\n",
" EST DEMEURÉ INCHANGÉ À 260 MILLIARDS DEUROS, STABLE PAR RAPPORT\n",
" À 2021, ET LA DETTE DE COURT TERME A ÉTÉ RÉDUITE DE 7 MILLIARDS DEUROS.\n",
"En janvier 2022, la normalisation de dobligations indexées sur linflation, la dette de court terme a été réduite\n",
"la politique monétaire en zone euro sur lequel a été enregistré un de 7 milliards deuros. En effet, le\n",
"était une perspective de moyen supplément dindexation supérieur dynamisme des recettes fiscales et\n",
"terme. Quelques semaines plus tard, de 17 milliards deuros à celui de la trésorerie levée lors de la crise\n",
"linvasion de lUkraine par la Russie lannée 2021. Il sest également sanit\n"
]
}
],
"source": [
"print(documents[0].get_content()[1000:10000])"
]
},
{
"cell_type": "markdown",
"id": "be161577-7b1e-4710-b721-f549feb8e6d0",
"metadata": {},
"source": [
"## Download Chinese PDF"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ac332ea3-cfff-4216-b292-62410a26c336",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"--2024-02-28 16:41:26-- https://www.dropbox.com/scl/fi/g5ojyzk4m44hl7neut6vc/chinese_pdf.pdf?rlkey=45reu51kjvdvic6zucr8v9sh3&dl=1\n",
"Resolving www.dropbox.com (www.dropbox.com)... 162.125.13.18\n",
"Connecting to www.dropbox.com (www.dropbox.com)|162.125.13.18|:443... connected.\n",
"HTTP request sent, awaiting response... 302 Found\n",
"Location: https://uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com/cd/0/inline/COJ69Wg2e7wH9S0ELzl4j4znoonRSQS-JJrH6mxy_vcrvY-KV7f10kMyQH6IYmtfMh_9xcDNOYnLkWkwMTYItwE1XQB5nqXbjmLJ4jLbDrMeu7-b49m796ctxevwnp7k1_U/file?dl=1# [following]\n",
"--2024-02-28 16:41:27-- https://uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com/cd/0/inline/COJ69Wg2e7wH9S0ELzl4j4znoonRSQS-JJrH6mxy_vcrvY-KV7f10kMyQH6IYmtfMh_9xcDNOYnLkWkwMTYItwE1XQB5nqXbjmLJ4jLbDrMeu7-b49m796ctxevwnp7k1_U/file?dl=1\n",
"Resolving uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com (uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com)... 162.125.13.15\n",
"Connecting to uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com (uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com)|162.125.13.15|:443... connected.\n",
"HTTP request sent, awaiting response... 302 Found\n",
"Location: /cd/0/inline2/COKEp-d6ZqzrIIaPRlanov72wwnd7GX5eNSPnsxug0A8pOpek8hO6eFxp84cY3_NMBRsAqtX-IIVPpcfYHNoV__mpu1SsOV8wV8a68DwVKaVJRJriY_KV8lEFocvLgf7c7mhrREbIJ1UBN2fx6S_qWegwVIen1z1-pw-K7icMnA3EKJNqM9DFtqx9ct0FI4vdYGsv8ckLF26WgAhs96k1cHn-VRJle4SKstdYs8EmBxiuFLXZRCL3gljwAsLu3J6WRvis9v7VJ2zNhgrcT-ZnVujlpQGoGWLLPmREKffK608Xfz1XE35DzO28e_mm4SUPRfsP2mvIUrJUtUrhobR4siqQRGojxi0S7-da4Y7fpB4Tw/file?dl=1 [following]\n",
"--2024-02-28 16:41:27-- https://uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com/cd/0/inline2/COKEp-d6ZqzrIIaPRlanov72wwnd7GX5eNSPnsxug0A8pOpek8hO6eFxp84cY3_NMBRsAqtX-IIVPpcfYHNoV__mpu1SsOV8wV8a68DwVKaVJRJriY_KV8lEFocvLgf7c7mhrREbIJ1UBN2fx6S_qWegwVIen1z1-pw-K7icMnA3EKJNqM9DFtqx9ct0FI4vdYGsv8ckLF26WgAhs96k1cHn-VRJle4SKstdYs8EmBxiuFLXZRCL3gljwAsLu3J6WRvis9v7VJ2zNhgrcT-ZnVujlpQGoGWLLPmREKffK608Xfz1XE35DzO28e_mm4SUPRfsP2mvIUrJUtUrhobR4siqQRGojxi0S7-da4Y7fpB4Tw/file?dl=1\n",
"Reusing existing connection to uc7a03fdb7d960dbedb23e9298ab.dl.dropboxusercontent.com:443.\n",
"HTTP request sent, awaiting response... 200 OK\n",
"Length: 8074860 (7.7M) [application/binary]\n",
"Saving to: chinese_pdf.pdf\n",
"\n",
"chinese_pdf.pdf 100%[===================>] 7.70M 37.9MB/s in 0.2s \n",
"\n",
"2024-02-28 16:41:28 (37.9 MB/s) - chinese_pdf.pdf saved [8074860/8074860]\n",
"\n"
]
}
],
"source": [
"!wget \"https://www.dropbox.com/scl/fi/g5ojyzk4m44hl7neut6vc/chinese_pdf.pdf?rlkey=45reu51kjvdvic6zucr8v9sh3&dl=1\" -O chinese_pdf.pdf"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "45235b17-08f0-48f1-92aa-06711225860b",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 0089f0b6-29ee-4e94-a8bf-49a137666f15\n",
".........."
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(language=\"ch_sim\")\n",
"result = await parser.aparse(\"./chinese_pdf.pdf\")\n",
"documents = result.get_text_documents(split_by_page=False)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f0d546cc-6549-4cf5-8b37-0896f4e8d43d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"中国投资有限责任公司2022年度报告 5\n",
"---\n",
"企业文化与核心价值观\n",
"使命 核心价值观\n",
" 致力于实现国家外汇资金多元化投资,在可接受风险范围内 责任 合力\n",
" 实现股东权益最大化,以服务于国家经济发展和深化金融体\n",
" 制改革的需要 忠于使命、勤勉尽责 立足大局、有效协同\n",
" 是公司遵奉的核心价值取向 是实现公司可持续发展的关键\n",
" 愿景 专业 进取\n",
" 成为受人尊重的国际一流主权财富基金 坚持良好的专业精神和职业操守 求知进取、追求卓越\n",
" 是公司成功的基石 是公司成功和发展壮大的内驱力\n",
"---\n",
"01 我们将一以贯之地践行全球发展倡议,充分维护投资东道国利益,\n",
" 积极投身可持续投资,助力世界经济实现更高质量、更有韧性的发展。\n",
" 致 辞\n",
" 3 中国投资有限责任公司2022年度报告 中国投资有限责任公司2022年度报告 4\n",
"---\n",
" “行之力则知愈进,知之深则行愈达。”站在新的历史起点上,中投公司\n",
" 将继续秉承精益求精、追求卓越的专业精神,与国内外合作伙伴一起深化\n",
" 合作,共聚力量、共迎挑战、共享成果,开启打造世界一流主权财富基金\n",
" 的新篇章,为助力全球经济发展作出新贡献! #Ave彭纯\n",
" 董事长\n",
" 2022年,是中投公司成立十五周年。\n",
"董事长致辞 自2007年成立以来,中投公司坚守长期机构投资者定位,坚持国际化、市场化、专业化、负责任原则,搭\n",
" 建起符合大型国际投资机构特点的治理架构,形成了系统完备的投资管理体系,经受住了国际金融危机、世纪\n",
" 疫情等多个历史罕见的风险与挑战。如今,公司对外投资业务覆盖国际市场主要资产类别以及全球110多个国家\n",
" 和地区,培养了一支高素质专业化的投资管理人才队伍,搭建了互利共赢的投资合作“朋友圈”,长期投资收\n",
" 益超越董事会制定的考核目标,为促进国家外汇资产保值增值、服务国内国际双循环作出了积极贡献,在推动\n",
" 全球投资合作、助力世界经济增长中贡献了中投力量,书写了中国主权财富基金不平凡的创业发展史。\n",
"5 中国投资有限责任公司2022年度报告 中国投资有限责任公司2022年度报告 6\n",
"---\n",
" 2022年以来,全球地缘政治风险显著攀升,产业链供应链持续调整重构,美欧央行大幅加息,国际资本 我们守正创新,坚决践行双碳与可持续发展理念。更加包容、更加普惠、更有韧性的发展是全球\n",
"市场剧烈震荡,MSCI全球股票指数、彭博全球债券指数一度自高点下跌超过22%、13%。面对风高浪急的国 可持续发展的关键。我们积极履行负责任投资者理念,制定《关于践行双碳目标和可持续投资行动的意见》,\n",
"际环境和前所未有的巨大挑战,公司保持战略定力,发挥长期机构投资者优势,不断优化资产配置和投资策 积极开展气候变化、能源转型等主题投资。我们发布《运营碳中和行动计划》,明确时间表和路线图,全力实\n",
"略,着力提升总组合韧性,加强重点领域风险防控,年度投资收益跑赢大市;截至2022年底,过去十年对外 现节能减排目标。我们探索以绿色资源引领乡村发展的新方法,在四个定点帮扶县持续推进巩固脱贫成果与乡\n",
"投资年化净收益率按美元计算为6.43%,超出十年业绩目标26个基点;自成立以来累计年化国有资本增值率达 村振兴的有效衔接,助力民生保障与产业扶持,积极履行企业社会责任。\n",
"到12.67%,圆满完成五年战略规划主要目标任务。 面向未来,我们坚信,发展与合作是破解全球性问题的“钥匙”。中投公司将一以贯之地践行全球发展倡\n",
" 我们矢志不渝,积极打造世界一流主权财富基金。长期资本对于促进世界经济持续发展有着不 议,秉持互利共赢理念,以资本为纽带,促进国际产业交流合作,推动世界互联互通;充分维护投资东道国利\n",
"可替代的作用。我们坚持国际化、市场化、专业化、负责任原则,快速恢复常态化对外交流交往,按照互利共 益,与东道国共创价值、共享价值;积极投身可持续投资,推动被投企业履行社会责任,助力世界经济实现更\n",
"赢原则深化与国内外各类机构合作,持续为世界经济发展提供长期资本支持。我们积极创新对外投资方式,稳 高质量、更有韧性的发展。\n",
"健运行多支新型双边基金,新设相关投资合作平台,深入推进中国市场价值创造,促进被投资公司拓展市场空\n",
"间,助推国际投资与产业合作高质量发展。 经济全球化的潮流不可阻挡。我们呼吁各国携起手来,做多边主义的坚定维护者,打造更加开放有序的投\n",
" 资环境,便利资本和资源要素在全球顺畅流动。我们尊重各方的利益关切,在开放中捕捉投资机遇,以务实合\n",
" 我们直面挑战,着力加强自主投资能力建设。面对持续动荡的国际金融市场,我们锚定配置方 作应对共同挑战,并肩前进分享发展红利,推动世界经济平稳运行和持续增长。\n",
"向,强化研究驱动,有序实施组合调整、策略优化,及时调整公开市场投资布局,质量并重推进非公开市场投\n",
"资,完成另类资产投资占比50%的资产配置目标,对外投资总组合的韧性和质量不断提高。我们持续深化投资 “行之力则知愈进,知之深则行愈达。”过去的十五年,是中投人不惧挑战、接续奋斗的十五\n",
"管理体制机制改革,统一非公开市场投资决策制度流程,配强投资决策专职委员并设立支持团队,投资管理科 年。 2023年是中投人落实新一轮战略规划的开局之年。上半年,在风高浪急的国际环境下,中投公司锚定战略目\n",
"学化、专业化水平得到进一步提升。 标,统筹好发展和安全,取得了良好业绩,实现了良好开局。近期,公司部分董事更换,我们对离任董事在指导和支\n",
" 持公司完善公司治理、深化投资管理体制机制改革、应对国际市场风险挑战等方面所作的贡献表示衷心感谢,对新\n",
" 我们勇担使命,坚定走好中国特色金融发展之路。面对新征程新要求,我们坚持发挥“积极股 任董事表示热烈欢迎。站在新的历史起点上,中投公司将完整、准确、全面贯彻新发展理念,积极助力构建新发展格\n",
"东”作用,督促控参股金融企业优化产品服务、加大资源倾斜力度,全力支持稳经济稳增长。我们积极创新完 局,牢牢把握高质量发展首要任务,继续秉承精益求精、追求卓越的专业精神,与国内外合作伙伴一起深化合作,共\n",
"善“汇金模式”,推动优化国有金融资本布局,以市场化方式参与问题金融机构救助,助力金融市场稳定健康 聚力量、共迎挑战、共享成果,开启打造世界一流主权财富基金的新篇章,为助力全球经济发展作出新贡献!\n",
"发展。我们主动适应新形势新要求,围绕国有金融资本管理体系建设等重大课题深入研究,压实派出董事自主\n",
"履职责任,不断提升机构化履职能力。\n",
" 我们坚守底线,持续夯实全面风险管理体系。面对风高浪急的国际环境,我们优化风险管理委员\n",
"会设置,修订全面风险管理基本制度,增加风险类别的覆盖度,全面提升风险预见、应对、处置水平。在对外投\n",
"资方面,我们严守法律合规底线,健全地缘政治、气候变化等非传统风险防控机制,突出抓好流动性管理,对外\n",
"投资总组合风险保持在董事会规定的容忍度内。在国有金融资本受托管理方面,我们建立健全控参股金融企业风\n",
"险监测体系,全面开展多维度风险画像,推动控参股金融企业风险减存量、控增量、防变量取得积极成效。\n",
"7 中国投资有限责任公司2022年度报告 中国投资有限责任公司2022年度报告 8\n",
"---\n",
"02 中投公司的组建宗旨是实现国家外汇资金多元化投资,在可接受风\n",
" 险范围内实现股东权益最大化,以服务于国家宏观经济发展和深化\n",
" 公 司 介 绍 金融体制改革的需要。\n",
" 9 中国投资有限责任公司2022年度报告 中国投资有限责任公司2022年度报告 10\n",
"---\n",
"公司概况中国投资有限责任公司(以下简称“中投公司”)依照《中华人民共和国公司法》(以下简称“《公司 公司治理 中投公司按照《公司法》及《中国投资有限责任公司章程》(以下简称“《中投公司章程》”)中的有关规\n",
"法》”)于2007年9月成立,总部设在北京。中投公司的初始资本金为2000亿美元,由中国财政部发行1.55万 定,设立了董事会、监事会和执行委员会(以下简称“执委会”),三者之间权责明确、独立履职、有效制衡。\n",
"亿元人民币特别国债募集。截至2022年底,公司总资产达1.24万亿美元。 2022年,中投公司健全完善董事会、监事会运行机制,强化下设专门委员会的职能发挥,持续提升公司治\n",
" 中投公司的组建宗旨是实现国家外汇资金多元化投资,在可接受风险范围内实现股东权益最大化,以服务于 理效能。公司根据业务发展需要,优化调整投资管理架构,完善投资决策和投后管理制度机制,深化全面风险管\n",
"国家宏观经济发展和深化金融体制改革的需要。 理体系建设,全面提升机构化投资能力。\n",
" 中投公司开展境外投资业务与境内金融机构股权管理工作。其中,境外投资业务由下设子公司⸺中投国际\n",
"有限责任公司(以下简称“中投国际”)和中投海外直接投资有限责任公司(以下简称“中投海外”)承担,业\n",
"务范围包括公开市场股票和债券投资,对冲基金和多资产,泛行业私募股权和私募信用投资,房地产、基础设\n",
"施、资源商品、农业等领域的基金投资与直接投资,以及多双边基金管理等。 组织架构图\n",
" 中央汇金投资有限责任公司(以下简称“中央汇金”)作为中投公司的子公司,根据国务院授权,对国有重\n",
"点金融企业进行股权投资,以出资额为限代表国家依法对国有重点金融企业行使出资人权利和履行出资人义务。 董事会 监事会\n",
"中央汇金不开展商业性经营活动,不干预其控股的国有重点金融企业的日常经营活动。 提名与\n",
" 薪酬委员会\n",
" 中投国际和中投海外开展的境外业务与中央汇金开展的境内业务之间实行严格的“防火墙”政策和措施。\n",
" 战略与\n",
" 社会责任\n",
" 委员会\n",
" 风险管理 执行 国际咨询 监督 审计\n",
" 委员会 委员会 委员会 委员会 委员会\n",
" 境外投资 管理与支持 境内股权\n",
" 业务部门 部门 管理部门\n",
"11 中国投资有限责任公司2022年度报告 中国投资有限责任公司2022年度报告 12\n",
"---\n",
"董事会 沈如军\n",
" 党委委员、执行董事、副总经理\n",
" 中投公司董事会行使《公司法》和《中投公司章程》中规定的有限责任公司董事会的职权,主要包括:审核 1964年出生,管理学博士,高级会计师。历任中国工商银行计划财务部副总经理、\n",
"和批准公司的发展战略、经营方针和投资计划;确定公司需向股东报告的重大事项;制定公司年度预决算方案; 北京市分行副行长、财务会计部总经理、山东省分行行长,交通银行执行董事、副\n",
"任免公司高级管理人员;决定或授权批准设立内部管理机构等。 行长。现任本公司党委委员、执行董事、副总经理。\n",
" 董事会由执行董事、非执行董事、独立董事以及职工董事构成。 丛亮\n",
" 2022年,面对复杂严峻的国际经济形势,董事会加强对公司重大经营管理事项的指导和督促,及时听取投 非执行董事\n",
"资形势、经营管理、风险防控等汇报,认真审议经营计划、财务预算和决算、业绩考核等重要议题,深入谋划中 1971年出生,经济学博士。历任国家发展和改革委员会国民经济综合司副司长、司\n",
"投公司新一轮战略规划,明确发展目标、基本原则和重点举措,为公司下一阶段改革发展描绘新的蓝图。董事会 长,国家发展和改革委员会秘书长、新闻发言人,国家发展和改革委员会副主任,\n",
"专门委员会根据授权,重点关注关系企业长远发展的重大事项,为董事会出谋划策,推动公司高质量发展迈上新 国家粮食和物资储备局局长。现任国家发展和改革委员会副主任,并兼任本公司非\n",
"台阶。 执行董事。\n",
" 许宏才\n",
" 非执行董事\n",
"董事会成员 1963年出生,经济学学士。历任财政部预算司副司长、司长,财政部部长助理,财\n",
" 政部副部长。现任全国人大财政经济委员会副主任委员、全国人大常委会预算工作\n",
" 彭 纯 \n"
]
}
],
"source": [
"print(documents[0].get_content()[1000:10000])"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "640f0679-7f7e-4b0a-a46d-b099ae382fe2",
"metadata": {},
"outputs": [],
"source": [
"# download another copy with a different name to avoid hitting pdf cache\n",
"!wget \"https://www.dropbox.com/scl/fi/g5ojyzk4m44hl7neut6vc/chinese_pdf.pdf?rlkey=45reu51kjvdvic6zucr8v9sh3&dl=1\" -O chinese_pdf2.pdf"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "bfcacf90-ca67-4bfd-b023-be0af2cb18c5",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 99538f59-24f7-4f1e-ab27-4081933fa5ee\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"base_parser = LlamaParse(language=\"en\")\n",
"result = await base_parser.aparse(\"./chinese_pdf2.pdf\")\n",
"base_documents = result.get_text_documents(split_by_page=False)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "b264ed4e-647a-4f51-9f79-fdf82b76762a",
"metadata": {},
"outputs": [],
"source": [
"print(base_documents[0].get_content()[1000:10000])"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama_parse",
"language": "python",
"name": "llama_parse"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
@@ -1,312 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "97c79c38-38a3-40f3-ba2e-250649347d63",
"metadata": {},
"source": [
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/parse/demo_starter_multimodal.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>"
]
},
{
"cell_type": "markdown",
"id": "4e081457",
"metadata": {},
"source": [
"# Multimodal Parsing using LlamaParse\n",
"\n",
"This cookbook shows you how to use LlamaParse to parse any document with the multimodal capabilities of Multi-Modal LLMs from Anthropic/ OpenAI.\n",
"\n",
"LlamaParse allows you to plug in external, multimodal model vendors for parsing - we handle the error correction, validation, and scalability/reliability for you.\n"
]
},
{
"cell_type": "markdown",
"id": "qOdqBxCS51Ow",
"metadata": {},
"source": [
"### Installation"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "H_Vqcylb50vm",
"metadata": {},
"outputs": [],
"source": [
"!pip install llama-cloud-services"
]
},
{
"cell_type": "markdown",
"id": "15e60ecf-519c-41fc-911b-765adaf8bad4",
"metadata": {},
"source": [
"### Setup\n",
"\n",
"Here we setup `LLAMA_CLOUD_API_KEY` for using `LlamaParse`."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "91a9e532-1454-40e0-bbf0-fd442c350121",
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"# API access to llama-cloud\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"<YOUR LLAMACLOUD API KEY>\""
]
},
{
"cell_type": "markdown",
"id": "LGwBNPNotZRQ",
"metadata": {},
"source": [
"## Download Data\n",
"\n",
"For this demonstration, we will use OpenAI's recent paper `Evaluation of OpenAI o1: Opportunities and Challenges of AGI`."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "IjtKDQRLrylI",
"metadata": {},
"outputs": [],
"source": [
"!wget \"https://arxiv.org/pdf/2409.18486\" -O \"o1.pdf\""
]
},
{
"cell_type": "markdown",
"id": "4e29a9d7-5bd9-4fb8-8ec1-4c128a748662",
"metadata": {},
"source": [
"## Initialize LlamaParse\n",
"\n",
"Initialize LlamaParse in multimodal mode, and specify the vendor.\n",
"\n",
"**NOTE**: optionally you can specify the Anthropic/ OpenAI API key. If you choose to do so LlamaParse will only charge you 1 credit (0.3c) per page. \n",
"\n",
"\n",
"Using your own API key may incur additional costs from your model provider and could result in failed pages or documents if you do not have sufficient usage limits."
]
},
{
"cell_type": "markdown",
"id": "1b5d6da6",
"metadata": {},
"source": [
"### With anthropic-sonnet-3.5"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f2e9d9cf-8189-4fcb-b34f-cde6cc0b59c8",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id dd9d5e0f-160e-486a-89a2-6005e5a1c2ac\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model_name=\"anthropic-sonnet-3.5\",\n",
" target_pages=\"24\"\n",
" # invalidate_cache=True\n",
")\n",
"result = await parser.aparse(\"o1.pdf\")\n",
"nodes = result.get_text_nodes(split_by_page=False)"
]
},
{
"cell_type": "markdown",
"id": "4f3c51b0-7878-48d7-9bc3-02b516500128",
"metadata": {},
"source": [
"### With GPT-4o\n",
"\n",
"For comparison, we will also parse the document using GPT-4o."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6fc3f258-50ae-4988-b904-c105463a498f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 6a4dea44-4f90-406b-b290-9e98620b1232\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser_gpt4o = LlamaParse(\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model=\"openai-gpt4o\",\n",
" target_pages=\"24\",\n",
" # invalidate_cache=True\n",
")\n",
"result = await parser_gpt4o.aparse(\"o1.pdf\")\n",
"nodes = result.get_markdown_nodes(split_by_page=False)"
]
},
{
"cell_type": "markdown",
"id": "44c20f7a-2901-4dd0-b635-a4b33c5664c1",
"metadata": {},
"source": [
"### View Results\n",
"\n",
"Let's visualize the results along with the original document page."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "778698aa-da7e-4081-b3b5-0372f228536f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page: 25\n",
"\n",
"| Participant_ID | clinical Description Reference |\n",
"|-----------------|----------------------------------|\n",
"| Attribute | Value | Basic Personal Information: Subject 098_S_0896 is a 72.0-year-old Female who has completed 15 years of education. The ethnicity is Not Hisp/Latino and race is White. Marital status is Married. Initially diagnosed as AD, as of the date 2007-10-24, the final diagnosis was Dementia. |\n",
"| Age | 72.0 |\n",
"| Sex | Female |\n",
"| Education | 15 |\n",
"| Race | White | Biomarker Measurements: The subject's genetic profile includes an ApoE4 status of 0.0... |\n",
"| DX_bl | AD |\n",
"| DX | Dementia |\n",
"| ... | ... | Cognitive and Neurofunctional Assessments: The Mini-Mental State Examination score stands at 29.0. The Clinical Dementia Rating, sum of boxes, is 1.0. ADAS 11 and 13 scores are 4.67 and 4.67 respectively, with a score of 1.0 in delayed word recall... |\n",
"| APOE4 | 1.0 |\n",
"| TAU | 212.5 |\n",
"| ... | ... |\n",
"| MMSE | 29.0 | Volumetric Data: Under MRI conditions at a field strength of 1.5 Tesla MRI Tesla, using Cross Sectional FreeSurfer (FreeSurfer Version 4.3), the imaging data recorded includes ventricles volume at 54422.0, hippocampus volume at 6677.0, whole brain volume at 1147980.0, entorhinal cortex volume at 2782.0, fusiform gyrus volume at 19432.0, and middle temporal area volume at 24951.0. The intracranial volume measured is 1799580.0.... |\n",
"| CDRSB | 0.0 |\n",
"| ... | ... |\n",
"| FLDSTRENG | 1.5 Tesla MRI |\n",
"| Ventricles | 84599 |\n",
"| Hippocampus | 5319 |\n",
"| ... | ... |\n",
"\n",
"Figure 2: An example of a patient table and its corresponding clinical description.\n",
"\n",
"skills. Mathematics, as a highly structured and logic-driven discipline, provides an ideal testing ground for evaluating this reasoning ability. To investigate o1-preview's performance, we designed a series of tests covering various difficulty levels. We begin with high school-level math competition problems in this section, followed by college-level mathematics problems in the next section, allowing us to observe the model's logical reasoning across varying levels of complexity.\n",
"\n",
"In this section, we selected two primary areas of mathematics: algebra and counting and probability in this section. We chose these two topics because of their heavy reliance on problem-solving skills and their frequent use in assessing logical and abstract thinking [46]. The dataset used in testing is from the MATH dataset [46]. The problems in the dataset cover a wide range of subjects, including Prealgebra, Intermediate Algebra, Algebra, Geometry, Counting and Probability, Number Theory, and Precalculus. Each problem is categorized based on difficulty, ranked from level 1 to 5, according to the Art of Problem Solving (AoPS). The dataset mainly comprises problems from various high school math competitions, including the American Mathematics Competitions (AMC) 10 and 12, as well as the American Invitational Mathematics Examination (AIME), and other similar contests. Each problem comes with detailed reference solutions, allowing for a comprehensive comparison of o1-preview's solutions.\n",
"\n",
"In addition to evaluating the final answers produced by o1-preview, our analysis delves into the step-by-step reasoning process of the o1-preview's solutions. By comparing o1-preview's solutions with the dataset's solutions, we assess its ability to engage in logical reasoning, handle abstract problem-solving tasks, and apply structured approaches to reach correct answers. This deeper analysis offers insights into o1-preview's overall reasoning capabilities, using mathematics as a reliable indicator for logical and structured thought processes.\n"
]
}
],
"source": [
"# using Sonnet-3.5\n",
"print(nodes[0].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "1511a30f-3efc-4142-9668-7dc056a24d0c",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page: 25\n",
"\n",
"\n",
"| Participant_ID | clinical Description Reference |\n",
"|----------------|--------------------------------|\n",
"| **Attribute** | **Value** |\n",
"| Age | 72.0 |\n",
"| Sex | Female |\n",
"| Education | 15 |\n",
"| Race | White |\n",
"| DX_bl | AD |\n",
"| DX | Dementia |\n",
"| ... | ... |\n",
"| APOE4 | 1.0 |\n",
"| TAU | 212.5 |\n",
"| ... | ... |\n",
"| MMSE | 29.0 |\n",
"| CDRSB | 0.0 |\n",
"| ... | ... |\n",
"| FLDSTRENG | 1.5 Tesla MRI |\n",
"| Ventricles | 84599 |\n",
"| Hippocampus | 5319 |\n",
"| ... | ... |\n",
"\n",
"**Basic Personal Information:** Subject 098_S_0896 is a 72.0-year-old Female who has completed 15 years of education. The ethnicity is Not Hisp/Latino and race is White. Marital status is Married. Initially diagnosed as AD, as of the date 2007-10-24, the final diagnosis was Dementia.\n",
"\n",
"**Biomarker Measurements:** The subject's genetic profile includes an ApoE4 status of 0.0...\n",
"\n",
"**Cognitive and Neurofunctional Assessments:** The Mini-Mental State Examination score stands at 29.0. The Clinical Dementia Rating, sum of boxes, is 1.0. ADAS 11 and 13 scores are 4.67 and 4.67 respectively, with a score of 1.0 in delayed word recall...\n",
"\n",
"**Volumetric Data:** Under MRI conditions at a field strength of 1.5 Tesla MRI Tesla, using Cross-Sectional FreeSurfer (FreeSurfer Version 4.3), the imaging data recorded includes ventricles volume at 84422.0, hippocampus volume at 6677.0, whole brain volume at 1147980.0, entorhinal cortex volume at 27820.0, fusiform gyrus volume at 19432.0, and middle temporal area volume at 24951.0. The intracranial volume measured is 1799580.0...\n",
"\n",
"Figure 2: An example of a patient table and its corresponding clinical description.\n",
"\n",
"----\n",
"\n",
"Skills. Mathematics, as a highly structured and logic-driven discipline, provides an ideal testing ground for evaluating this reasoning ability. To investigate o1-previews performance, we designed a series of tests covering various difficulty levels. We begin with high school-level math competition problems in this section, followed by college-level mathematics problems in the next section, allowing us to observe the models logical reasoning across varying levels of complexity.\n",
"\n",
"In this section, we selected two primary areas of mathematics: algebra and counting and probability in this section. We chose these two topics because of their heavy reliance on problem-solving skills and their frequent use in assessing logical and abstract thinking [46]. The dataset used in testing is from the MATH dataset [46]. The problems in the dataset cover a wide range of subjects, including Prealgebra, Intermediate Algebra, Algebra, Geometry, Counting and Probability, Number Theory, and Precalculus. Each problem is categorized based on difficulty, ranked from level 1 to 5, according to the Art of Problem Solving (AoPS). The dataset mainly comprises problems from various high school math competitions, including the American Mathematics Competitions (AMC) 10 and 12, as well as the American Invitational Mathematics Examination (AIME), and other similar contests. Each problem comes with detailed reference solutions, allowing for a comprehensive comparison of o1-previews solutions.\n",
"\n",
"In addition to evaluating the final answers produced by o1-preview, our analysis delves into the step-by-step reasoning process of the o1-previews solutions. By comparing o1-previews solutions with the datasets solutions, we assess its ability to engage in logical reasoning, handle abstract problem-solving tasks, and apply structured approaches to reach correct answers. This deeper analysis offers insights into o1-previews overall reasoning capabilities, using mathematics as a reliable indicator for logical and structured thought processes.\n"
]
}
],
"source": [
"# using GPT-4o\n",
"print(nodes[0].get_content(metadata_mode=\"all\"))"
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "llamacloud",
"language": "python",
"name": "llamacloud"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
@@ -1,516 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Table Extraction with LlamaParse\n",
"\n",
"This notebook will show you how to extract tables and save them as CSV files thanks to LlamaParse advanced parsing capabilities."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**1. Install needed dependencies**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"! pip install llama-cloud-services pandas"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**2. Set you LLAMA_CLOUD_API_KEY as env variable**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"LLAMA_CLOUD_API_KEY: ··········\n"
]
}
],
"source": [
"import os\n",
"from getpass import getpass\n",
"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = getpass(\"LLAMA_CLOUD_API_KEY: \")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**3. Initialiaze the parser**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(result_type=\"markdown\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**4. Get data**\n",
"\n",
"This is a PDF with _lots_ of tables!"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"--2025-07-16 16:20:41-- https://assets.accessible-digital-documents.com/uploads/2017/01/sample-tables.pdf\n",
"Resolving assets.accessible-digital-documents.com (assets.accessible-digital-documents.com)... 3.166.135.2, 3.166.135.62, 3.166.135.51, ...\n",
"Connecting to assets.accessible-digital-documents.com (assets.accessible-digital-documents.com)|3.166.135.2|:443... connected.\n",
"HTTP request sent, awaiting response... 200 OK\n",
"Length: 145494 (142K) [application/pdf]\n",
"Saving to: sample-tables.pdf\n",
"\n",
"sample-tables.pdf 100%[===================>] 142.08K --.-KB/s in 0.04s \n",
"\n",
"2025-07-16 16:20:41 (3.72 MB/s) - sample-tables.pdf saved [145494/145494]\n",
"\n"
]
}
],
"source": [
"! wget https://assets.accessible-digital-documents.com/uploads/2017/01/sample-tables.pdf"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**5. Parse document**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id b53949f7-9017-4b6a-b30c-be6227271ed2\n"
]
}
],
"source": [
"json_result = parser.get_json_result(\"sample-tables.pdf\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**6. Get tables!**"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"tables = parser.get_tables(json_result, \"tables/\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**7. Load tables**\n",
"\n",
"Let's show one example table!"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"application/vnd.google.colaboratory.intrinsic+json": {
"summary": "{\n \"name\": \"display(df\",\n \"rows\": 8,\n \"fields\": [\n {\n \"column\": \"Rainfall\",\n \"properties\": {\n \"dtype\": \"string\",\n \"num_unique_values\": 5,\n \"samples\": [\n \"Average\",\n \"\",\n \"24 hour high\"\n ],\n \"semantic_type\": \"\",\n \"description\": \"\"\n }\n },\n {\n \"column\": \"Americas\",\n \"properties\": {\n \"dtype\": \"number\",\n \"std\": 908,\n \"min\": 9,\n \"max\": 2010,\n \"num_unique_values\": 8,\n \"samples\": [\n 104,\n 133,\n 2010\n ],\n \"semantic_type\": \"\",\n \"description\": \"\"\n }\n },\n {\n \"column\": \"Asia\",\n \"properties\": {\n \"dtype\": \"object\",\n \"num_unique_values\": 7,\n \"samples\": [\n \"\",\n 201.0,\n 28.0\n ],\n \"semantic_type\": \"\",\n \"description\": \"\"\n }\n },\n {\n \"column\": \"Europe\",\n \"properties\": {\n \"dtype\": \"object\",\n \"num_unique_values\": 7,\n \"samples\": [\n \"\",\n 193.0,\n 29.0\n ],\n \"semantic_type\": \"\",\n \"description\": \"\"\n }\n },\n {\n \"column\": \"Africa\",\n \"properties\": {\n \"dtype\": \"object\",\n \"num_unique_values\": 7,\n \"samples\": [\n \"\",\n 144.0,\n 20.0\n ],\n \"semantic_type\": \"\",\n \"description\": \"\"\n }\n }\n ]\n}",
"type": "dataframe"
},
"text/html": [
"\n",
" <div id=\"df-94a74c8f-1062-4a80-8d3f-32f0fbadf7bb\" class=\"colab-df-container\">\n",
" <div>\n",
"<style scoped>\n",
" .dataframe tbody tr th:only-of-type {\n",
" vertical-align: middle;\n",
" }\n",
"\n",
" .dataframe tbody tr th {\n",
" vertical-align: top;\n",
" }\n",
"\n",
" .dataframe thead th {\n",
" text-align: right;\n",
" }\n",
"</style>\n",
"<table border=\"1\" class=\"dataframe\">\n",
" <thead>\n",
" <tr style=\"text-align: right;\">\n",
" <th></th>\n",
" <th>Rainfall</th>\n",
" <th>Americas</th>\n",
" <th>Asia</th>\n",
" <th>Europe</th>\n",
" <th>Africa</th>\n",
" </tr>\n",
" </thead>\n",
" <tbody>\n",
" <tr>\n",
" <th>0</th>\n",
" <td>(inches)</td>\n",
" <td>2010</td>\n",
" <td></td>\n",
" <td></td>\n",
" <td></td>\n",
" </tr>\n",
" <tr>\n",
" <th>1</th>\n",
" <td>Average</td>\n",
" <td>104</td>\n",
" <td>201.0</td>\n",
" <td>193.0</td>\n",
" <td>144.0</td>\n",
" </tr>\n",
" <tr>\n",
" <th>2</th>\n",
" <td>24 hour high</td>\n",
" <td>15</td>\n",
" <td>26.0</td>\n",
" <td>27.0</td>\n",
" <td>18.0</td>\n",
" </tr>\n",
" <tr>\n",
" <th>3</th>\n",
" <td>12 hour high</td>\n",
" <td>9</td>\n",
" <td>10.0</td>\n",
" <td>11.0</td>\n",
" <td>12.0</td>\n",
" </tr>\n",
" <tr>\n",
" <th>4</th>\n",
" <td></td>\n",
" <td>2009</td>\n",
" <td></td>\n",
" <td></td>\n",
" <td></td>\n",
" </tr>\n",
" <tr>\n",
" <th>5</th>\n",
" <td>Average</td>\n",
" <td>133</td>\n",
" <td>244.0</td>\n",
" <td>155.0</td>\n",
" <td>166.0</td>\n",
" </tr>\n",
" <tr>\n",
" <th>6</th>\n",
" <td>24 hour high</td>\n",
" <td>27</td>\n",
" <td>28.0</td>\n",
" <td>29.0</td>\n",
" <td>20.0</td>\n",
" </tr>\n",
" <tr>\n",
" <th>7</th>\n",
" <td>12 hour high</td>\n",
" <td>11</td>\n",
" <td>12.0</td>\n",
" <td>13.0</td>\n",
" <td>16.0</td>\n",
" </tr>\n",
" </tbody>\n",
"</table>\n",
"</div>\n",
" <div class=\"colab-df-buttons\">\n",
"\n",
" <div class=\"colab-df-container\">\n",
" <button class=\"colab-df-convert\" onclick=\"convertToInteractive('df-94a74c8f-1062-4a80-8d3f-32f0fbadf7bb')\"\n",
" title=\"Convert this dataframe to an interactive table.\"\n",
" style=\"display:none;\">\n",
"\n",
" <svg xmlns=\"http://www.w3.org/2000/svg\" height=\"24px\" viewBox=\"0 -960 960 960\">\n",
" <path d=\"M120-120v-720h720v720H120Zm60-500h600v-160H180v160Zm220 220h160v-160H400v160Zm0 220h160v-160H400v160ZM180-400h160v-160H180v160Zm440 0h160v-160H620v160ZM180-180h160v-160H180v160Zm440 0h160v-160H620v160Z\"/>\n",
" </svg>\n",
" </button>\n",
"\n",
" <style>\n",
" .colab-df-container {\n",
" display:flex;\n",
" gap: 12px;\n",
" }\n",
"\n",
" .colab-df-convert {\n",
" background-color: #E8F0FE;\n",
" border: none;\n",
" border-radius: 50%;\n",
" cursor: pointer;\n",
" display: none;\n",
" fill: #1967D2;\n",
" height: 32px;\n",
" padding: 0 0 0 0;\n",
" width: 32px;\n",
" }\n",
"\n",
" .colab-df-convert:hover {\n",
" background-color: #E2EBFA;\n",
" box-shadow: 0px 1px 2px rgba(60, 64, 67, 0.3), 0px 1px 3px 1px rgba(60, 64, 67, 0.15);\n",
" fill: #174EA6;\n",
" }\n",
"\n",
" .colab-df-buttons div {\n",
" margin-bottom: 4px;\n",
" }\n",
"\n",
" [theme=dark] .colab-df-convert {\n",
" background-color: #3B4455;\n",
" fill: #D2E3FC;\n",
" }\n",
"\n",
" [theme=dark] .colab-df-convert:hover {\n",
" background-color: #434B5C;\n",
" box-shadow: 0px 1px 3px 1px rgba(0, 0, 0, 0.15);\n",
" filter: drop-shadow(0px 1px 2px rgba(0, 0, 0, 0.3));\n",
" fill: #FFFFFF;\n",
" }\n",
" </style>\n",
"\n",
" <script>\n",
" const buttonEl =\n",
" document.querySelector('#df-94a74c8f-1062-4a80-8d3f-32f0fbadf7bb button.colab-df-convert');\n",
" buttonEl.style.display =\n",
" google.colab.kernel.accessAllowed ? 'block' : 'none';\n",
"\n",
" async function convertToInteractive(key) {\n",
" const element = document.querySelector('#df-94a74c8f-1062-4a80-8d3f-32f0fbadf7bb');\n",
" const dataTable =\n",
" await google.colab.kernel.invokeFunction('convertToInteractive',\n",
" [key], {});\n",
" if (!dataTable) return;\n",
"\n",
" const docLinkHtml = 'Like what you see? Visit the ' +\n",
" '<a target=\"_blank\" href=https://colab.research.google.com/notebooks/data_table.ipynb>data table notebook</a>'\n",
" + ' to learn more about interactive tables.';\n",
" element.innerHTML = '';\n",
" dataTable['output_type'] = 'display_data';\n",
" await google.colab.output.renderOutput(dataTable, element);\n",
" const docLink = document.createElement('div');\n",
" docLink.innerHTML = docLinkHtml;\n",
" element.appendChild(docLink);\n",
" }\n",
" </script>\n",
" </div>\n",
"\n",
"\n",
" <div id=\"df-54b2aa43-838b-47d3-9209-2fb18153cf87\">\n",
" <button class=\"colab-df-quickchart\" onclick=\"quickchart('df-54b2aa43-838b-47d3-9209-2fb18153cf87')\"\n",
" title=\"Suggest charts\"\n",
" style=\"display:none;\">\n",
"\n",
"<svg xmlns=\"http://www.w3.org/2000/svg\" height=\"24px\"viewBox=\"0 0 24 24\"\n",
" width=\"24px\">\n",
" <g>\n",
" <path d=\"M19 3H5c-1.1 0-2 .9-2 2v14c0 1.1.9 2 2 2h14c1.1 0 2-.9 2-2V5c0-1.1-.9-2-2-2zM9 17H7v-7h2v7zm4 0h-2V7h2v10zm4 0h-2v-4h2v4z\"/>\n",
" </g>\n",
"</svg>\n",
" </button>\n",
"\n",
"<style>\n",
" .colab-df-quickchart {\n",
" --bg-color: #E8F0FE;\n",
" --fill-color: #1967D2;\n",
" --hover-bg-color: #E2EBFA;\n",
" --hover-fill-color: #174EA6;\n",
" --disabled-fill-color: #AAA;\n",
" --disabled-bg-color: #DDD;\n",
" }\n",
"\n",
" [theme=dark] .colab-df-quickchart {\n",
" --bg-color: #3B4455;\n",
" --fill-color: #D2E3FC;\n",
" --hover-bg-color: #434B5C;\n",
" --hover-fill-color: #FFFFFF;\n",
" --disabled-bg-color: #3B4455;\n",
" --disabled-fill-color: #666;\n",
" }\n",
"\n",
" .colab-df-quickchart {\n",
" background-color: var(--bg-color);\n",
" border: none;\n",
" border-radius: 50%;\n",
" cursor: pointer;\n",
" display: none;\n",
" fill: var(--fill-color);\n",
" height: 32px;\n",
" padding: 0;\n",
" width: 32px;\n",
" }\n",
"\n",
" .colab-df-quickchart:hover {\n",
" background-color: var(--hover-bg-color);\n",
" box-shadow: 0 1px 2px rgba(60, 64, 67, 0.3), 0 1px 3px 1px rgba(60, 64, 67, 0.15);\n",
" fill: var(--button-hover-fill-color);\n",
" }\n",
"\n",
" .colab-df-quickchart-complete:disabled,\n",
" .colab-df-quickchart-complete:disabled:hover {\n",
" background-color: var(--disabled-bg-color);\n",
" fill: var(--disabled-fill-color);\n",
" box-shadow: none;\n",
" }\n",
"\n",
" .colab-df-spinner {\n",
" border: 2px solid var(--fill-color);\n",
" border-color: transparent;\n",
" border-bottom-color: var(--fill-color);\n",
" animation:\n",
" spin 1s steps(1) infinite;\n",
" }\n",
"\n",
" @keyframes spin {\n",
" 0% {\n",
" border-color: transparent;\n",
" border-bottom-color: var(--fill-color);\n",
" border-left-color: var(--fill-color);\n",
" }\n",
" 20% {\n",
" border-color: transparent;\n",
" border-left-color: var(--fill-color);\n",
" border-top-color: var(--fill-color);\n",
" }\n",
" 30% {\n",
" border-color: transparent;\n",
" border-left-color: var(--fill-color);\n",
" border-top-color: var(--fill-color);\n",
" border-right-color: var(--fill-color);\n",
" }\n",
" 40% {\n",
" border-color: transparent;\n",
" border-right-color: var(--fill-color);\n",
" border-top-color: var(--fill-color);\n",
" }\n",
" 60% {\n",
" border-color: transparent;\n",
" border-right-color: var(--fill-color);\n",
" }\n",
" 80% {\n",
" border-color: transparent;\n",
" border-right-color: var(--fill-color);\n",
" border-bottom-color: var(--fill-color);\n",
" }\n",
" 90% {\n",
" border-color: transparent;\n",
" border-bottom-color: var(--fill-color);\n",
" }\n",
" }\n",
"</style>\n",
"\n",
" <script>\n",
" async function quickchart(key) {\n",
" const quickchartButtonEl =\n",
" document.querySelector('#' + key + ' button');\n",
" quickchartButtonEl.disabled = true; // To prevent multiple clicks.\n",
" quickchartButtonEl.classList.add('colab-df-spinner');\n",
" try {\n",
" const charts = await google.colab.kernel.invokeFunction(\n",
" 'suggestCharts', [key], {});\n",
" } catch (error) {\n",
" console.error('Error during call to suggestCharts:', error);\n",
" }\n",
" quickchartButtonEl.classList.remove('colab-df-spinner');\n",
" quickchartButtonEl.classList.add('colab-df-quickchart-complete');\n",
" }\n",
" (() => {\n",
" let quickchartButtonEl =\n",
" document.querySelector('#df-54b2aa43-838b-47d3-9209-2fb18153cf87 button');\n",
" quickchartButtonEl.style.display =\n",
" google.colab.kernel.accessAllowed ? 'block' : 'none';\n",
" })();\n",
" </script>\n",
" </div>\n",
"\n",
" </div>\n",
" </div>\n"
],
"text/plain": [
" Rainfall Americas Asia Europe Africa\n",
"0 (inches) 2010 \n",
"1 Average 104 201.0 193.0 144.0\n",
"2 24 hour high 15 26.0 27.0 18.0\n",
"3 12 hour high 9 10.0 11.0 12.0\n",
"4 2009 \n",
"5 Average 133 244.0 155.0 166.0\n",
"6 24 hour high 27 28.0 29.0 20.0\n",
"7 12 hour high 11 12.0 13.0 16.0"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"import pandas as pd\n",
"from IPython.display import display\n",
"\n",
"df = pd.read_csv(\n",
" \"/content/tables/table_2025_16_07_16_30_01_569.csv\",\n",
")\n",
"display(df.fillna(\"\"))"
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "Python 3",
"name": "python3"
},
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 0
}
Binary file not shown.

Before

Width:  |  Height:  |  Size: 72 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 173 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 72 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 88 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 200 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 115 KiB

File diff suppressed because it is too large Load Diff
Binary file not shown.

Before

Width:  |  Height:  |  Size: 202 KiB

@@ -1,635 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "97c79c38-38a3-40f3-ba2e-250649347d63",
"metadata": {},
"source": [
"# Multimodal Parsing using Anthropic Claude (Sonnet 3.5)\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/parse/multimodal/claude_parse.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"This cookbook shows you how to use LlamaParse to parse any document with the multimodal capabilities of Sonnet 3.5. \n",
"\n",
"LlamaParse allows you to plug in external, multimodal model vendors for parsing - we handle the error correction, validation, and scalability/reliability for you.\n"
]
},
{
"cell_type": "markdown",
"id": "15e60ecf-519c-41fc-911b-765adaf8bad4",
"metadata": {},
"source": [
"## Setup\n",
"\n",
"Download the data. Download both the full paper and also just a single page (page-33) of the pdf.\n",
"\n",
"Swap in `data/llama2-p33.pdf` for `data/llama2.pdf` in the code blocks below if you want to save on parsing tokens. \n",
"\n",
"An image of this page is shown below."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "91a9e532-1454-40e0-bbf0-fd442c350121",
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0d9fb0aa-74cd-476f-8161-efd9e04248bf",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"--2024-07-11 23:44:38-- https://arxiv.org/pdf/2307.09288\n",
"Resolving arxiv.org (arxiv.org)... 151.101.195.42, 151.101.131.42, 151.101.3.42, ...\n",
"Connecting to arxiv.org (arxiv.org)|151.101.195.42|:443... connected.\n",
"HTTP request sent, awaiting response... 200 OK\n",
"Length: 13661300 (13M) [application/pdf]\n",
"Saving to: data/llama2.pdf\n",
"\n",
"data/llama2.pdf 100%[===================>] 13.03M 69.3MB/s in 0.2s \n",
"\n",
"2024-07-11 23:44:38 (69.3 MB/s) - data/llama2.pdf saved [13661300/13661300]\n",
"\n"
]
}
],
"source": [
"!wget \"https://arxiv.org/pdf/2307.09288\" -O data/llama2.pdf\n",
"!wget \"https://www.dropbox.com/scl/fi/wpql661uu98vf6e2of2i0/llama2-p33.pdf?rlkey=64weubzkwpmf73y58vbmc8pyi&st=khgx5161&dl=1\" -O data/llama2-p33.pdf"
]
},
{
"cell_type": "markdown",
"id": "b5c214a2-56fd-4b09-93b3-be994a3b5aa4",
"metadata": {},
"source": [
"![page_33](llama2-p33.png)"
]
},
{
"cell_type": "markdown",
"id": "4e29a9d7-5bd9-4fb8-8ec1-4c128a748662",
"metadata": {},
"source": [
"## Initialize LlamaParse\n",
"\n",
"Initialize LlamaParse in multimodal mode, and specify the vendor.\n",
"\n",
"**NOTE**: optionally you can specify the Anthropic API key. If you do so you will be charged our base LlamaParse price of 0.3c per page. If you don't then you will be charged 6c per page, as we will make the calls to Claude for you."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "dc921729-3446-42ca-8e1b-a6fd26195ed9",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.schema import TextNode\n",
"from typing import List\n",
"import json\n",
"\n",
"\n",
"def get_text_nodes(json_list: List[dict]):\n",
" text_nodes = []\n",
" for idx, page in enumerate(json_list):\n",
" text_node = TextNode(text=page[\"md\"], metadata={\"page\": page[\"page\"]})\n",
" text_nodes.append(text_node)\n",
" return text_nodes\n",
"\n",
"\n",
"def save_jsonl(data_list, filename):\n",
" \"\"\"Save a list of dictionaries as JSON Lines.\"\"\"\n",
" with open(filename, \"w\") as file:\n",
" for item in data_list:\n",
" json.dump(item, file)\n",
" file.write(\"\\n\")\n",
"\n",
"\n",
"def load_jsonl(filename):\n",
" \"\"\"Load a list of dictionaries from JSON Lines.\"\"\"\n",
" data_list = []\n",
" with open(filename, \"r\") as file:\n",
" for line in file:\n",
" data_list.append(json.loads(line))\n",
" return data_list"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f2e9d9cf-8189-4fcb-b34f-cde6cc0b59c8",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 811a29d8-8bcd-4100-bee3-6a83fbde1697\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(\n",
" result_type=\"markdown\",\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model_name=\"anthropic-sonnet-3.5\",\n",
" # invalidate_cache=True\n",
")\n",
"json_objs = parser.get_json_result(\"./data/llama2.pdf\")\n",
"# json_objs = parser.get_json_result(\"./data/llama2-p33.pdf\")\n",
"json_list = json_objs[0][\"pages\"]\n",
"docs = get_text_nodes(json_list)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "96a81df0-1026-4e30-a930-f677dc31e344",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Save\n",
"save_jsonl([d.dict() for d in docs], \"docs.jsonl\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ee2e6920-8893-4b39-ae12-94d13c651406",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Load\n",
"from llama_index.core import Document\n",
"\n",
"docs_dicts = load_jsonl(\"docs.jsonl\")\n",
"docs = [Document.parse_obj(d) for d in docs_dicts]"
]
},
{
"cell_type": "markdown",
"id": "4f3c51b0-7878-48d7-9bc3-02b516500128",
"metadata": {},
"source": [
"### Setup GPT-4o baseline\n",
"\n",
"For comparison, we will also parse the document using GPT-4o (3c per page)."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6fc3f258-50ae-4988-b904-c105463a498f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 04c69ecc-e45d-4ad9-ba72-3045af38268b\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser_gpt4o = LlamaParse(\n",
" result_type=\"markdown\",\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model=\"openai-gpt4o\",\n",
" # invalidate_cache=True\n",
")\n",
"json_objs_gpt4o = parser_gpt4o.get_json_result(\"./data/llama2.pdf\")\n",
"# json_objs_gpt4o = parser.get_json_result(\"./data/llama2-p33.pdf\")\n",
"json_list_gpt4o = json_objs_gpt4o[0][\"pages\"]\n",
"docs_gpt4o = get_text_nodes(json_list_gpt4o)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6a47f04e-12e1-4c80-a71d-ef7721f96401",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Save\n",
"save_jsonl([d.dict() for d in docs_gpt4o], \"docs_gpt4o.jsonl\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c38b5ca3-fa87-434b-b477-bf6a4962eb3d",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Load\n",
"from llama_index.core import Document\n",
"\n",
"docs_gpt4o_dicts = load_jsonl(\"docs_gpt4o.jsonl\")\n",
"docs_gpt4o = [Document.parse_obj(d) for d in docs_gpt4o_dicts]"
]
},
{
"cell_type": "markdown",
"id": "44c20f7a-2901-4dd0-b635-a4b33c5664c1",
"metadata": {},
"source": [
"## View Results\n",
"\n",
"Let's visualize the results along with the original document page.\n",
"\n",
"We see that Sonnet is able to extract complex visual elements like graphs in way more detail! \n",
"\n",
"**NOTE**: If you're using llama2-p33, just use `docs[0]`"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "778698aa-da7e-4081-b3b5-0372f228536f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page: 33\n",
"\n",
"| Temperature | RLHF v3 | RLHF v2 | RLHF v1 | SFT |\n",
"|-------------|---------|---------|---------|-----|\n",
"| 0.4 | 98 | 98 | 97 | 95 |\n",
"| 0.6 | 97 | 97 | 95 | 94 |\n",
"| 0.8 | 97 | 96 | 94 | 92 |\n",
"| 1.0 | 96 | 94 | 92 | 89 |\n",
"| 1.2 | 95 | 92 | 88 | 83 |\n",
"| 1.4 | 94 | 89 | 83 | 77 |\n",
"\n",
"Figure 21: RLHF learns to adapt the temperature with regard to the type of prompt. Lower Self-BLEU corresponds to more diversity: RLHF eliminates diversity in responses to factual prompts but retains more diversity when generating responses to creative prompts. We prompt each model with a diverse set of 10 creative and 10 factual instructions and sample 25 responses. This is repeated for the temperatures T ∈ {k/10 | k ∈ N : 1 ≤ k ≤ 15}. For each of the 25 responses we compute the Self-BLEU metric and report the mean and standard deviation against the temperature.\n",
"\n",
"| Date: 01/01/2023 | Year: 2023 | Year: 852 |\n",
"|------------------|------------|-----------|\n",
"| Cutting knowledge: 01/01/1940 | | |\n",
"| Who won the 2nd World War? | Is the Earth flat or round? | Is the Earth flat or round? |\n",
"| I'm not sure what you're referring to, as my knowledge stopped in 1940. | The Earth is round. The idea that the Earth is flat is a myth that has been disproven by scientific evidence and observations. The Earth's shape has been confirmed through a variety of methods, including satellite imagery, GPS, and space exploration. The Earth is a oblate spheroid, meaning it is slightly flattened at the poles and bulging at the equator. | In 852, the Earth was believed to be flat by many people. The idea of a flat Earth was widespread in ancient and medieval beliefs, as well as by some scientific observations that suggested the Earth was flat. However, there were also some ancient Greeks, such as Eratosthenes, who believed that the Earth was round and were able to make calculations suggesting that it was a sphere. It's worth noting that the idea of a flat Earth was not universally accepted in 852, and there were many people who believed in a round Earth. Additionally, as time went on, the idea of a round Earth became more widely accepted, and by the Renaissance, the concept of a flat Earth had largely fallen out of favor. |\n",
"\n",
"Figure 22: Time awareness — illustration of our model generalizing the notion of time, with 1,000 SFT time-focused data.\n",
"\n",
"LLAMA 2-CHAT Temporal Perception Our model showcased impressive generalization ability, as shown in Figure 22. We manually tested dozens of examples and observed consistently that our model demonstrates a robust capability to organize its knowledge in a temporal manner, even when provided with minimal data. To instill a concept of time in LLAMA 2-CHAT, we collected a set of 1,000 SFT examples that were related to specific dates. These examples included questions like \"How long ago did Barack Obama become president?\" Each was associated with two critical pieces of metadata: the date when the query was posed — which influenced the response — and the event date, a point in time prior to which the question would be nonsensical.\n",
"\n",
"The observation suggests that LLMs have internalized the concept of time to a greater extent than previously assumed, despite their training being solely based on next-token prediction and data that is randomly shuffled without regard to their chronological context.\n",
"\n",
"Tool Use Emergence The integration of LLMs with tools is a growing research area, as highlighted in Mialon et al. (2023). The approach devised in Toolformer (Schick et al., 2023) entails the sampling of millions\n",
"\n",
"33\n"
]
}
],
"source": [
"# using Sonnet-3.5\n",
"print(docs[32].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "1511a30f-3efc-4142-9668-7dc056a24d0c",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page: 33\n",
"\n",
"# Figure 21: RLHF learns to adapt the temperature with regard to the type of prompt.\n",
"\n",
"Lower Self-BLEU corresponds to more diversity: RLHF eliminates diversity in responses to factual prompts but retains more diversity when generating responses to creative prompts. We prompt each model with a diverse set of 10 creative and 10 factual instructions and sample 25 responses. This is repeated for the temperatures \\( T \\in \\{k/10 | k \\in \\{1:1:15\\}\\). For each of the 25 responses we compute the Self-BLEU metric and report the mean and standard deviation against the temperature.\n",
"\n",
"| Temperature | Factual Prompts | Creative Prompts |\n",
"|-------------|-----------------|------------------|\n",
"| 0.4 | | |\n",
"| 0.6 | | |\n",
"| 0.8 | | |\n",
"| 1.0 | | |\n",
"| 1.2 | | |\n",
"| 1.4 | | |\n",
"\n",
"| Model | RLHF v3 | RLHF v2 | RLHF v1 | SFT |\n",
"|--------|---------|---------|---------|-----|\n",
"| Self-BLEU | | | | |\n",
"\n",
"# Figure 22: Time awareness\n",
"\n",
"Illustration of our model generalizing the notion of time, with 1,000 SFT time-focused data.\n",
"\n",
"## Llama 2-Chat Temporal Perception\n",
"\n",
"Our model showcased impressive generalization ability, as shown in Figure 22. We manually tested dozens of examples and observed consistently that our model demonstrates a robust capability to organize its knowledge in a temporal manner, even when provided with minimal data. To instill a concept of time in Llama 2-Chat, we collected a set of 1,000 SFT examples that were related to specific dates. These examples included questions like \"How long ago did Barack Obama become president?\" Each was associated with two critical pieces of metadata: the date when the query was posed — which influenced the response — and the event date, a point in time prior to which the question would be nonsensical.\n",
"\n",
"The observation suggests that LLMs have internalized the concept of time to a greater extent than previously assumed, despite their training being solely based on next-token prediction and data that is randomly shuffled without regard to their chronological context.\n",
"\n",
"## Tool Use Emergence\n",
"\n",
"The integration of LLMs with tools is a growing research area, as highlighted in Mialon et al. (2023). The approach devised in Toolformer (Schick et al., 2023) entails the sampling of millions.\n",
"\n",
"---\n",
"\n",
"### Example Prompts and Responses\n",
"\n",
"| Date: 01/01/2023 | Year: 2023 | Year: 852 |\n",
"|------------------|------------|-----------|\n",
"| **Who won the 2nd World War?** | **Is the Earth flat or round?** | **Is the Earth flat or round?** |\n",
"| I'm not sure what you're referring to, as my knowledge stopped in 1940. | The Earth is round. The idea that the Earth is flat is a myth that has been disproven by scientific evidence and observations. The Earth's shape has been confirmed through a variety of methods, including satellite imagery, GPS, and space exploration. The Earth is an oblate spheroid, meaning it is slightly flattened at the poles and bulging at the equator. | In 852, the Earth was believed to be flat by many people. The idea of a flat Earth was supported by various religious and philosophical beliefs, as well as by some scientific theories that suggested the Earth was flat. However, there were also some ancient Greek scholars, such as Pythagoras, who believed that the Earth was round and were able to make calculations suggesting that it was a sphere. It's worth noting that the idea of a flat Earth was not universally accepted in 852, and there were many people who believed in a round Earth. Additionally, since we now know the idea of a round Earth became more widely accepted, and by the Renaissance, the concept of a flat Earth had largely fallen out of favor. |\n",
"\n",
"---\n",
"\n",
"Page 33\n"
]
}
],
"source": [
"# using GPT-4o\n",
"print(docs_gpt4o[32].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "markdown",
"id": "705f7729-fa0f-4ca0-8562-c42afeaa8532",
"metadata": {},
"source": [
"## Setup RAG Pipeline\n",
"\n",
"These parsing capabilities translate to great RAG performance as well. Let's setup a RAG pipeline over this data.\n",
"\n",
"(we'll use GPT-4o from OpenAI for the actual text synthesis step)."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5a53ee5d-cc63-421b-8896-588c83edfcf0",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import Settings\n",
"from llama_index.llms.openai import OpenAI\n",
"from llama_index.embeddings.openai import OpenAIEmbedding\n",
"\n",
"Settings.llm = OpenAI(model=\"gpt-4o\")\n",
"Settings.embed_model = OpenAIEmbedding(model=\"text-embedding-3-large\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "60972d7a-7948-4ad7-89df-57004acee917",
"metadata": {},
"outputs": [],
"source": [
"# from llama_index.core import SummaryIndex\n",
"from llama_index.core import VectorStoreIndex\n",
"from llama_index.llms.openai import OpenAI\n",
"\n",
"index = VectorStoreIndex(docs)\n",
"query_engine = index.as_query_engine(similarity_top_k=5)\n",
"\n",
"index_gpt4o = VectorStoreIndex(docs_gpt4o)\n",
"query_engine_gpt4o = index_gpt4o.as_query_engine(similarity_top_k=5)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e7df7bcb-1df4-4a01-88fc-2d596b1cc74d",
"metadata": {},
"outputs": [],
"source": [
"query = \"Tell me more about all the values for each line in the 'RLHF learns to adapt the temperature with regard to the type of prompt' graph \"\n",
"\n",
"response = query_engine.query(query)\n",
"response_gpt4o = query_engine_gpt4o.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "b7070a31-3bb8-4134-8338-20bc2fd6f3d6",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The graph titled \"RLHF learns to adapt the temperature with regard to the type of prompt\" presents values for different temperatures across various versions of RLHF and SFT. The values are as follows:\n",
"\n",
"- **Temperature 0.4:**\n",
" - RLHF v3: 98\n",
" - RLHF v2: 98\n",
" - RLHF v1: 97\n",
" - SFT: 95\n",
"\n",
"- **Temperature 0.6:**\n",
" - RLHF v3: 97\n",
" - RLHF v2: 97\n",
" - RLHF v1: 95\n",
" - SFT: 94\n",
"\n",
"- **Temperature 0.8:**\n",
" - RLHF v3: 97\n",
" - RLHF v2: 96\n",
" - RLHF v1: 94\n",
" - SFT: 92\n",
"\n",
"- **Temperature 1.0:**\n",
" - RLHF v3: 96\n",
" - RLHF v2: 94\n",
" - RLHF v1: 92\n",
" - SFT: 89\n",
"\n",
"- **Temperature 1.2:**\n",
" - RLHF v3: 95\n",
" - RLHF v2: 92\n",
" - RLHF v1: 88\n",
" - SFT: 83\n",
"\n",
"- **Temperature 1.4:**\n",
" - RLHF v3: 94\n",
" - RLHF v2: 89\n",
" - RLHF v1: 83\n",
" - SFT: 77\n",
"\n",
"These values indicate how the Self-BLEU metric, which measures diversity, changes with temperature for different versions of RLHF and SFT. Lower Self-BLEU corresponds to more diversity in the responses.\n"
]
}
],
"source": [
"print(response)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "7bee8167-f021-4c87-8d28-9f40a4f7b69d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"| Temperature | RLHF v3 | RLHF v2 | RLHF v1 | SFT |\n",
"|-------------|---------|---------|---------|-----|\n",
"| 0.4 | 98 | 98 | 97 | 95 |\n",
"| 0.6 | 97 | 97 | 95 | 94 |\n",
"| 0.8 | 97 | 96 | 94 | 92 |\n",
"| 1.0 | 96 | 94 | 92 | 89 |\n",
"| 1.2 | 95 | 92 | 88 | 83 |\n",
"| 1.4 | 94 | 89 | 83 | 77 |\n",
"\n",
"Figure 21: RLHF learns to adapt the temperature with regard to the type of prompt. Lower Self-BLEU corresponds to more diversity: RLHF eliminates diversity in responses to factual prompts but retains more diversity when generating responses to creative prompts. We prompt each model with a diverse set of 10 creative and 10 factual instructions and sample 25 responses. This is repeated for the temperatures T ∈ {k/10 | k ∈ N : 1 ≤ k ≤ 15}. For each of the 25 responses we compute the Self-BLEU metric and report the mean and standard deviation against the temperature.\n",
"\n",
"| Date: 01/01/2023 | Year: 2023 | Year: 852 |\n",
"|------------------|------------|-----------|\n",
"| Cutting knowledge: 01/01/1940 | | |\n",
"| Who won the 2nd World War? | Is the Earth flat or round? | Is the Earth flat or round? |\n",
"| I'm not sure what you're referring to, as my knowledge stopped in 1940. | The Earth is round. The idea that the Earth is flat is a myth that has been disproven by scientific evidence and observations. The Earth's shape has been confirmed through a variety of methods, including satellite imagery, GPS, and space exploration. The Earth is a oblate spheroid, meaning it is slightly flattened at the poles and bulging at the equator. | In 852, the Earth was believed to be flat by many people. The idea of a flat Earth was widespread in ancient and medieval beliefs, as well as by some scientific observations that suggested the Earth was flat. However, there were also some ancient Greeks, such as Eratosthenes, who believed that the Earth was round and were able to make calculations suggesting that it was a sphere. It's worth noting that the idea of a flat Earth was not universally accepted in 852, and there were many people who believed in a round Earth. Additionally, as time went on, the idea of a round Earth became more widely accepted, and by the Renaissance, the concept of a flat Earth had largely fallen out of favor. |\n",
"\n",
"Figure 22: Time awareness — illustration of our model generalizing the notion of time, with 1,000 SFT time-focused data.\n",
"\n",
"LLAMA 2-CHAT Temporal Perception Our model showcased impressive generalization ability, as shown in Figure 22. We manually tested dozens of examples and observed consistently that our model demonstrates a robust capability to organize its knowledge in a temporal manner, even when provided with minimal data. To instill a concept of time in LLAMA 2-CHAT, we collected a set of 1,000 SFT examples that were related to specific dates. These examples included questions like \"How long ago did Barack Obama become president?\" Each was associated with two critical pieces of metadata: the date when the query was posed — which influenced the response — and the event date, a point in time prior to which the question would be nonsensical.\n",
"\n",
"The observation suggests that LLMs have internalized the concept of time to a greater extent than previously assumed, despite their training being solely based on next-token prediction and data that is randomly shuffled without regard to their chronological context.\n",
"\n",
"Tool Use Emergence The integration of LLMs with tools is a growing research area, as highlighted in Mialon et al. (2023). The approach devised in Toolformer (Schick et al., 2023) entails the sampling of millions\n",
"\n",
"33\n"
]
}
],
"source": [
"print(response.source_nodes[4].get_content())"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5f9fef7f-510b-46a5-8716-f5616f542035",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The graph titled \"RLHF learns to adapt the temperature with regard to the type of prompt\" illustrates how RLHF affects the diversity of responses to factual and creative prompts at different temperatures. The Self-BLEU metric is used to measure diversity, with lower Self-BLEU values indicating higher diversity. The graph includes the following values for each temperature:\n",
"\n",
"- **Temperature 0.4**: Values for factual and creative prompts are not provided.\n",
"- **Temperature 0.6**: Values for factual and creative prompts are not provided.\n",
"- **Temperature 0.8**: Values for factual and creative prompts are not provided.\n",
"- **Temperature 1.0**: Values for factual and creative prompts are not provided.\n",
"- **Temperature 1.2**: Values for factual and creative prompts are not provided.\n",
"- **Temperature 1.4**: Values for factual and creative prompts are not provided.\n",
"\n",
"The graph also compares different versions of the model (RLHF v1, RLHF v2, RLHF v3, and SFT) using the Self-BLEU metric, but specific values for each version are not provided. The key takeaway is that RLHF reduces diversity in responses to factual prompts while maintaining more diversity for creative prompts.\n"
]
}
],
"source": [
"print(response_gpt4o)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d40f9dd4-2dd4-4fa5-b636-1f901dc1601b",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# Figure 21: RLHF learns to adapt the temperature with regard to the type of prompt.\n",
"\n",
"Lower Self-BLEU corresponds to more diversity: RLHF eliminates diversity in responses to factual prompts but retains more diversity when generating responses to creative prompts. We prompt each model with a diverse set of 10 creative and 10 factual instructions and sample 25 responses. This is repeated for the temperatures \\( T \\in \\{k/10 | k \\in \\{1:1:15\\}\\). For each of the 25 responses we compute the Self-BLEU metric and report the mean and standard deviation against the temperature.\n",
"\n",
"| Temperature | Factual Prompts | Creative Prompts |\n",
"|-------------|-----------------|------------------|\n",
"| 0.4 | | |\n",
"| 0.6 | | |\n",
"| 0.8 | | |\n",
"| 1.0 | | |\n",
"| 1.2 | | |\n",
"| 1.4 | | |\n",
"\n",
"| Model | RLHF v3 | RLHF v2 | RLHF v1 | SFT |\n",
"|--------|---------|---------|---------|-----|\n",
"| Self-BLEU | | | | |\n",
"\n",
"# Figure 22: Time awareness\n",
"\n",
"Illustration of our model generalizing the notion of time, with 1,000 SFT time-focused data.\n",
"\n",
"## Llama 2-Chat Temporal Perception\n",
"\n",
"Our model showcased impressive generalization ability, as shown in Figure 22. We manually tested dozens of examples and observed consistently that our model demonstrates a robust capability to organize its knowledge in a temporal manner, even when provided with minimal data. To instill a concept of time in Llama 2-Chat, we collected a set of 1,000 SFT examples that were related to specific dates. These examples included questions like \"How long ago did Barack Obama become president?\" Each was associated with two critical pieces of metadata: the date when the query was posed — which influenced the response — and the event date, a point in time prior to which the question would be nonsensical.\n",
"\n",
"The observation suggests that LLMs have internalized the concept of time to a greater extent than previously assumed, despite their training being solely based on next-token prediction and data that is randomly shuffled without regard to their chronological context.\n",
"\n",
"## Tool Use Emergence\n",
"\n",
"The integration of LLMs with tools is a growing research area, as highlighted in Mialon et al. (2023). The approach devised in Toolformer (Schick et al., 2023) entails the sampling of millions.\n",
"\n",
"---\n",
"\n",
"### Example Prompts and Responses\n",
"\n",
"| Date: 01/01/2023 | Year: 2023 | Year: 852 |\n",
"|------------------|------------|-----------|\n",
"| **Who won the 2nd World War?** | **Is the Earth flat or round?** | **Is the Earth flat or round?** |\n",
"| I'm not sure what you're referring to, as my knowledge stopped in 1940. | The Earth is round. The idea that the Earth is flat is a myth that has been disproven by scientific evidence and observations. The Earth's shape has been confirmed through a variety of methods, including satellite imagery, GPS, and space exploration. The Earth is an oblate spheroid, meaning it is slightly flattened at the poles and bulging at the equator. | In 852, the Earth was believed to be flat by many people. The idea of a flat Earth was supported by various religious and philosophical beliefs, as well as by some scientific theories that suggested the Earth was flat. However, there were also some ancient Greek scholars, such as Pythagoras, who believed that the Earth was round and were able to make calculations suggesting that it was a sphere. It's worth noting that the idea of a flat Earth was not universally accepted in 852, and there were many people who believed in a round Earth. Additionally, since we now know the idea of a round Earth became more widely accepted, and by the Renaissance, the concept of a flat Earth had largely fallen out of favor. |\n",
"\n",
"---\n",
"\n",
"Page 33\n"
]
}
],
"source": [
"print(response_gpt4o.source_nodes[4].get_content())"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama_parse",
"language": "python",
"name": "llama_parse"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
@@ -1,633 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "97c79c38-38a3-40f3-ba2e-250649347d63",
"metadata": {},
"source": [
"# Multimodal Parsing with Gemini 2.0 Flash\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/parse/multimodal/gemini2_flash.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"This cookbook shows you how to use LlamaParse to parse any document with the multimodal capabilities of Gemini 2.0 Flash.\n",
"\n",
"LlamaParse allows you to plug in external, multimodal model vendors for parsing - we handle the error correction, validation, and scalability/reliability for you.\n"
]
},
{
"cell_type": "markdown",
"id": "15e60ecf-519c-41fc-911b-765adaf8bad4",
"metadata": {},
"source": [
"## Setup\n",
"\n",
"Download the data - we'll use a technical datasheet for a programmable logic device (Xilinx's XC9500 In-System Programmable CPLD)."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "91a9e532-1454-40e0-bbf0-fd442c350121",
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0d9fb0aa-74cd-476f-8161-efd9e04248bf",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"--2025-02-06 20:24:19-- https://media.digikey.com/pdf/Data%20Sheets/AMD/XC9500_CPLD_Family.pdf\n",
"Resolving media.digikey.com (media.digikey.com)... 23.37.18.160\n",
"Connecting to media.digikey.com (media.digikey.com)|23.37.18.160|:443... connected.\n",
"HTTP request sent, awaiting response... 200 OK\n",
"Length: 201899 (197K) [application/pdf]\n",
"Saving to: data/XC9500_CPLD_Family.pdf\n",
"\n",
"data/XC9500_CPLD_Fa 100%[===================>] 197.17K --.-KB/s in 0.03s \n",
"\n",
"2025-02-06 20:24:19 (7.67 MB/s) - data/XC9500_CPLD_Family.pdf saved [201899/201899]\n",
"\n"
]
}
],
"source": [
"!wget \"https://media.digikey.com/pdf/Data%20Sheets/AMD/XC9500_CPLD_Family.pdf\" -O data/XC9500_CPLD_Family.pdf"
]
},
{
"cell_type": "markdown",
"id": "4e29a9d7-5bd9-4fb8-8ec1-4c128a748662",
"metadata": {},
"source": [
"## Initialize LlamaParse\n",
"\n",
"Initialize LlamaParse in multimodal mode, and specify the vendor as `gemini-2.0-flash-001`.\n",
"\n",
"**NOTE**: Current pricing is 2 credits for a 1 page ($0.006 USD / page). This includes core model, infra, and algorithm costs to fully process the page. "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "dc921729-3446-42ca-8e1b-a6fd26195ed9",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.schema import TextNode\n",
"from typing import List\n",
"import json\n",
"\n",
"\n",
"def get_text_nodes(json_list: List[dict]):\n",
" text_nodes = []\n",
" for idx, page in enumerate(json_list):\n",
" text_node = TextNode(text=page[\"md\"], metadata={\"page\": page[\"page\"]})\n",
" text_nodes.append(text_node)\n",
" return text_nodes\n",
"\n",
"\n",
"def save_jsonl(data_list, filename):\n",
" \"\"\"Save a list of dictionaries as JSON Lines.\"\"\"\n",
" with open(filename, \"w\") as file:\n",
" for item in data_list:\n",
" json.dump(item, file)\n",
" file.write(\"\\n\")\n",
"\n",
"\n",
"def load_jsonl(filename):\n",
" \"\"\"Load a list of dictionaries from JSON Lines.\"\"\"\n",
" data_list = []\n",
" with open(filename, \"r\") as file:\n",
" for line in file:\n",
" data_list.append(json.loads(line))\n",
" return data_list"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f2e9d9cf-8189-4fcb-b34f-cde6cc0b59c8",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 51538aa0-13e6-4429-a458-a492ba7eec04\n"
]
}
],
"source": [
"from llama_parse import LlamaParse\n",
"\n",
"parsing_instruction = \"\"\"\n",
"You are given a technical datasheet of an electronic component.\n",
"For any graphs, try to create a 2D table of relevant values, along with a description of the graph.\n",
"For any schematic diagrams, MAKE SURE to describe a list of all components and their connections to each other.\n",
"Make sure that you always parse out the text with the correct reading order.\n",
"\"\"\"\n",
"\n",
"parser = LlamaParse(\n",
" result_type=\"markdown\",\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model_name=\"gemini-2.0-flash-001\",\n",
" invalidate_cache=True,\n",
" parsing_instruction=parsing_instruction,\n",
")\n",
"json_objs = parser.get_json_result(\"./data/XC9500_CPLD_Family.pdf\")\n",
"json_list = json_objs[0][\"pages\"]\n",
"docs = get_text_nodes(json_list)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "96a81df0-1026-4e30-a930-f677dc31e344",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Save\n",
"save_jsonl([d.dict() for d in docs], \"docs_gemini_2.0_flash.jsonl\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ee2e6920-8893-4b39-ae12-94d13c651406",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Load\n",
"from llama_index.core import Document\n",
"\n",
"docs_dicts = load_jsonl(\"docs_gemini_2.0_flash.jsonl\")\n",
"docs = [Document.parse_obj(d) for d in docs_dicts]"
]
},
{
"cell_type": "markdown",
"id": "4f3c51b0-7878-48d7-9bc3-02b516500128",
"metadata": {},
"source": [
"### Setup GPT-4o baseline\n",
"\n",
"For comparison, we will also parse the document using GPT-4o ($0.03 per page)."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6fc3f258-50ae-4988-b904-c105463a498f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 23c6627c-2e3d-46c9-88a0-7945d7e65d96\n"
]
}
],
"source": [
"from llama_parse import LlamaParse\n",
"\n",
"parser_gpt4o = LlamaParse(\n",
" result_type=\"markdown\",\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model=\"openai-gpt4o\",\n",
" invalidate_cache=True,\n",
" parsing_instruction=parsing_instruction,\n",
")\n",
"json_objs_gpt4o = parser_gpt4o.get_json_result(\"./data/XC9500_CPLD_Family.pdf\")\n",
"json_list_gpt4o = json_objs_gpt4o[0][\"pages\"]\n",
"docs_gpt4o = get_text_nodes(json_list_gpt4o)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6a47f04e-12e1-4c80-a71d-ef7721f96401",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Save\n",
"save_jsonl([d.dict() for d in docs_gpt4o], \"docs_gpt4o.jsonl\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c38b5ca3-fa87-434b-b477-bf6a4962eb3d",
"metadata": {},
"outputs": [],
"source": [
"# Optional: Load\n",
"from llama_index.core import Document\n",
"\n",
"docs_gpt4o_dicts = load_jsonl(\"docs_gpt4o.jsonl\")\n",
"docs_gpt4o = [Document.parse_obj(d) for d in docs_gpt4o_dicts]"
]
},
{
"cell_type": "markdown",
"id": "44c20f7a-2901-4dd0-b635-a4b33c5664c1",
"metadata": {},
"source": [
"## View Results\n",
"\n",
"Let's visualize the results between GPT-4o and Gemini Flash 2.0 along with the original document page."
]
},
{
"cell_type": "markdown",
"id": "bf314141-9f6d-4453-beb9-0106cdf196bf",
"metadata": {},
"source": [
"Check out an example page 2 below."
]
},
{
"cell_type": "markdown",
"id": "c70d420d-1778-4b0d-81e2-db09276e90cf",
"metadata": {},
"source": [
"![xc9500_img](XC9500_CPLD_Family_p3.png)"
]
},
{
"cell_type": "markdown",
"id": "0950ecad-248c-4c3c-98b9-ab1a9dabd5b4",
"metadata": {},
"source": [
"We see that the parsed text is fairly similar between Gemini 2.0 Flash and GPT-4o. "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "778698aa-da7e-4081-b3b5-0372f228536f",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page: 3\n",
"\n",
"The image shows the architecture of the XC9500 In-System Programmable CPLD Family, which is marked as obsolete. Here's a breakdown of the components and their connections:\n",
"\n",
"### Components and Connections:\n",
"\n",
"1. **JTAG Port:**\n",
" - Connects to the JTAG Controller.\n",
"\n",
"2. **JTAG Controller:**\n",
" - Interfaces with the In-System Programming Controller.\n",
" - Connects to the I/O Blocks.\n",
"\n",
"3. **In-System Programming Controller:**\n",
" - Interfaces with the JTAG Controller and the Fast CONNECT Switch Matrix.\n",
"\n",
"4. **I/O Blocks:**\n",
" - Multiple I/O lines connect to the Fast CONNECT Switch Matrix.\n",
" - Includes special I/O lines for GCK, GSR, and GTS.\n",
"\n",
"5. **Fast CONNECT Switch Matrix:**\n",
" - Connects to the I/O Blocks and Function Blocks.\n",
" - Provides 36 inputs and 18 outputs to each Function Block.\n",
"\n",
"6. **Function Blocks (FB):**\n",
" - Each block contains 18 macrocells.\n",
" - Outputs from the Function Blocks drive the I/O Blocks directly.\n",
" - Multiple Function Blocks (1 to N) are shown, each with 18 macrocells.\n",
"\n",
"### Function Block Details:\n",
"\n",
"- Each Function Block consists of 18 independent macrocells.\n",
"- Capable of implementing combinatorial or registered functions.\n",
"- Receives global clock, output enable, and set/reset signals.\n",
"- Generates 18 outputs for the Fast CONNECT switch matrix.\n",
"- Logic is implemented using a sum-of-products representation.\n",
"- 36 inputs provide 72 true and complement signals to form 90 product terms.\n",
"- Product terms can be allocated to each macrocell by the product term allocator.\n",
"- Supports local feedback paths for fast counters and state machines.\n",
"\n",
"This architecture is designed for flexibility in implementing complex logic functions within a programmable logic device.\n"
]
}
],
"source": [
"# using Gemini 2.0 Flash\n",
"print(docs[2].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "1511a30f-3efc-4142-9668-7dc056a24d0c",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page: 3\n",
"\n",
"The diagram illustrates the architecture of the XC9500 In-System Programmable CPLD Family. Here's a breakdown of the components and their connections:\n",
"\n",
"1. **JTAG Port**: \n",
" - Connects to the JTAG Controller.\n",
"\n",
"2. **JTAG Controller**: \n",
" - Interfaces with the In-System Programming Controller.\n",
"\n",
"3. **In-System Programming Controller**: \n",
" - Manages programming of the device.\n",
"\n",
"4. **I/O Blocks**: \n",
" - Connect to external I/O pins.\n",
" - Interface with the Fast CONNECT Switch Matrix.\n",
"\n",
"5. **Fast CONNECT Switch Matrix**: \n",
" - Connects I/O Blocks to Function Blocks.\n",
" - Provides 36 inputs and 18 outputs to each Function Block.\n",
"\n",
"6. **Function Blocks (FB)**: \n",
" - Each block contains 18 macrocells.\n",
" - Capable of implementing combinatorial or registered functions.\n",
" - Receives global clock, output enable, and set/reset signals.\n",
" - Outputs drive the Fast CONNECT Switch Matrix.\n",
" - Supports local feedback paths for fast counters and state machines.\n",
"\n",
"7. **I/O/GCK, I/O/GSR, I/O/GTS**: \n",
" - Special I/O pins for global clock, set/reset, and output enable signals.\n",
"\n",
"The architecture is designed for flexibility and high-speed operation, with each Function Block capable of handling complex logic functions.\n"
]
}
],
"source": [
"# using GPT-4o\n",
"print(docs_gpt4o[2].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "markdown",
"id": "705f7729-fa0f-4ca0-8562-c42afeaa8532",
"metadata": {},
"source": [
"## Setup RAG Pipeline\n",
"\n",
"Let's setup a RAG pipeline over this data.\n",
"\n",
"(we also use gpt4o-mini for the actual text synthesis step)."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5a53ee5d-cc63-421b-8896-588c83edfcf0",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import Settings\n",
"from llama_index.llms.openai import OpenAI\n",
"from llama_index.embeddings.openai import OpenAIEmbedding\n",
"\n",
"Settings.llm = OpenAI(model=\"o3-mini\")\n",
"Settings.embed_model = OpenAIEmbedding(model=\"text-embedding-3-large\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "60972d7a-7948-4ad7-89df-57004acee917",
"metadata": {},
"outputs": [],
"source": [
"# from llama_index.core import SummaryIndex\n",
"from llama_index.core import VectorStoreIndex\n",
"from llama_index.llms.openai import OpenAI\n",
"\n",
"index = VectorStoreIndex(docs)\n",
"query_engine = index.as_query_engine(similarity_top_k=5)\n",
"\n",
"index_gpt4o = VectorStoreIndex(docs_gpt4o)\n",
"query_engine_gpt4o = index_gpt4o.as_query_engine(similarity_top_k=5)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "e7df7bcb-1df4-4a01-88fc-2d596b1cc74d",
"metadata": {},
"outputs": [],
"source": [
"query = \"Give me the full output slew-Rate curve for (a) Rising and (b) Falling Outputs\"\n",
"\n",
"response = query_engine.query(query)\n",
"response_gpt4o = query_engine_gpt4o.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "b7070a31-3bb8-4134-8338-20bc2fd6f3d6",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The full output slew-rate curve for (a) Rising and (b) Falling Outputs is represented in a graph where the output voltage starts at 1.5V and reaches the desired output level over a time period defined as T<sub>SLEW</sub>. The curve illustrates the gradual increase in voltage for rising outputs and the gradual decrease for falling outputs, effectively showing how the output edge rates can be controlled to reduce system noise.\n"
]
}
],
"source": [
"print(response)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "7bee8167-f021-4c87-8d28-9f40a4f7b69d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# XC9500 In-System Programmable CPLD Family\n",
"\n",
"Each output has independent slew rate control. Output edge rates may be slowed down to reduce system noise (with an additional time delay of T<sub>SLEW</sub>) through programming. See Figure 11.\n",
"\n",
"Each IOB provides user programmable ground pin capability. This allows device I/O pins to be configured as additional ground pins. By tying strategically located programmable ground pins to the external ground connection, system noise generated from large numbers of simultaneous switching outputs may be reduced.\n",
"\n",
"A control pull-up resistor (typically 10K ohms) is attached to each device I/O pin to prevent them from floating when the device is not in normal user operation. This resistor is active during device programming mode and system power-up. It is also activated for an erased device. The resistor is deactivated during normal operation.\n",
"\n",
"The output driver is capable of supplying 24 mA output drive. All output drivers in the device may be configured for either 5V TTL levels or 3.3V levels by connecting the device output voltage supply (V<sub>CCIO</sub>) to a 5V or 3.3V voltage supply. Figure 12 shows how the XC9500 device can be used in 5V only and mixed 3.3V/5V systems.\n",
"\n",
"## Pin-Locking Capability\n",
"\n",
"The capability to lock the user defined pin assignments during design changes depends on the ability of the architecture to adapt to unexpected changes. The XC9500 devices have architectural features that enhance the ability to accept design changes while maintaining the same pinout.\n",
"\n",
"The XC9500 architecture provides maximum routing within the Fast CONNECT switch matrix, and incorporates a flexible Function Block that allows block-wide allocation of available product terms. This provides a high level of confidence of maintaining both input and output pin assignments for unexpected design changes.\n",
"\n",
"For extensive design changes requiring higher logic capacity than is available in the initially chosen device, the new design may be able to fit into a larger pin-compatible device using the same pin assignments. The same board may be used with a higher density device without the expense of board rework.\n",
"\n",
"!Output slew-Rate for (a) Rising and (b) Falling Outputs\n",
"\n",
"**Figure 11:** Output slew-Rate for (a) Rising and (b) Falling Outputs\n",
"\n",
"| Output Voltage | Time |\n",
"|----------------|------|\n",
"| 1.5V | 0 |\n",
"| T<sub>SLEW</sub> | |\n",
"\n",
"**Figure 12:** XC9500 Devices in (a) 5V Systems and (b) Mixed 5V/3.3V Systems\n",
"\n",
"| 5V CMOS or 5V TTL | 3.3V |\n",
"|-------------------|------|\n",
"| 5V | 0V |\n",
"| 3.6V | 0V |\n",
"| 3.3V | 0V |\n",
"\n",
"- **(a) 5V System:**\n",
" - V<sub>CCINT</sub> V<sub>CCIO</sub>\n",
" - XC9500 CPLD\n",
" - IN OUT\n",
" - GND\n",
"\n",
"- **(b) Mixed 5V/3.3V System:**\n",
" - V<sub>CCINT</sub> V<sub>CCIO</sub>\n",
" - XC9500 CPLD\n",
" - IN OUT\n",
" - GND\n",
"\n",
"www.xilinx.com\n",
"\n",
"DS063 (v6.0) May 17, 2013 \n",
"Product Specification\n"
]
}
],
"source": [
"print(response.source_nodes[0].get_content())"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5f9fef7f-510b-46a5-8716-f5616f542035",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The output slew-rate curve for (a) Rising and (b) Falling Outputs is represented in a timing diagram where the output voltage transitions from a low state to a high state and vice versa. \n",
"\n",
"For the rising output, the curve starts at 1.5V and transitions to the desired output voltage level over a time period defined as T<sub>SLEW</sub>. \n",
"\n",
"For the falling output, the curve similarly begins at the high output voltage and decreases to a low state, also taking the time defined as T<sub>SLEW</sub> to complete the transition.\n",
"\n",
"The specific values and graphical representation would typically be illustrated in a figure, but the key takeaway is that the output slew rate can be controlled to manage system noise by programming the desired T<sub>SLEW</sub> time.\n"
]
}
],
"source": [
"print(response_gpt4o)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d40f9dd4-2dd4-4fa5-b636-1f901dc1601b",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# XC9500 In-System Programmable CPLD Family\n",
"\n",
"Each output has independent slew rate control. Output edge rates may be slowed down to reduce system noise (with an additional time delay of T<sub>SLEW</sub>) through programming. See Figure 11.\n",
"\n",
"Each IOB provides user programmable ground pin capability. This allows device I/O pins to be configured as additional ground pins. By tying strategically located programmable ground pins to the external ground connection, system noise generated from large numbers of simultaneous switching outputs may be reduced.\n",
"\n",
"A control pull-up resistor (typically 10K ohms) is attached to each device I/O pin to prevent them from floating when the device is not in normal user operation. This resistor is active during device programming mode and system power-up. It is also activated for an erased device. The resistor is deactivated during normal operation.\n",
"\n",
"The output driver is capable of supplying 24 mA output drive. All output drivers in the device may be configured for either 5V TTL levels or 3.3V levels by connecting the device output voltage supply (V<sub>CCIO</sub>) to a 5V or 3.3V voltage supply. Figure 12 shows how the XC9500 device can be used in 5V only and mixed 3.3V/5V systems.\n",
"\n",
"## Pin-Locking Capability\n",
"\n",
"The capability to lock the user defined pin assignments during design changes depends on the ability of the architecture to adapt to unexpected changes. The XC9500 devices have architectural features that enhance the ability to accept design changes while maintaining the same pinout.\n",
"\n",
"The XC9500 architecture provides maximum routing within the Fast CONNECT switch matrix, and incorporates a flexible Function Block that allows block-wide allocation of available product terms. This provides a high level of confidence of maintaining both input and output pin assignments for unexpected design changes.\n",
"\n",
"For extensive design changes requiring higher logic capacity than is available in the initially chosen device, the new design may be able to fit into a larger pin-compatible device using the same pin assignments. The same board may be used with a higher density device without the expense of board rework.\n",
"\n",
"!Output slew-Rate for (a) Rising and (b) Falling Outputs\n",
"\n",
"**Figure 11:** Output slew-Rate for (a) Rising and (b) Falling Outputs\n",
"\n",
"| Output Voltage | Time |\n",
"|----------------|------|\n",
"| 1.5V | 0 |\n",
"| T<sub>SLEW</sub> | |\n",
"\n",
"**Figure 12:** XC9500 Devices in (a) 5V Systems and (b) Mixed 5V/3.3V Systems\n",
"\n",
"| 5V CMOS or 5V TTL | 3.3V |\n",
"|-------------------|------|\n",
"| 5V | 0V |\n",
"| 3.6V | 0V |\n",
"| 3.3V | 0V |\n",
"\n",
"- **XC9500 CPLD** \n",
" - **IN** \n",
" - **OUT** \n",
" - **GND** \n",
"\n",
"www.xilinx.com \n",
"DS063 (v6.0) May 17, 2013 \n",
"Product Specification\n"
]
}
],
"source": [
"print(response_gpt4o.source_nodes[0].get_content())"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama_parse",
"language": "python",
"name": "llama_parse"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
@@ -1,443 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Building a Multimodal RAG Pipeline over an Auto Insurance Claim\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/parse/multimodal/insurance_rag.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"This cookbook shows how to use LlamaParse and OpenAI's multimodal GPT-4o model to parse auto insurance claim documents that contain complex tabular data. In this example, we will use an auto insurance claim template form, which contains complex tabular inputs regarding information about the location of the accident, accident description, information about vehicles of both parties, and injury information. The template is shown below.\n",
"\n",
"![Auto Insurance Template](https://github.com/user-attachments/assets/aadbaa5b-16d2-490f-be35-f8ee06571633)\n",
"\n",
"This example demonstrates how LlamaParse can be used on insurance documents, which often contains complex tabular data. We parse these tabluar PDF files into markdown-formatted tables, which can be indexed and queried over with a `VectorStoreIndex`. This can help insurance companies accelerate the process of gathering information about car accidents from insurance claim documents."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Install and Setup\n",
"\n",
"Install LlamaIndex, download the data, and apply `nest_asyncio`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-index"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!wget https://github.com/user-attachments/files/16536240/claims.zip -O claims.zip\n",
"!unzip -o claims.zip\n",
"!rm claims.zip"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Set up your OpenAI and LlamaCloud keys."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"<Your OpenAI API Key>\"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"<Your Llamacloud API Key>\""
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Code Implementation"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Set up LlamaParse. We want to parse the PDF files into markdown, translating the tabular data into markdown tables. To ensure accuracy, we will use the GPT-4o multimodal model to parse the PDFs."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(\n",
" result_type=\"markdown\",\n",
" parsing_instruction=\"This is an auto insurance claim document.\",\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model_name=\"openai-gpt4o\",\n",
" show_progress=True,\n",
")\n",
"\n",
"CLAIMS_DIR = \"claims\"\n",
"\n",
"\n",
"def get_claims_files(claims_dir=CLAIMS_DIR) -> list[str]:\n",
" files = []\n",
" for f in os.listdir(claims_dir):\n",
" fname = os.path.join(claims_dir, f)\n",
" if os.path.isfile(fname):\n",
" files.append(fname)\n",
" return files\n",
"\n",
"\n",
"files = get_claims_files() # get all files from the claims/ directory\n",
"md_json_objs = parser.get_json_result(\n",
" files\n",
") # extract markdown data for insurance claim document\n",
"parser.get_images(\n",
" md_json_objs, download_path=\"data_images\"\n",
") # extract images from PDFs and save them to ./data_images/"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"# extract list of pages for insurance claim doc\n",
"md_json_list = []\n",
"for obj in md_json_objs:\n",
" md_json_list.extend(obj[\"pages\"])"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Create helper functions to create a list of `TextNode`s from the markdown tables to feed into the `VectorStoreIndex`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import re\n",
"from pathlib import Path\n",
"import typing as t\n",
"from llama_index.core.schema import TextNode, ImageNode\n",
"\n",
"\n",
"def get_page_number(file_name):\n",
" \"\"\"Gets page number of images using regex on file names\"\"\"\n",
" match = re.search(r\"-page-(\\d+)\\.jpg$\", str(file_name))\n",
" if match:\n",
" return int(match.group(1))\n",
" return 0\n",
"\n",
"\n",
"def _get_sorted_image_files(image_dir):\n",
" \"\"\"Get image files sorted by page.\"\"\"\n",
" raw_files = [f for f in list(Path(image_dir).iterdir()) if f.is_file()]\n",
" sorted_files = sorted(raw_files, key=get_page_number)\n",
" return sorted_files\n",
"\n",
"\n",
"def get_text_nodes(json_dicts, image_dir) -> t.List[TextNode]:\n",
" \"\"\"Creates nodes from json + images\"\"\"\n",
"\n",
" nodes = []\n",
"\n",
" docs = [doc[\"md\"] for doc in json_dicts] # extract text\n",
" image_files = _get_sorted_image_files(image_dir) # extract images\n",
"\n",
" for idx, doc in enumerate(docs):\n",
" # adds both a text node and the corresponding image node (jpg of the page) for each page\n",
" node = TextNode(\n",
" text=doc,\n",
" metadata={\"image_path\": str(image_files[idx]), \"page_num\": idx + 1},\n",
" )\n",
" image_node = ImageNode(\n",
" image_path=str(image_files[idx]),\n",
" metadata={\"page_num\": idx + 1, \"text_node_id\": node.id_},\n",
" )\n",
" nodes.extend([node, image_node])\n",
"\n",
" return nodes\n",
"\n",
"\n",
"text_nodes = get_text_nodes(md_json_list, \"data_images\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Index the documents."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import (\n",
" VectorStoreIndex,\n",
" StorageContext,\n",
" load_index_from_storage,\n",
" Settings,\n",
")\n",
"from llama_index.embeddings.openai import OpenAIEmbedding\n",
"from llama_index.llms.openai import OpenAI\n",
"\n",
"embed_model = OpenAIEmbedding(model=\"text-embedding-3-large\")\n",
"llm = OpenAI(\"gpt-4o\")\n",
"\n",
"Settings.llm = llm\n",
"Settings.embed_model = embed_model\n",
"\n",
"if not os.path.exists(\"storage_insurance\"):\n",
" index = VectorStoreIndex(text_nodes, embed_model=embed_model)\n",
" index.storage_context.persist(persist_dir=\"./storage_insurance\")\n",
"else:\n",
" ctx = StorageContext.from_defaults(persist_dir=\"./storage_insurance\")\n",
" index = load_index_from_storage(ctx)\n",
"\n",
"query_engine = index.as_query_engine()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Example queries are shown below."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"Michael Johnson filed the insurance claim for the accident that happened on Sunset Blvd."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"from IPython.display import display, Markdown\n",
"\n",
"response = query_engine.query(\n",
" \"Who filed the insurance claim for the accident that happened on Sunset Blvd?\"\n",
")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"Ms. Patel's accident occurred on March 10, 2023, at approximately 9:15 AM in the Boise Towne Square Mall parking lot. She was heading west at a parking space and, after checking her mirrors and blind spots, did not see any approaching vehicles. However, Michael Chen, the driver of another vehicle, was driving too fast through the parking lot and failed to stop in time, resulting in a collision with Ms. Patel's vehicle. This caused significant damage to the rear bumper and trunk of her car."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\"How did Ms. Patel's accident happen?\")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"Mr. Johnson's red sedan, a 2020 Honda Accord, was damaged on the front passenger side, including a dented fender and a broken headlight. The estimated repair cost is $3,500."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\"How was Mr. Johnson's red sedan damaged?\")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"Mr. Doe's Honda Accord sustained damage to the front bumper, hood, fenders, head/tail lights, windshield, and doors."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\"How was Mr. Doe's Honda Accord damaged?\")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"The witness for Ms. Patel's accident is Sophia Rodriguez. She can be contacted at 5554567890."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\n",
" \"Who are some witnesses for the Ms. Patel's accident and how can we contact them?\"\n",
")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"Yes, Ms. Johnson sustained injuries. She experienced minor injuries, including a bruised knee and some whiplash."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\n",
" \"Did Ms. Johnson sustain any injuries? If so, what were those injuries?\"\n",
")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"Mark Johnson is liable for the damages from the accident on Lombard Street. He was driving a delivery van that collided with the rear of Emily Rodriguez's vehicle. In rear-end collisions, the driver who hits the vehicle in front is typically at fault because they are expected to maintain a safe distance and be able to stop in time to avoid a collision."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"chat_engine = index.as_chat_engine()\n",
"response = chat_engine.chat(\n",
" \"Given the accident that happened on Lombard Street, name a party that is liable for the damages and explain why.\"\n",
")\n",
"display(Markdown(str(response)))"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama-parse-5ZmnAQ0r-py3.11",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -1,999 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"id": "93ae9bad-b8cc-43de-ba7d-387e0155674c",
"metadata": {},
"source": [
"# Building a Natively Multimodal RAG Pipeline (over a Slide Deck)\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/parse/multimodal/multimodal_rag_slide_deck.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"In this cookbook we show you how to build a multimodal RAG pipeline over a slide deck, with text, tables, images, diagrams, and complex layouts.\n",
"\n",
"A gap of text-based RAG is that they struggle with purely text-based representations of complex documents. For instance, if a page contains a lot of images and diagrams, a text parser would need to rely on raw OCR to extract out text. You can also use a multimodal model (e.g. gpt-4o and up) to do text extraction, but this is inherently a lossy conversion.\n",
"\n",
"Instead a **native multimodal pipeline** stores both a text and image representation of a document chunk. They are indexed via embeddings (text or image), and during synthesis both text and image are directly fed to the multimodal model for synthesis.\n",
"\n",
"This can have the following advantages:\n",
"- **Robustness**: This solution is more robust than a pure text or even a pure image-based approach. In a pure text RAG approach, the parsing piece can be lossy. In a pure image-based approach, multimodal OCR is not perfect and may lose out against text parsing for text-heavy documents.\n",
"- **Cost Optimization**: You may choose to dynamically include text-only, or text + image depending on the content of the page.\n",
"\n",
"![mm_rag_diagram](./multimodal_rag_slide_deck_img.png)"
]
},
{
"cell_type": "markdown",
"id": "54e8d9a7-5036-4d32-818f-00b2e888521f",
"metadata": {},
"source": [
"## Setup"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "70ccdd53-e68a-4199-aacb-cfe71ad1ff0b",
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()"
]
},
{
"cell_type": "markdown",
"id": "225c5556-a789-4386-a1ee-cce01dbeb6cf",
"metadata": {},
"source": [
"### Setup Observability\n",
"\n",
"We setup an integration with LlamaTrace (integration with Arize).\n",
"\n",
"If you haven't already done so, make sure to create an account here: https://llamatrace.com/login. Then create an API key and put it in the `PHOENIX_API_KEY` variable below."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0eabee1f-290a-4c85-b362-54f45c8559ae",
"metadata": {},
"outputs": [],
"source": [
"!pip install -U llama-index-callbacks-arize-phoenix"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "aaeb245c-730b-4c34-ad68-708fdde0e6cb",
"metadata": {},
"outputs": [],
"source": [
"# setup Arize Phoenix for logging/observability\n",
"import llama_index.core\n",
"import os\n",
"\n",
"PHOENIX_API_KEY = \"<PHOENIX_API_KEY>\"\n",
"os.environ[\"OTEL_EXPORTER_OTLP_HEADERS\"] = f\"api_key={PHOENIX_API_KEY}\"\n",
"llama_index.core.set_global_handler(\n",
" \"arize_phoenix\", endpoint=\"https://llamatrace.com/v1/traces\"\n",
")"
]
},
{
"cell_type": "markdown",
"id": "fbb362db-b1b1-4eea-be1a-b1f78b0779d7",
"metadata": {},
"source": [
"### Load Data\n",
"\n",
"Here we load the [Conoco Phillips 2023 investor meeting slide deck](https://static.conocophillips.com/files/2023-conocophillips-aim-presentation.pdf)."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8bce3407-a7d2-47e8-9eaf-ab297a94750c",
"metadata": {},
"outputs": [],
"source": [
"!mkdir data\n",
"!mkdir data_images\n",
"!wget \"https://static.conocophillips.com/files/2023-conocophillips-aim-presentation.pdf\" -O data/conocophillips.pdf"
]
},
{
"cell_type": "markdown",
"id": "246ba6b0-51af-42f9-b1b2-8d3e721ef782",
"metadata": {},
"source": [
"### Model Setup\n",
"\n",
"Setup models that will be used for downstream orchestration."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "16e2071d-bbc2-4707-8ae7-cb4e1fecafd3",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import Settings\n",
"from llama_index.llms.openai import OpenAI\n",
"from llama_index.embeddings.openai import OpenAIEmbedding\n",
"\n",
"embed_model = OpenAIEmbedding(model=\"text-embedding-3-large\")\n",
"llm = OpenAI(model=\"gpt-4o\")\n",
"\n",
"Settings.embed_model = embed_model\n",
"Settings.llm = llm"
]
},
{
"cell_type": "markdown",
"id": "e3f6416f-f580-4722-aaa9-7f3500408547",
"metadata": {},
"source": [
"## Use LlamaParse to Parse Text and Images\n",
"\n",
"In this example, use LlamaParse to parse both the text and images from the document.\n",
"\n",
"We parse out the text in two ways: \n",
"- in regular `text` mode using our default text layout algorithm\n",
"- in `markdown` mode using GPT-4o (`gpt4o_mode=True`). This also allows us to capture page screenshots"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "570089e5-238a-4dcc-af65-96e7393c2b4d",
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"\n",
"parser_text = LlamaParse(result_type=\"text\")\n",
"parser_gpt4o = LlamaParse(result_type=\"markdown\", gpt4o_mode=True)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "ef82a985-4088-4bb7-9a21-0318e1b9207d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Parsing text...\n",
"Started parsing the file under job_id 62f157a9-9ef9-4e5b-95ac-67093fa25800\n",
"..........Parsing PDF file...\n",
"Started parsing the file under job_id 1ddd5654-062b-4e19-b488-d66efc9c509d\n"
]
}
],
"source": [
"print(f\"Parsing text...\")\n",
"docs_text = parser_text.load_data(\"data/conocophillips.pdf\")\n",
"print(f\"Parsing PDF file...\")\n",
"md_json_objs = parser_gpt4o.get_json_result(\"data/conocophillips.pdf\")\n",
"md_json_list = md_json_objs[0][\"pages\"]"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5318fb7b-fe6a-4a8a-b82e-4ed7b4512c37",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# Commitment to Disciplined Reinvestment Rate\n",
"\n",
"| Period | Description | Reinvestment Rate | WTI Average |\n",
"|--------------|--------------------------------------|-------------------|-------------|\n",
"| 2012-2016 | Industry Growth Focus | >100% | ~$75/BBL |\n",
"| 2017-2022 | ConocoPhillips Strategy Reset | <60% | ~$63/BBL |\n",
"| 2023E | | | at $80/BBL |\n",
"| 2024-2028 | Disciplined Reinvestment Rate | ~50% | at $60/BBL |\n",
"| 2029-2032 | | ~6% CFO CAGR | at $60/BBL |\n",
"\n",
"- **Historic Reinvestment Rate**: Gray bars\n",
"- **Reinvestment Rate at $60/BBL WTI**: Blue bars\n",
"- **Reinvestment Rate at $80/BBL WTI**: Dashed blue lines\n",
"\n",
"Reinvestment rate and cash from operations (CFO) are non-GAAP measures. Definitions and reconciliations are included in the Appendix.\n"
]
}
],
"source": [
"print(md_json_list[10][\"md\"])"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "eeadb16c-97eb-4622-9551-b34d7f90d72f",
"metadata": {},
"outputs": [],
"source": [
"image_dicts = parser_gpt4o.get_images(md_json_objs, download_path=\"data_images\")"
]
},
{
"cell_type": "markdown",
"id": "fd3e098b-0606-4429-b48d-d4fe0140fc0e",
"metadata": {},
"source": [
"## Build Multimodal Index\n",
"\n",
"In this section we build the multimodal index over the parsed deck. \n",
"\n",
"We do this by creating **text** nodes from the document that contain metadata referencing the original image path.\n",
"\n",
"In this example we're indexing the text node for retrieval. The text node has a reference to both the parsed text as well as the image screenshot."
]
},
{
"cell_type": "markdown",
"id": "3aae2dee-9d85-4604-8a51-705d4db527f7",
"metadata": {},
"source": [
"#### Get Text Nodes"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "18c24174-05ce-417f-8dd2-79c3f375db03",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.schema import TextNode\n",
"from typing import Optional"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "8e331dfe-a627-4e23-8c57-70ab1d9342e4",
"metadata": {},
"outputs": [],
"source": [
"# get pages loaded through llamaparse\n",
"import re\n",
"\n",
"\n",
"def get_page_number(file_name):\n",
" match = re.search(r\"-page-(\\d+)\\.jpg$\", str(file_name))\n",
" if match:\n",
" return int(match.group(1))\n",
" return 0\n",
"\n",
"\n",
"def _get_sorted_image_files(image_dir):\n",
" \"\"\"Get image files sorted by page.\"\"\"\n",
" raw_files = [f for f in list(Path(image_dir).iterdir()) if f.is_file()]\n",
" sorted_files = sorted(raw_files, key=get_page_number)\n",
" return sorted_files"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "346fe5ef-171e-4a54-9084-7a7805103a13",
"metadata": {},
"outputs": [],
"source": [
"from copy import deepcopy\n",
"from pathlib import Path\n",
"\n",
"\n",
"# attach image metadata to the text nodes\n",
"def get_text_nodes(docs, image_dir=None, json_dicts=None):\n",
" \"\"\"Split docs into nodes, by separator.\"\"\"\n",
" nodes = []\n",
"\n",
" image_files = _get_sorted_image_files(image_dir) if image_dir is not None else None\n",
" md_texts = [d[\"md\"] for d in json_dicts] if json_dicts is not None else None\n",
"\n",
" doc_chunks = [c for d in docs for c in d.text.split(\"---\")]\n",
" for idx, doc_chunk in enumerate(doc_chunks):\n",
" chunk_metadata = {\"page_num\": idx + 1}\n",
" if image_files is not None:\n",
" image_file = image_files[idx]\n",
" chunk_metadata[\"image_path\"] = str(image_file)\n",
" if md_texts is not None:\n",
" chunk_metadata[\"parsed_text_markdown\"] = md_texts[idx]\n",
" chunk_metadata[\"parsed_text\"] = doc_chunk\n",
" node = TextNode(\n",
" text=\"\",\n",
" metadata=chunk_metadata,\n",
" )\n",
" nodes.append(node)\n",
"\n",
" return nodes"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f591669c-5a8e-491d-9cef-0b754abbf26f",
"metadata": {},
"outputs": [],
"source": [
"# this will split into pages\n",
"text_nodes = get_text_nodes(docs_text, image_dir=\"data_images\", json_dicts=md_json_list)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "32c13950-c1db-435f-b5b4-89d62b8b7744",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page_num: 11\n",
"image_path: data_images/1ddd5654-062b-4e19-b488-d66efc9c509d-page_39.jpg\n",
"parsed_text_markdown: # Commitment to Disciplined Reinvestment Rate\n",
"\n",
"| Period | Description | Reinvestment Rate | WTI Average |\n",
"|--------------|--------------------------------------|-------------------|-------------|\n",
"| 2012-2016 | Industry Growth Focus | >100% | ~$75/BBL |\n",
"| 2017-2022 | ConocoPhillips Strategy Reset | <60% | ~$63/BBL |\n",
"| 2023E | | | at $80/BBL |\n",
"| 2024-2028 | Disciplined Reinvestment Rate | ~50% | at $60/BBL |\n",
"| 2029-2032 | | ~6% CFO CAGR | at $60/BBL |\n",
"\n",
"- **Historic Reinvestment Rate**: Gray bars\n",
"- **Reinvestment Rate at $60/BBL WTI**: Blue bars\n",
"- **Reinvestment Rate at $80/BBL WTI**: Dashed blue lines\n",
"\n",
"Reinvestment rate and cash from operations (CFO) are non-GAAP measures. Definitions and reconciliations are included in the Appendix.\n",
"parsed_text: Commitment to Disciplined Reinvestment Rate\n",
" Industry ConocoPhillips\n",
" Strategy Reset Disciplined Reinvestment Rate is the Foundation for Superior\n",
" Growth Focus Returns on and of Capital, while Driving Durable CFO Growth\n",
" 100% <60% 50% 6% at $60/BBL WTI\n",
" Reinvestment Rate Reinvestment Rate Reinvestment Rate10-YearCFO CAGR Planning PriceMid-Cycle\n",
" 2024-2032\n",
" 2 100%\n",
" 1 75%\n",
" 1 50%\n",
" 1 WTIat $80/BBL at S80/BBL\n",
" 25% 'S75/BBL $63/BBL WTI\n",
" WTI WTI at S80/BBL at S60/BBL at S60/BBL\n",
" Average Average WTI WTI WTI\n",
" 0%\n",
" 2012-2016 2017-2022 2023E 2024-2028 2029-2032\n",
" Historic Reinvestment Rate Reinvestment Rate at $60/BBL WTI Reinvestment Rate at $80/BBL WTI\n",
" Reinvestment rate and cash from operations (CFO) are non-GAAP measures: Definitions and reconciliations are included in the Appendix ConocoPhillips\n"
]
}
],
"source": [
"print(text_nodes[10].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "markdown",
"id": "4f404f56-db1e-4ed7-9ba1-ead763546348",
"metadata": {},
"source": [
"#### Build Index\n",
"\n",
"Once the text nodes are ready, we feed into our vector store index abstraction, which will index these nodes into a simple in-memory vector store (of course, you should definitely check out our 40+ vector store integrations!)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "6ea53c31-0e38-421c-8d9b-0e3adaa1677e",
"metadata": {},
"outputs": [
{
"name": "stderr",
"output_type": "stream",
"text": [
"/Users/jerryliu/Programming/gpt_index/.venv/lib/python3.10/site-packages/tiktoken/core.py:50: RuntimeWarning: coroutine 'LlamaParse.aload_data' was never awaited\n",
" self._core_bpe = _tiktoken.CoreBPE(mergeable_ranks, special_tokens, pat_str)\n",
"RuntimeWarning: Enable tracemalloc to get the object allocation traceback\n"
]
}
],
"source": [
"import os\n",
"from llama_index.core import (\n",
" StorageContext,\n",
" VectorStoreIndex,\n",
" load_index_from_storage,\n",
")\n",
"\n",
"if not os.path.exists(\"storage_nodes\"):\n",
" index = VectorStoreIndex(text_nodes, embed_model=embed_model)\n",
" # save index to disk\n",
" index.set_index_id(\"vector_index\")\n",
" index.storage_context.persist(\"./storage_nodes\")\n",
"else:\n",
" # rebuild storage context\n",
" storage_context = StorageContext.from_defaults(persist_dir=\"storage_nodes\")\n",
" # load index\n",
" index = load_index_from_storage(storage_context, index_id=\"vector_index\")\n",
"\n",
"retriever = index.as_retriever()"
]
},
{
"cell_type": "markdown",
"id": "5f0e33a4-9422-498d-87ee-d917bdf74d80",
"metadata": {},
"source": [
"## Build Multimodal Query Engine\n",
"\n",
"We now use LlamaIndex abstractions to build a **custom query engine**. In contrast to a standard RAG query engine that will retrieve the text node and only put that into the prompt (response synthesis module), this custom query engine will also load the image document, and put both the text and image document into the response synthesis module."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "35a94be2-e289-41a6-92e4-d3cb428fb0c8",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.query_engine import CustomQueryEngine, SimpleMultiModalQueryEngine\n",
"from llama_index.core.retrievers import BaseRetriever\n",
"from llama_index.multi_modal_llms.openai import OpenAIMultiModal\n",
"from llama_index.core.schema import ImageNode, NodeWithScore, MetadataMode\n",
"from llama_index.core.prompts import PromptTemplate\n",
"from llama_index.core.base.response.schema import Response\n",
"from typing import Optional\n",
"\n",
"\n",
"gpt_4o = OpenAIMultiModal(model=\"gpt-4o\", max_new_tokens=4096)\n",
"\n",
"QA_PROMPT_TMPL = \"\"\"\\\n",
"Below we give parsed text from slides in two different formats, as well as the image.\n",
"\n",
"We parse the text in both 'markdown' mode as well as 'raw text' mode. Markdown mode attempts \\\n",
"to convert relevant diagrams into tables, whereas raw text tries to maintain the rough spatial \\\n",
"layout of the text.\n",
"\n",
"Use the image information first and foremost. ONLY use the text/markdown information \n",
"if you can't understand the image.\n",
"\n",
"---------------------\n",
"{context_str}\n",
"---------------------\n",
"Given the context information and not prior knowledge, answer the query. Explain whether you got the answer\n",
"from the parsed markdown or raw text or image, and if there's discrepancies, and your reasoning for the final answer.\n",
"\n",
"Query: {query_str}\n",
"Answer: \"\"\"\n",
"\n",
"QA_PROMPT = PromptTemplate(QA_PROMPT_TMPL)\n",
"\n",
"\n",
"class MultimodalQueryEngine(CustomQueryEngine):\n",
" \"\"\"Custom multimodal Query Engine.\n",
"\n",
" Takes in a retriever to retrieve a set of document nodes.\n",
" Also takes in a prompt template and multimodal model.\n",
"\n",
" \"\"\"\n",
"\n",
" qa_prompt: PromptTemplate\n",
" retriever: BaseRetriever\n",
" multi_modal_llm: OpenAIMultiModal\n",
"\n",
" def __init__(self, qa_prompt: Optional[PromptTemplate] = None, **kwargs) -> None:\n",
" \"\"\"Initialize.\"\"\"\n",
" super().__init__(qa_prompt=qa_prompt or QA_PROMPT, **kwargs)\n",
"\n",
" def custom_query(self, query_str: str):\n",
" # retrieve text nodes\n",
" nodes = self.retriever.retrieve(query_str)\n",
" # create ImageNode items from text nodes\n",
" image_nodes = [\n",
" NodeWithScore(node=ImageNode(image_path=n.metadata[\"image_path\"]))\n",
" for n in nodes\n",
" ]\n",
"\n",
" # create context string from text nodes, dump into the prompt\n",
" context_str = \"\\n\\n\".join(\n",
" [r.get_content(metadata_mode=MetadataMode.LLM) for r in nodes]\n",
" )\n",
" fmt_prompt = self.qa_prompt.format(context_str=context_str, query_str=query_str)\n",
"\n",
" # synthesize an answer from formatted text and images\n",
" llm_response = self.multi_modal_llm.complete(\n",
" prompt=fmt_prompt,\n",
" image_documents=[image_node.node for image_node in image_nodes],\n",
" )\n",
" return Response(\n",
" response=str(llm_response),\n",
" source_nodes=nodes,\n",
" metadata={\"text_nodes\": text_nodes, \"image_nodes\": image_nodes},\n",
" )\n",
"\n",
" return response"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "0890be59-fb12-4bb5-959b-b2d9600f7774",
"metadata": {},
"outputs": [],
"source": [
"query_engine = MultimodalQueryEngine(\n",
" retriever=index.as_retriever(similarity_top_k=9), multi_modal_llm=gpt_4o\n",
")"
]
},
{
"cell_type": "markdown",
"id": "a92aa4f1-7501-4711-b054-f02338e54e74",
"metadata": {},
"source": [
"### Define Baseline\n",
"\n",
"In addition, we define a \"baseline\" where we rely only on text-based indexing. Here we define an index using only the nodes that are parsed in text-mode from LlamaParse. \n",
"\n",
"**NOTE**: We don't currently include the markdown-parsed text because that was parsed with GPT-4o, so already uses a multimodal model during the text extraction phase.\n",
"\n",
"It is of course a valid experiment to compare RAG where multimodal extraction only happens during indexing, vs. the current multimodal RAG implementation where images are fed during synthesis to the LLM. "
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "c0b15a48-d177-4666-aec2-98ee90664642",
"metadata": {},
"outputs": [],
"source": [
"def get_nodes(docs):\n",
" \"\"\"Split docs into nodes, by separator.\"\"\"\n",
" nodes = []\n",
" for doc in docs:\n",
" doc_chunks = doc.text.split(\"\\n---\\n\")\n",
" for doc_chunk in doc_chunks:\n",
" node = TextNode(\n",
" text=doc_chunk,\n",
" metadata=deepcopy(doc.metadata),\n",
" )\n",
" nodes.append(node)\n",
"\n",
" return nodes"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2065d2c6-d6ba-4ee3-8e9e-dbc83cbcec1b",
"metadata": {},
"outputs": [],
"source": [
"base_nodes = get_nodes(docs_text)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "bcaea1a8-26c9-4385-8f62-32855aa898b6",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Our Differentiated Portfolio: Deep; Durable and Diverse\n",
" 20 BBOE of Resource Diverse Production Base\n",
" Under $40/BBL Cost of Supply 10-Year Plan Cumulative Production (BBOE)\n",
" S50 S32/BBL Lower 48 Alaska\n",
" Average Cost of Supply\n",
" 3 $40 GKA GWA\n",
" GPA WNS\n",
" $30 EMENA\n",
" 3 Norway\n",
" 8 $20\n",
" E Qatar Libya\n",
" Asia Pacific Canada\n",
" $10 Permian\n",
" APLNG Montney\n",
" S0\n",
" 10 15 20 Bakken\n",
" Resource (BBOE) Eagle Ford Other Malaysia ChinaSurmont\n",
" Lower 48 Canada Alaska EMENA Asia Pacific\n",
"Costs assumemid-cycle price environment of S60/BBL WTI:\n",
" ConocoPhillips\n"
]
}
],
"source": [
"print(base_nodes[13].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "f6bcfbc6-4e9b-41ad-ad81-1c4245b95cd5",
"metadata": {},
"outputs": [],
"source": [
"base_index = VectorStoreIndex(base_nodes, embed_model=embed_model)\n",
"base_query_engine = base_index.as_query_engine(llm=llm, similarity_top_k=9)"
]
},
{
"cell_type": "markdown",
"id": "1f94ef26-0df5-4468-a156-903d686f02ce",
"metadata": {},
"source": [
"## Build a Multimodal Agent\n",
"\n",
"Build an agent around the multimodal query engine. This gives you agent capabilities like query planning/decomposition and memory around a central QA interface."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "5b7a8c5f-39fc-4d04-8c56-3642f5718437",
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.tools import QueryEngineTool\n",
"from llama_index.core.agent import FunctionCallingAgentWorker\n",
"\n",
"\n",
"vector_tool = QueryEngineTool.from_defaults(\n",
" query_engine=query_engine,\n",
" name=\"vector_tool\",\n",
" description=(\n",
" \"Useful for retrieving specific context from the data. Do NOT select if question asks for a summary of the data.\"\n",
" ),\n",
")\n",
"agent = FunctionCallingAgentWorker.from_tools(\n",
" [vector_tool], llm=llm, verbose=True\n",
").as_agent()"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "2b4f7eb1-d247-45fa-bb41-c02fc353a22a",
"metadata": {},
"outputs": [],
"source": [
"# define a similar agent for the baseline\n",
"base_vector_tool = QueryEngineTool.from_defaults(\n",
" query_engine=base_query_engine,\n",
" name=\"vector_tool\",\n",
" description=(\n",
" \"Useful for retrieving specific context from the data. Do NOT select if question asks for a summary of the data.\"\n",
" ),\n",
")\n",
"base_agent = FunctionCallingAgentWorker.from_tools(\n",
" [base_vector_tool], llm=llm, verbose=True\n",
").as_agent()"
]
},
{
"cell_type": "markdown",
"id": "2336f98b-c0a1-413a-849d-8a89bacb90b5",
"metadata": {},
"source": [
"## Try out Queries\n",
"\n",
"Let's try out queries against these documents and compare against each other."
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d78e53cf-35cb-4ef8-b03e-1b47ba15ae64",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Added user message to memory: Tell me about the diverse geographies where Conoco Phillips has a production base\n",
"=== Calling Function ===\n",
"Calling function: vector_tool with args: {\"input\": \"Conoco Phillips production base geographies\"}\n",
"=== Function Output ===\n",
"ConocoPhillips' production base geographies include:\n",
"\n",
"1. **Lower 48** (Permian, Eagle Ford, Bakken, Other)\n",
"2. **Alaska** (GKA, GWA, GPA, WNS)\n",
"3. **EMENA** (Norway, Libya, Qatar)\n",
"4. **Asia Pacific** (APLNG, Malaysia, China)\n",
"5. **Canada** (Montney, Surmont)\n",
"\n",
"This information was derived from the image on page 14, which provides a detailed breakdown of the diverse production base and the regions involved. The parsed markdown and raw text also support this information, but the image provides the clearest and most comprehensive view. There are no discrepancies between the image and the parsed text in this case.\n",
"=== LLM Response ===\n",
"ConocoPhillips has a diverse production base spread across various geographies, including:\n",
"\n",
"1. **Lower 48**:\n",
" - Permian Basin\n",
" - Eagle Ford\n",
" - Bakken\n",
" - Other regions within the continental United States\n",
"\n",
"2. **Alaska**:\n",
" - Greater Kuparuk Area (GKA)\n",
" - Greater Prudhoe Area (GPA)\n",
" - Greater Willow Area (GWA)\n",
" - Western North Slope (WNS)\n",
"\n",
"3. **EMENA (Europe, Middle East, and North Africa)**:\n",
" - Norway\n",
" - Libya\n",
" - Qatar\n",
"\n",
"4. **Asia Pacific**:\n",
" - Australia Pacific LNG (APLNG)\n",
" - Malaysia\n",
" - China\n",
"\n",
"5. **Canada**:\n",
" - Montney\n",
" - Surmont\n",
"\n",
"These regions highlight the global reach and diverse geographical footprint of ConocoPhillips' production operations.\n",
"Added user message to memory: Tell me about the diverse geographies where Conoco Phillips has a production base\n",
"=== Calling Function ===\n",
"Calling function: vector_tool with args: {\"input\": \"diverse geographies where Conoco Phillips has a production base\"}\n",
"=== Function Output ===\n",
"ConocoPhillips has a diverse production base that includes the Lower 48 (Permian, Bakken, Eagle Ford), Alaska, Canada (Montney, Surmont), EMENA (Norway, Libya), Asia Pacific (Malaysia, China, APLNG), and Qatar.\n",
"=== LLM Response ===\n",
"ConocoPhillips has a diverse production base spanning several key geographies:\n",
"\n",
"1. **Lower 48 (United States)**: This includes major production areas such as the Permian Basin, Bakken Formation, and Eagle Ford Shale.\n",
"2. **Alaska**: Significant operations in the North Slope region.\n",
"3. **Canada**: Operations in the Montney Formation and the Surmont oil sands project.\n",
"4. **EMENA (Europe, Middle East, and North Africa)**: Notable operations in Norway and Libya.\n",
"5. **Asia Pacific**: Includes operations in Malaysia, China, and the Australia Pacific LNG (APLNG) project.\n",
"6. **Qatar**: Involvement in the country's energy sector.\n",
"\n",
"These regions highlight the company's extensive and varied geographical footprint in the energy production industry.\n"
]
}
],
"source": [
"query = (\n",
" \"Tell me about the diverse geographies where Conoco Phillips has a production base\"\n",
")\n",
"response = agent.query(query)\n",
"base_response = base_agent.query(query)"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "355d2aa4-c26f-480e-b512-4446acbd9227",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ConocoPhillips has a diverse production base spread across various geographies, including:\n",
"\n",
"1. **Lower 48**:\n",
" - Permian Basin\n",
" - Eagle Ford\n",
" - Bakken\n",
" - Other regions within the continental United States\n",
"\n",
"2. **Alaska**:\n",
" - Greater Kuparuk Area (GKA)\n",
" - Greater Prudhoe Area (GPA)\n",
" - Greater Willow Area (GWA)\n",
" - Western North Slope (WNS)\n",
"\n",
"3. **EMENA (Europe, Middle East, and North Africa)**:\n",
" - Norway\n",
" - Libya\n",
" - Qatar\n",
"\n",
"4. **Asia Pacific**:\n",
" - Australia Pacific LNG (APLNG)\n",
" - Malaysia\n",
" - China\n",
"\n",
"5. **Canada**:\n",
" - Montney\n",
" - Surmont\n",
"\n",
"These regions highlight the global reach and diverse geographical footprint of ConocoPhillips' production operations.\n"
]
}
],
"source": [
"print(str(response))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d584c560-8f49-4c10-a4db-2e0d3b7085d2",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"page_num: 14\n",
"image_path: data_images/1ddd5654-062b-4e19-b488-d66efc9c509d-page_12.jpg\n",
"parsed_text_markdown: # Our Differentiated Portfolio: Deep, Durable and Diverse\n",
"\n",
"## ~20 BBOE of Resource\n",
"Under $40/BBL Cost of Supply\n",
"\n",
"### ~ $32/BBL\n",
"Average Cost of Supply\n",
"\n",
"### WTI Cost of Supply ($/BBL)\n",
"\n",
"| Cost ($/BBL) | Resource (BBOE) |\n",
"|--------------|-----------------|\n",
"| $0 | 0 |\n",
"| $10 | |\n",
"| $20 | |\n",
"| $30 | |\n",
"| $40 | |\n",
"| $50 | |\n",
"\n",
"- **Legend:**\n",
" - Lower 48\n",
" - Canada\n",
" - Alaska\n",
" - EMENA\n",
" - Asia Pacific\n",
"\n",
"*Costs assume a mid-cycle price environment of $60/BBL WTI.*\n",
"\n",
"## Diverse Production Base\n",
"10-Year Plan Cumulative Production (BBOE)\n",
"\n",
"| Region | Sub-region |\n",
"|--------------|-----------------|\n",
"| Lower 48 | Permian |\n",
"| | Eagle Ford |\n",
"| | Bakken |\n",
"| | Other |\n",
"| Alaska | GKA |\n",
"| | GWA |\n",
"| | GPA |\n",
"| | WNS |\n",
"| EMENA | Norway |\n",
"| | Libya |\n",
"| | Qatar |\n",
"| Asia Pacific | APLNG |\n",
"| | Malaysia |\n",
"| | China |\n",
"| Canada | Montney |\n",
"| | Surmont |\n",
"parsed_text: Our Differentiated Portfolio: Deep; Durable and Diverse\n",
" 20 BBOE of Resource Diverse Production Base\n",
" Under $40/BBL Cost of Supply 10-Year Plan Cumulative Production (BBOE)\n",
" S50 S32/BBL Lower 48 Alaska\n",
" Average Cost of Supply\n",
" 3 $40 GKA GWA\n",
" GPA WNS\n",
" $30 EMENA\n",
" 3 Norway\n",
" 8 $20\n",
" E Qatar Libya\n",
" Asia Pacific Canada\n",
" $10 Permian\n",
" APLNG Montney\n",
" S0\n",
" 10 15 20 Bakken\n",
" Resource (BBOE) Eagle Ford Other Malaysia ChinaSurmont\n",
" Lower 48 Canada Alaska EMENA Asia Pacific\n",
"Costs assumemid-cycle price environment of S60/BBL WTI:\n",
" ConocoPhillips\n"
]
}
],
"source": [
"print(response.source_nodes[7].get_content(metadata_mode=\"all\"))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d21d694b-6618-4d04-a6f6-8b0c2625f539",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ConocoPhillips has a diverse production base spanning several key geographies:\n",
"\n",
"1. **Lower 48 (United States)**: This includes major production areas such as the Permian Basin, Bakken Formation, and Eagle Ford Shale.\n",
"2. **Alaska**: Significant operations in the North Slope region.\n",
"3. **Canada**: Operations in the Montney Formation and the Surmont oil sands project.\n",
"4. **EMENA (Europe, Middle East, and North Africa)**: Notable operations in Norway and Libya.\n",
"5. **Asia Pacific**: Includes operations in Malaysia, China, and the Australia Pacific LNG (APLNG) project.\n",
"6. **Qatar**: Involvement in the country's energy sector.\n",
"\n",
"These regions highlight the company's extensive and varied geographical footprint in the energy production industry.\n"
]
}
],
"source": [
"print(str(base_response))"
]
},
{
"cell_type": "code",
"execution_count": null,
"id": "d3afccae-ad8d-4c5d-9d93-810dba413a5d",
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Our Differentiated Portfolio: Deep; Durable and Diverse\n",
" 20 BBOE of Resource Diverse Production Base\n",
" Under $40/BBL Cost of Supply 10-Year Plan Cumulative Production (BBOE)\n",
" S50 S32/BBL Lower 48 Alaska\n",
" Average Cost of Supply\n",
" 3 $40 GKA GWA\n",
" GPA WNS\n",
" $30 EMENA\n",
" 3 Norway\n",
" 8 $20\n",
" E Qatar Libya\n",
" Asia Pacific Canada\n",
" $10 Permian\n",
" APLNG Montney\n",
" S0\n",
" 10 15 20 Bakken\n",
" Resource (BBOE) Eagle Ford Other Malaysia ChinaSurmont\n",
" Lower 48 Canada Alaska EMENA Asia Pacific\n",
"Costs assumemid-cycle price environment of S60/BBL WTI:\n",
" ConocoPhillips\n"
]
}
],
"source": [
"print(base_response.source_nodes[1].get_content(metadata_mode=\"all\"))"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama_index_v3",
"language": "python",
"name": "llama_index_v3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 5
}
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -1,834 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Building a RAG Pipeline over IKEA Product Instruction Manuals\n",
"\n",
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/parse/multimodal/product_manual_rag.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"This cookbook shows how to use LlamaParse and OpenAI's multimodal models to query over IKEA instruction manual PDFs, which mainly contain images and diagrams to show how one can assemble the product.\n",
"\n",
"LlamaParse and multimodal LLMs can interpret these diagrams and translate them into textual instructions. With textual assistance, confusing visual instructions within the IKEA product manuals can be made easier to understand and interpret. Additionally, textual instructions can be helpful for those who are visually impaired."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Install and Setup\n",
"\n",
"Install LlamaIndex, download the data, and apply `nest_asyncio`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-index llama-parse llama-index-multi-modal-llms-openai git+https://github.com/openai/CLIP.git"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!wget https://github.com/user-attachments/files/16461058/data.zip -O data.zip\n",
"!unzip -o data.zip\n",
"!rm data.zip"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Set up your OpenAI and LlamaCloud keys."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"<Your OpenAI API Key>\"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"<Your LlamaCloud API Key>\""
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Code Implementation"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Set up LlamaParse. We will parse the PDF files into markdown and use the GPT-4o multimodal model to parse the PDFs."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Load data from the parser."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"parser = LlamaParse(\n",
" result_type=\"markdown\",\n",
" parsing_instruction=\"You are given IKEA assembly instruction manuals\",\n",
" use_vendor_multimodal_model=True,\n",
" vendor_multimodal_model_name=\"openai-gpt4o\",\n",
" show_progress=True,\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"DATA_DIR = \"data\"\n",
"\n",
"\n",
"def get_data_files(data_dir=DATA_DIR) -> list[str]:\n",
" files = []\n",
" for f in os.listdir(data_dir):\n",
" fname = os.path.join(data_dir, f)\n",
" if os.path.isfile(fname):\n",
" files.append(fname)\n",
" return files\n",
"\n",
"\n",
"files = get_data_files()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Load data into docs, and save images from PDFs into `data_images` directory."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"md_json_objs = parser.get_json_result(files)\n",
"md_json_list = md_json_objs[0][\"pages\"]\n",
"image_dicts = parser.get_images(md_json_objs, download_path=\"data_images\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Create helper functions to create a list of `TextNode`s from the markdown tables to feed into the `VectorStoreIndex`."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import re\n",
"from pathlib import Path\n",
"import typing as t\n",
"from llama_index.core.schema import TextNode\n",
"\n",
"\n",
"def get_page_number(file_name):\n",
" \"\"\"Gets page number of images using regex on file names\"\"\"\n",
" match = re.search(r\"-page-(\\d+)\\.jpg$\", str(file_name))\n",
" if match:\n",
" return int(match.group(1))\n",
" return 0\n",
"\n",
"\n",
"def _get_sorted_image_files(image_dir):\n",
" \"\"\"Get image files sorted by page.\"\"\"\n",
" raw_files = [f for f in list(Path(image_dir).iterdir()) if f.is_file()]\n",
" sorted_files = sorted(raw_files, key=get_page_number)\n",
" return sorted_files\n",
"\n",
"\n",
"def get_text_nodes(json_dicts, image_dir) -> t.List[TextNode]:\n",
" \"\"\"Creates nodes from json + images\"\"\"\n",
"\n",
" nodes = []\n",
"\n",
" docs = [doc[\"md\"] for doc in json_dicts] # extract text\n",
" image_files = _get_sorted_image_files(image_dir) # extract images\n",
"\n",
" for idx, doc in enumerate(docs):\n",
" # adds both a text node and the corresponding image node (jpg of the page) for each page\n",
" node = TextNode(\n",
" text=doc,\n",
" metadata={\"image_path\": str(image_files[idx]), \"page_num\": idx + 1},\n",
" )\n",
" nodes.append(node)\n",
"\n",
" return nodes\n",
"\n",
"\n",
"text_nodes = get_text_nodes(md_json_list, \"data_images\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Index the documents."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import (\n",
" VectorStoreIndex,\n",
" StorageContext,\n",
" load_index_from_storage,\n",
" Settings,\n",
")\n",
"from llama_index.embeddings.openai import OpenAIEmbedding\n",
"from llama_index.llms.openai import OpenAI\n",
"\n",
"embed_model = OpenAIEmbedding(model=\"text-embedding-3-large\")\n",
"llm = OpenAI(\"gpt-4o\")\n",
"\n",
"Settings.llm = llm\n",
"Settings.embed_model = embed_model\n",
"\n",
"if not os.path.exists(\"storage_ikea\"):\n",
" index = VectorStoreIndex(text_nodes, embed_model=embed_model)\n",
" index.storage_context.persist(persist_dir=\"./storage_ikea\")\n",
"else:\n",
" ctx = StorageContext.from_defaults(persist_dir=\"./storage_ikea\")\n",
" index = load_index_from_storage(ctx)\n",
"\n",
"retriever = index.as_retriever()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Create a custom query engine that uses GPT-4o's multimodal model."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.query_engine import CustomQueryEngine\n",
"from llama_index.core.retrievers import BaseRetriever\n",
"from llama_index.multi_modal_llms.openai import OpenAIMultiModal\n",
"from llama_index.core.schema import NodeWithScore, MetadataMode\n",
"from llama_index.core.base.response.schema import Response\n",
"from llama_index.core.prompts import PromptTemplate\n",
"from llama_index.core.schema import ImageNode\n",
"\n",
"QA_PROMPT_TMPL = \"\"\"\\\n",
"Below we give parsed text from slides in two different formats, as well as the image.\n",
"\n",
"We parse the text in both 'markdown' mode as well as 'raw text' mode. Markdown mode attempts \\\n",
"to convert relevant diagrams into tables, whereas raw text tries to maintain the rough spatial \\\n",
"layout of the text.\n",
"\n",
"Use the image information first and foremost. ONLY use the text/markdown information \n",
"if you can't understand the image.\n",
"\n",
"---------------------\n",
"{context_str}\n",
"---------------------\n",
"Given the context information and not prior knowledge, answer the query. Explain whether you got the answer\n",
"from the parsed markdown or raw text or image, and if there's discrepancies, and your reasoning for the final answer.\n",
"\n",
"Query: {query_str}\n",
"Answer: \"\"\"\n",
"\n",
"QA_PROMPT = PromptTemplate(QA_PROMPT_TMPL)\n",
"\n",
"gpt_4o_mm = OpenAIMultiModal(model=\"gpt-4o\", max_new_tokens=4096)\n",
"\n",
"\n",
"class MultimodalQueryEngine(CustomQueryEngine):\n",
" qa_prompt: PromptTemplate\n",
" retriever: BaseRetriever\n",
" multi_modal_llm: OpenAIMultiModal\n",
"\n",
" def __init__(\n",
" self,\n",
" qa_prompt: PromptTemplate,\n",
" retriever: BaseRetriever,\n",
" multi_modal_llm: OpenAIMultiModal,\n",
" ):\n",
" super().__init__(\n",
" qa_prompt=qa_prompt, retriever=retriever, multi_modal_llm=multi_modal_llm\n",
" )\n",
"\n",
" def custom_query(self, query_str: str):\n",
" # retrieve most relevant nodes\n",
" nodes = self.retriever.retrieve(query_str)\n",
"\n",
" # create image nodes from the image associated with those nodes\n",
" image_nodes = [\n",
" NodeWithScore(node=ImageNode(image_path=n.node.metadata[\"image_path\"]))\n",
" for n in nodes\n",
" ]\n",
"\n",
" # create context string from parsed markdown text\n",
" ctx_str = \"\\n\\n\".join(\n",
" [r.node.get_content(metadata_mode=MetadataMode.LLM) for r in nodes]\n",
" )\n",
" # prompt for the LLM\n",
" fmt_prompt = self.qa_prompt.format(context_str=ctx_str, query_str=query_str)\n",
"\n",
" # use the multimodal LLM to interpret images and generate a response to the prompt\n",
" llm_repsonse = self.multi_modal_llm.complete(\n",
" prompt=fmt_prompt,\n",
" image_documents=[image_node.node for image_node in image_nodes],\n",
" )\n",
" return Response(\n",
" response=str(llm_repsonse),\n",
" source_nodes=nodes,\n",
" metadata={\"text_nodes\": text_nodes, \"image_nodes\": image_nodes},\n",
" )"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Create a query engine instance."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"query_engine = MultimodalQueryEngine(\n",
" qa_prompt=QA_PROMPT,\n",
" retriever=index.as_retriever(similarity_top_k=9),\n",
" multi_modal_llm=gpt_4o_mm,\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"\n",
"## Example Queries"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"The query asks about the parts included in the Uppspel, but the provided images and parsed text do not contain any information about the Uppspel. Instead, they contain information about other IKEA products such as SMÅGÖRA, FREDDE, and TUFFING.\n",
"\n",
"Therefore, based on the provided images and parsed text, I cannot determine the parts included in the Uppspel. The answer cannot be derived from the given information."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"from IPython.display import display, Markdown\n",
"\n",
"response = query_engine.query(\"What parts are included in the Uppspel?\")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"The Tuffing is a bunk bed frame with a minimalist design, featuring a metal frame and safety rails on the top bunk. The image provided shows the Tuffing bunk bed with a ladder for access to the top bunk and a simple, sturdy construction.\n",
"\n",
"I got the answer from the image provided. The image clearly shows the design and structure of the Tuffing bunk bed. There were no discrepancies between the parsed markdown or raw text and the image. The image was the primary source for understanding what the Tuffing looks like."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\"What does the Tuffing look like?\")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"The query asks for step 4 of assembling the Nordli. Based on the provided information, step 4 is described in the parsed text as follows:\n",
"\n",
"**Step 4:**\n",
"- Insert the provided tool into the hole as shown.\n",
"- Ensure the structure is properly aligned and secure.\n",
"- Push down firmly to lock the structure in place.\n",
"\n",
"This information was derived from the parsed text, as the image provided does not contain step-by-step instructions for the Nordli assembly. There are no discrepancies between the parsed markdown and raw text for this step."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\"What is step 4 of assembling the Nordli?\")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/markdown": [
"If you're confused with reading the manual, you should contact IKEA customer service for assistance. This information is derived from the image on page 2, which shows a person with a question mark next to an IKEA box and another person making a phone call to IKEA. This visual cue indicates that contacting IKEA customer service is the recommended action if you need help."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = query_engine.query(\n",
" \"What should I do if I'm confused with reading the manual?\"\n",
")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"You can also create an agent around the query engine and chat with the agent."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core.agent import FunctionCallingAgentWorker\n",
"from llama_index.core.tools import QueryEngineTool\n",
"\n",
"query_engine_tool = QueryEngineTool.from_defaults(\n",
" query_engine=query_engine,\n",
" name=\"query_engine_tool\",\n",
" description=\"Useful for retrieving specific context from the data. Do NOT select if question asks for a summary of the data.\",\n",
")\n",
"agent = FunctionCallingAgentWorker.from_tools(\n",
" [query_engine_tool], llm=llm, verbose=True\n",
").as_agent()"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Added user message to memory: Give a step-by-step instruction guide on how to assemble the Smagora\n",
"=== Calling Function ===\n",
"Calling function: query_engine_tool with args: {\"input\": \"step-by-step instruction guide on how to assemble the Smagora\"}\n",
"=== Function Output ===\n",
"The step-by-step instruction guide on how to assemble the Smågåra crib is provided in the images. The images show detailed visual instructions for each step of the assembly process, including the tools required, the parts involved, and the specific actions to be taken.\n",
"\n",
"Here is a summary of the steps based on the images:\n",
"\n",
"1. **Tools Required**:\n",
" - Flathead screwdriver\n",
" - Phillips screwdriver\n",
" - Hammer\n",
"\n",
"2. **Preparation**:\n",
" - Do not assemble alone; assemble with a partner.\n",
" - Do not assemble on a hard surface; use a soft surface to avoid damage.\n",
" - If you have questions or need assistance, contact IKEA customer service.\n",
"\n",
"3. **Step 1**:\n",
" - Insert 12 screws into the designated holes on the frame.\n",
"\n",
"4. **Step 2**:\n",
" - Align the side panels with the headboard and footboard.\n",
" - Use 4 connectors and secure them with bolts and washers.\n",
" - Tighten using the provided tool.\n",
" - Carefully flip the structure as shown.\n",
"\n",
"5. **Step 3**:\n",
" - Use the provided Allen key to tighten the screws into the designated holes.\n",
" - Ensure the screws are properly aligned and tightened.\n",
" - Repeat this process for all four screws.\n",
" - Make sure the screws are flush with the surface.\n",
"\n",
"6. **Step 4**:\n",
" - Insert the provided tool into the hole as shown.\n",
" - Ensure the structure is properly aligned and secure.\n",
" - Push down firmly to lock the structure in place.\n",
"\n",
"7. **Step 5**:\n",
" - Insert 4 dowels into the designated holes on the board.\n",
"\n",
"8. **Step 6**:\n",
" - Align the board with the dowels and insert it into the corresponding slots on the frame.\n",
"\n",
"9. **Step 7**:\n",
" - Insert the top panel into the side panels.\n",
" - Use 4 screws to secure the top panel.\n",
" - Ensure the screws are properly aligned and tightened using the provided tool.\n",
"\n",
"10. **Step 8**:\n",
" - Carefully flip the assembled structure upright.\n",
" - Use 2 screws to secure the bottom panel.\n",
" - Tighten the screws with the provided tool.\n",
"\n",
"These steps are derived from the images provided, which offer a clear and detailed visual guide for assembling the Smågåra crib.\n",
"=== LLM Response ===\n",
"Here is a step-by-step instruction guide on how to assemble the Smågåra crib:\n",
"\n",
"### Tools Required:\n",
"- Flathead screwdriver\n",
"- Phillips screwdriver\n",
"- Hammer\n",
"- Allen key (provided in the package)\n",
"\n",
"### Preparation:\n",
"- **Safety First**: Assemble with a partner to ensure safety and ease.\n",
"- **Surface**: Assemble on a soft surface to avoid damaging the parts.\n",
"- **Assistance**: If you have questions or need help, contact IKEA customer service.\n",
"\n",
"### Step-by-Step Assembly:\n",
"\n",
"#### Step 1: Insert Screws into the Frame\n",
"1. Insert 12 screws into the designated holes on the frame.\n",
"2. Ensure the screws are properly aligned.\n",
"\n",
"#### Step 2: Align and Secure Side Panels\n",
"1. Align the side panels with the headboard and footboard.\n",
"2. Use 4 connectors and secure them with bolts and washers.\n",
"3. Tighten the bolts using the provided tool.\n",
"4. Carefully flip the structure as shown in the instructions.\n",
"\n",
"#### Step 3: Tighten Screws\n",
"1. Use the provided Allen key to tighten the screws into the designated holes.\n",
"2. Ensure the screws are properly aligned and tightened.\n",
"3. Repeat this process for all four screws.\n",
"4. Make sure the screws are flush with the surface.\n",
"\n",
"#### Step 4: Lock the Structure\n",
"1. Insert the provided tool into the hole as shown.\n",
"2. Ensure the structure is properly aligned and secure.\n",
"3. Push down firmly to lock the structure in place.\n",
"\n",
"#### Step 5: Insert Dowels\n",
"1. Insert 4 dowels into the designated holes on the board.\n",
"\n",
"#### Step 6: Align and Insert the Board\n",
"1. Align the board with the dowels.\n",
"2. Insert the board into the corresponding slots on the frame.\n",
"\n",
"#### Step 7: Secure the Top Panel\n",
"1. Insert the top panel into the side panels.\n",
"2. Use 4 screws to secure the top panel.\n",
"3. Ensure the screws are properly aligned and tightened using the provided tool.\n",
"\n",
"#### Step 8: Secure the Bottom Panel\n",
"1. Carefully flip the assembled structure upright.\n",
"2. Use 2 screws to secure the bottom panel.\n",
"3. Tighten the screws with the provided tool.\n",
"\n",
"By following these steps, you should be able to assemble the Smågåra crib successfully. If you encounter any issues, refer to the visual instructions provided in the package or contact IKEA customer service for assistance.\n"
]
},
{
"data": {
"text/markdown": [
"Here is a step-by-step instruction guide on how to assemble the Smågåra crib:\n",
"\n",
"### Tools Required:\n",
"- Flathead screwdriver\n",
"- Phillips screwdriver\n",
"- Hammer\n",
"- Allen key (provided in the package)\n",
"\n",
"### Preparation:\n",
"- **Safety First**: Assemble with a partner to ensure safety and ease.\n",
"- **Surface**: Assemble on a soft surface to avoid damaging the parts.\n",
"- **Assistance**: If you have questions or need help, contact IKEA customer service.\n",
"\n",
"### Step-by-Step Assembly:\n",
"\n",
"#### Step 1: Insert Screws into the Frame\n",
"1. Insert 12 screws into the designated holes on the frame.\n",
"2. Ensure the screws are properly aligned.\n",
"\n",
"#### Step 2: Align and Secure Side Panels\n",
"1. Align the side panels with the headboard and footboard.\n",
"2. Use 4 connectors and secure them with bolts and washers.\n",
"3. Tighten the bolts using the provided tool.\n",
"4. Carefully flip the structure as shown in the instructions.\n",
"\n",
"#### Step 3: Tighten Screws\n",
"1. Use the provided Allen key to tighten the screws into the designated holes.\n",
"2. Ensure the screws are properly aligned and tightened.\n",
"3. Repeat this process for all four screws.\n",
"4. Make sure the screws are flush with the surface.\n",
"\n",
"#### Step 4: Lock the Structure\n",
"1. Insert the provided tool into the hole as shown.\n",
"2. Ensure the structure is properly aligned and secure.\n",
"3. Push down firmly to lock the structure in place.\n",
"\n",
"#### Step 5: Insert Dowels\n",
"1. Insert 4 dowels into the designated holes on the board.\n",
"\n",
"#### Step 6: Align and Insert the Board\n",
"1. Align the board with the dowels.\n",
"2. Insert the board into the corresponding slots on the frame.\n",
"\n",
"#### Step 7: Secure the Top Panel\n",
"1. Insert the top panel into the side panels.\n",
"2. Use 4 screws to secure the top panel.\n",
"3. Ensure the screws are properly aligned and tightened using the provided tool.\n",
"\n",
"#### Step 8: Secure the Bottom Panel\n",
"1. Carefully flip the assembled structure upright.\n",
"2. Use 2 screws to secure the bottom panel.\n",
"3. Tighten the screws with the provided tool.\n",
"\n",
"By following these steps, you should be able to assemble the Smågåra crib successfully. If you encounter any issues, refer to the visual instructions provided in the package or contact IKEA customer service for assistance."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = agent.chat(\n",
" \"Give a step-by-step instruction guide on how to assemble the Smagora\"\n",
")\n",
"display(Markdown(str(response)))"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Added user message to memory: How do I assemble the Fredde?\n",
"=== Calling Function ===\n",
"Calling function: query_engine_tool with args: {\"input\": \"step-by-step instruction guide on how to assemble the Fredde\"}\n",
"=== Function Output ===\n",
"The query asks for a step-by-step instruction guide on how to assemble the Fredde. However, based on the provided images and parsed text, there is no specific mention or visual representation of the Fredde assembly instructions. The images and text provided are related to other IKEA products such as Tuffing and Smågöra, but not Fredde.\n",
"\n",
"Therefore, I cannot provide the step-by-step instructions for assembling the Fredde from the given information. If you have the specific instructions for Fredde, please provide them, and I can assist you further.\n",
"=== LLM Response ===\n",
"It appears that the specific step-by-step instructions for assembling the Fredde desk are not available in the provided data. However, I can offer a general guide based on typical assembly procedures for IKEA furniture. For the most accurate and detailed instructions, please refer to the assembly manual that comes with the product.\n",
"\n",
"### General Assembly Guide for Fredde Desk:\n",
"\n",
"#### Tools Required:\n",
"- Phillips screwdriver\n",
"- Flathead screwdriver\n",
"- Allen key (usually provided in the package)\n",
"- Hammer (if needed for dowels)\n",
"\n",
"### Step-by-Step Assembly:\n",
"\n",
"#### Step 1: Unpack and Organize\n",
"1. **Unpack** all the parts and hardware.\n",
"2. **Organize** the parts by type and size to make the assembly process easier.\n",
"\n",
"#### Step 2: Assemble the Main Frame\n",
"1. **Connect the Side Panels**: Attach the side panels to the back panel using screws and dowels as indicated in the manual.\n",
"2. **Secure the Bottom Panel**: Attach the bottom panel to the side panels.\n",
"\n",
"#### Step 3: Attach the Shelves\n",
"1. **Install the Lower Shelves**: Insert the lower shelves into the designated slots and secure them with screws.\n",
"2. **Install the Upper Shelves**: Repeat the process for the upper shelves.\n",
"\n",
"#### Step 4: Attach the Desktop\n",
"1. **Align the Desktop**: Place the desktop on top of the frame, ensuring it is properly aligned.\n",
"2. **Secure the Desktop**: Use screws to secure the desktop to the frame.\n",
"\n",
"#### Step 5: Install Additional Features\n",
"1. **Attach Monitor Shelf**: If the Fredde desk includes a monitor shelf, attach it to the back panel using screws.\n",
"2. **Install Side Extensions**: Attach any side extensions or additional shelves as per the instructions.\n",
"\n",
"#### Step 6: Final Adjustments\n",
"1. **Check Stability**: Ensure all screws are tightened and the desk is stable.\n",
"2. **Adjust Height**: If the desk has adjustable height features, set it to the desired height.\n",
"\n",
"#### Step 7: Clean Up\n",
"1. **Remove Packaging**: Dispose of any packaging materials.\n",
"2. **Organize Tools**: Put away your tools and clean the workspace.\n",
"\n",
"For the most accurate and detailed instructions, please refer to the assembly manual that comes with the Fredde desk. If you encounter any issues, IKEA customer service can provide additional support.\n"
]
},
{
"data": {
"text/markdown": [
"It appears that the specific step-by-step instructions for assembling the Fredde desk are not available in the provided data. However, I can offer a general guide based on typical assembly procedures for IKEA furniture. For the most accurate and detailed instructions, please refer to the assembly manual that comes with the product.\n",
"\n",
"### General Assembly Guide for Fredde Desk:\n",
"\n",
"#### Tools Required:\n",
"- Phillips screwdriver\n",
"- Flathead screwdriver\n",
"- Allen key (usually provided in the package)\n",
"- Hammer (if needed for dowels)\n",
"\n",
"### Step-by-Step Assembly:\n",
"\n",
"#### Step 1: Unpack and Organize\n",
"1. **Unpack** all the parts and hardware.\n",
"2. **Organize** the parts by type and size to make the assembly process easier.\n",
"\n",
"#### Step 2: Assemble the Main Frame\n",
"1. **Connect the Side Panels**: Attach the side panels to the back panel using screws and dowels as indicated in the manual.\n",
"2. **Secure the Bottom Panel**: Attach the bottom panel to the side panels.\n",
"\n",
"#### Step 3: Attach the Shelves\n",
"1. **Install the Lower Shelves**: Insert the lower shelves into the designated slots and secure them with screws.\n",
"2. **Install the Upper Shelves**: Repeat the process for the upper shelves.\n",
"\n",
"#### Step 4: Attach the Desktop\n",
"1. **Align the Desktop**: Place the desktop on top of the frame, ensuring it is properly aligned.\n",
"2. **Secure the Desktop**: Use screws to secure the desktop to the frame.\n",
"\n",
"#### Step 5: Install Additional Features\n",
"1. **Attach Monitor Shelf**: If the Fredde desk includes a monitor shelf, attach it to the back panel using screws.\n",
"2. **Install Side Extensions**: Attach any side extensions or additional shelves as per the instructions.\n",
"\n",
"#### Step 6: Final Adjustments\n",
"1. **Check Stability**: Ensure all screws are tightened and the desk is stable.\n",
"2. **Adjust Height**: If the desk has adjustable height features, set it to the desired height.\n",
"\n",
"#### Step 7: Clean Up\n",
"1. **Remove Packaging**: Dispose of any packaging materials.\n",
"2. **Organize Tools**: Put away your tools and clean the workspace.\n",
"\n",
"For the most accurate and detailed instructions, please refer to the assembly manual that comes with the Fredde desk. If you encounter any issues, IKEA customer service can provide additional support."
],
"text/plain": [
"<IPython.core.display.Markdown object>"
]
},
"metadata": {},
"output_type": "display_data"
}
],
"source": [
"response = agent.chat(\"How do I assemble the Fredde?\")\n",
"display(Markdown(str(response)))"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama-parse-5ZmnAQ0r-py3.11",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
@@ -1,335 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# LlamaParse - Parsing Financial Powerpoints 📊\n",
"\n",
"In this cookbook we show you how to use LlamaParse to parse a financial powerpoint."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Installation\n",
"\n",
"Parsing instruction are part of the LlamaParse API. They can be access by directly specifying the parsing_instruction parameter in the API or by using LlamaParse python module (which we will use for this tutorial).\n",
"\n",
"To install llama-parse, just get it from `pip`:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-index\n",
"%pip install llama-cloud-services\n",
"%pip install torch transformers python-pptx Pillow"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## API Key\n",
"\n",
"The use of LlamaParse requires an API key which you can get here: https://cloud.llamaindex.ai/parse"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import os\n",
"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"llx-...\"\n",
"os.environ[\"OPENAI_API_KEY\"] = \"sk-...\""
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"**NOTE**: Since LlamaParse is natively async, running the sync code in a notebook requires the use of nest_asyncio.\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Importing the package\n",
"\n",
"To import llama_parse simply do:"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud_services import LlamaParse"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Using LlamaParse to Parse Presentations\n",
"\n",
"Like Powerpoints, presentations are often hard to extract for RAG. With LlamaParse we can now parse them and unclock their content of presentations for RAG.\n",
"\n",
"Let's download a financial report from the World Meteorological Association."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"! mkdir data; wget \"https://meetings.wmo.int/Cg-19/PublishingImages/SitePages/FINAC-43/7%20-%20EC-77-Doc%205%20Financial%20Statements%20for%202022%20(FINAC).pptx\" -O data/presentation.pptx"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Parsing the presentation\n",
"\n",
"Now let's parse it into Markdown with LlamaParse and the default LlamaIndex parser.\n",
"\n",
"\n"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"#### Llama Index default"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import SimpleDirectoryReader\n",
"\n",
"vanilla_documents = SimpleDirectoryReader(\"./data/\").load_data()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"#### Llama Parse"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 56724c0d-e45a-4e30-ae8c-e416173c608a\n"
]
}
],
"source": [
"llama_parse_documents = LlamaParse(result_type=\"markdown\").load_data(\n",
" \"./data/presentation.pptx\"\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Let's take a look at the parsed output from an example slide (see image below).\n",
"\n",
"As we can see the table is faithfully extracted!"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"ation and mitigation\n",
"---\n",
"|Item|31 Dec 2022|31 Dec 2021|Change|\n",
"|---|---|---|---|\n",
"|Payables and accruals|4,685|4,066|619|\n",
"|Employee benefits|127,215|84,676|42,539|\n",
"|Contributions received in advance|6,975|10,192|(3,217)|\n",
"|Unearned revenue from exchange transactions|20|651|(631)|\n",
"|Deferred Revenue|71,301|55,737|15,564|\n",
"|Borrowings|28,229|29,002|(773)|\n",
"|Funds held in trust|30,373|29,014|1,359|\n",
"|Provisions|1,706|1,910|(204)|\n",
"|Total Liabilities|270,504|215,248|55,256|\n",
"---\n",
"## Liabilities\n",
"\n",
"Employee Ben\n"
]
}
],
"source": [
"print(llama_parse_documents[0].get_content()[-2800:-2300])"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Compared against the original slide image.\n",
"![Demo](demo_ppt_financial_1.png)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## Comparing the two for RAG\n",
"\n",
"The main difference between LlamaParse and the previous directory reader approach, it that LlamaParse will extract the document in a structured format, allowing better RAG."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Query Engine on SimpleDirectoryReader results"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_index.core import VectorStoreIndex, SimpleDirectoryReader\n",
"\n",
"vanilla_index = VectorStoreIndex.from_documents(vanilla_documents)\n",
"vanilla_query_engine = vanilla_index.as_query_engine()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Query Engine on LlamaParse Results\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"llama_parse_index = VectorStoreIndex.from_documents(llama_parse_documents)\n",
"llama_parse_query_engine = llama_parse_index.as_query_engine()"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Liability provision\n",
"What was the liability provision as of Dec 31 2021?\n",
"\n",
"<!-- <img src=\"https://drive.usercontent.google.com/download?id=184jVq0QyspDnmCyRfV0ebmJJxmAOJHba&authuser=0\" /> -->"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The liability provision as of December 31, 2021, included Employee Benefit Liabilities, Contributions received in advance (assessed contributions), and Deferred revenue.\n"
]
}
],
"source": [
"vanilla_response = vanilla_query_engine.query(\n",
" \"What was the liability provision as of Dec 31 2021?\"\n",
")\n",
"print(vanilla_response)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"The liability provision as of December 31, 2021, was 1,910 CHF.\n"
]
}
],
"source": [
"llama_parse_response = llama_parse_query_engine.query(\n",
" \"What was the liability provision as of Dec 31 2021?\"\n",
")\n",
"print(llama_parse_response)"
]
}
],
"metadata": {
"colab": {
"provenance": []
},
"kernelspec": {
"display_name": "llama_parse",
"language": "python",
"name": "llama_parse"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 4
}
Binary file not shown.

Before

Width:  |  Height:  |  Size: 350 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 47 KiB

@@ -1,602 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"<a href=\"https://colab.research.google.com/github/run-llama/llama_cloud_services/blob/main/examples/parse/parsing_instructions.ipynb\" target=\"_parent\"><img src=\"https://colab.research.google.com/assets/colab-badge.svg\" alt=\"Open In Colab\"/></a>\n",
"\n",
"# Parsing documents with Instructions\n",
"\n",
"Parsing instructions allow you to guide our parsing model in the same way you would instruct an LLM.\n",
"\n",
"These instructions can be useful for improving the parser's performance on complex document layouts, extracting data in a specific format, or transforming the document in other ways.\n",
"\n",
"### Why This Matters:\n",
"Traditional document parsing can be rigid and error-prone, often missing crucial context and nuances in complex layouts. Our instruction-based parsing allows you to:\n",
"\n",
"1. Extract specific information with pinpoint accuracy\n",
"2. Handle complex document layouts with ease\n",
"3. Transform unstructured data into structured formats effortlessly\n",
"4. Save hours of manual data entry and verification\n",
"5. Reduce errors in document processing workflows\n",
"\n",
"In this demonstration, we showcase how parsing instructions can be used to extract specific information from unstructured documents. Below are the documents we use for testing:\n",
"\n",
"1. McDonald's Receipt - Extracting the price of each order and the final amount to be paid.\n",
"\n",
"2. Expense Report Document - Extracting employee name, employee ID, position, department, date ranges, individual expense items with dates, categories, and amounts.\n",
"\n",
"3. Purchase Order Document - Identifying the PO number, vendor details, shipping terms, and an itemized list of products with quantities and unit prices.\n",
"\n",
"Let's jump into these real-world examples and see how parsing instructions can help us extract specific information."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Installation"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!pip install llama-cloud-services"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Setup API Key"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"import nest_asyncio\n",
"\n",
"nest_asyncio.apply()\n",
"\n",
"import os\n",
"\n",
"os.environ[\"LLAMA_CLOUD_API_KEY\"] = \"llx-...\""
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### McDonald's Receipt\n",
"\n",
"Here we extract the price of each order and the final amount to be paid."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"<img src=\"mcdonalds_receipt.png\" alt=\"Alt Text\" width=\"500\">"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 66643b81-e2f4-408b-890b-8e116472210b\n"
]
}
],
"source": [
"from llama_cloud_services import LlamaParse\n",
"\n",
"vanilaParsing = LlamaParse(result_type=\"markdown\").load_data(\"./mcdonalds_receipt.png\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# Rate us HIGHLY SATISFIED\n",
"\n",
"Purchase any sandwich and receive a FREE ITEM\n",
"\n",
"Go to WWW.mcdvoice.com within 7 days of purchase of equal or lesser value and tell us about your visit.\n",
"\n",
"Validation Code: 31278-01121-21018-20481-00081-0\n",
"\n",
"Valid at participating US McDonald's\n",
"\n",
"Expires 30 days after receipt date\n",
"\n",
"# McDonald's Restaurant #312782378\n",
"\n",
"PINE RD NW\n",
"\n",
"RICE MN 56367-9740\n",
"\n",
"TEL# 320 393 4600\n",
"\n",
"KS# 12/08/2022 08:48 PM\n",
"\n",
"# Order\n",
"\n",
"|Happy Meal 6 Pc|$4.89|\n",
"|---|---|\n",
"|Creamy Ranch Cup| |\n",
"|Extra Kids Fry| |\n",
"|Wreck It Ralph 2 Snack| |\n",
"|Oreo McFlurry|$2.69|\n",
"\n",
"# Summary\n",
"\n",
"|Subtotal|$7.58|\n",
"|---|---|\n",
"|Tax|$0.52|\n",
"|Take-Out Total|$8.10|\n",
"|Cash Tendered|$10.00|\n",
"|Change|$1.90|\n",
"\n",
"### Not ACCEPTING APPLICATIONS *++ McDonald's Restaurant Rice\n",
"\n",
"Text to #36453 apply 31278\n"
]
}
],
"source": [
"print(vanilaParsing[0].text)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 1a04fdbb-5415-4a36-a1bd-26bfb5d618fa\n"
]
}
],
"source": [
"parsingInstruction = \"\"\"The provided document is a McDonald's receipt.\n",
" Provide the price of each order and final amount to be paid.\"\"\"\n",
"withInstructionParsing = LlamaParse(\n",
" result_type=\"markdown\", parsing_instruction=parsingInstruction\n",
").load_data(\"./mcdonalds_receipt.png\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Here are the prices for each order from the McDonald's receipt:\n",
"\n",
"1. Happy Meal 6 Pc: $4.89\n",
"2. Snack Oreo McFlurry: $2.69\n",
"\n",
"**Subtotal:** $7.58\n",
"**Tax:** $0.52\n",
"**Total Amount to be Paid:** $8.10\n",
"\n",
"The cash tendered was $10.00, and the change given was $1.90.\n"
]
}
],
"source": [
"print(withInstructionParsing[0].text)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Expense Report Document\n",
"\n",
"Here we extract employee name, employee ID, position, department, date ranges, individual expense items with dates, categories, and amounts."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"<img src=\"expense_report_document.png\" alt=\"Alt Text\" width=\"500\">"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id b6bcc6e1-7d30-4522-9abd-ace196781a70\n"
]
}
],
"source": [
"vanilaParsing = LlamaParse(result_type=\"markdown\").load_data(\n",
" \"./expense_report_document.pdf\"\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# QUANTUM DYNAMICS CORPORATION\n",
"\n",
"# EMPLOYEE EXPENSE REPORT\n",
"\n",
"# FISCAL YEAR 2024\n",
"\n",
"# EMPLOYEE INFORMATION:\n",
"\n",
"Name: Dr. Alexandra Chen-Martinez, PhD\n",
"\n",
"Employee ID: QD-2022-1457\n",
"\n",
"Department: Advanced Research & Development\n",
"\n",
"Cost Center: CC-ARD-NA-003\n",
"\n",
"Project Codes: QD-QUANTUM-2024-01, QD-AI-2024-03\n",
"\n",
"Position: Principal Research Scientist\n",
"\n",
"Reporting Manager: Dr. James Thompson\n",
"\n",
"# TRIP/EXPENSE PERIOD:\n",
"\n",
"Start Date: November 15, 2024\n",
"\n",
"End Date: December 10, 2024\n",
"\n",
"Purpose: International Conference Attendance & Client Meetings\n",
"\n",
"Locations: Tokyo, Japan → Singapore → Sydney, Australia\n",
"\n",
"# CURRENCY CONVERSION RATES APPLIED:\n",
"\n",
"JPY (¥) → USD: 0.0068 (as of 11/15/2024)\n",
"\n",
"SGD (S$) → USD: 0.74 (as of 11/28/2024)\n",
"\n",
"AUD (A$) → USD: 0.65 (as of 12/03/2024)\n",
"\n",
"# ITEMIZED EXPENSES:\n",
"\n",
"|Date|Category|Description|Original|Currency|USD|\n",
"|---|---|---|---|---|---|\n",
"|11/15/2024|Transportation|JFK → NRT Business Class|4,250.00|USD|4,250.00|\n",
"|Booking Ref: QF78956 - Corporate Rate Applied|Booking Ref: QF78956 - Corporate Rate Applied|Booking Ref: QF78956 - Corporate Rate Applied|Booking Ref: QF78956 - Corporate Rate Applied|Booking Ref: QF78956 - Corporate Rate Applied|Booking Ref: QF78956 - Corporate Rate Applied|\n",
"|Project Code: QD-QUANTUM-2024-01|Project Code: QD-QUANTUM-2024-01|Project Code: QD-QUANTUM-2024-01|Project Code: QD-QUANTUM-2024-01|Project Code: QD-QUANTUM-2024-01|Project Code: QD-QUANTUM-2024-01|\n",
"|11/16/2024|Accommodation|Hilton Tokyo - 5 nights|225,000|JPY|1,530.00|\n",
"|Confirmation: HTK-2024-78956|Confirmation: HTK-2024-78956|Confirmation: HTK-2024-78956|Confirmation: HTK-2024-78956|Confirmation: HTK-2024-78956|Confirmation: HTK-2024-78956|\n"
]
}
],
"source": [
"print(vanilaParsing[0].text)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id 7b0d05bb-947b-4475-8d0f-f10386f7446e\n"
]
}
],
"source": [
"parsingInstruction = \"\"\"You are provided with an expense report. \n",
"Extract employee name, employee id, position, department, date ranges, individual expense items with dates, categories, and amounts.\"\"\"\n",
"\n",
"withInstructionParsing = LlamaParse(\n",
" result_type=\"markdown\", parsing_instruction=parsingInstruction\n",
").load_data(\"./expense_report_document.pdf\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"**Employee Information:**\n",
"- **Name:** Dr. Alexandra Chen-Martinez, PhD\n",
"- **Employee ID:** QD-2022-1457\n",
"- **Position:** Principal Research Scientist\n",
"- **Department:** Advanced Research & Development\n",
"\n",
"**Trip/Expense Period:**\n",
"- **Start Date:** November 15, 2024\n",
"- **End Date:** December 10, 2024\n",
"\n",
"**Expense Items:**\n",
"1. **Date:** 11/15/2024\n",
"- **Category:** Transportation\n",
"- **Description:** JFK → NRT Business Class\n",
"- **Original Amount:** $4,250.00\n",
"- **Currency:** USD\n",
"- **USD Amount:** $4,250.00\n",
"- **Booking Reference:** QF78956 - Corporate Rate Applied\n",
"- **Project Code:** QD-QUANTUM-2024-01\n",
"\n",
"2. **Date:** 11/16/2024\n",
"- **Category:** Accommodation\n",
"- **Description:** Hilton Tokyo - 5 nights\n",
"- **Original Amount:** ¥225,000\n",
"- **Currency:** JPY\n",
"- **USD Amount:** $1,530.00\n",
"- **Confirmation:** HTK-2024-78956\n",
"\n",
"**Locations:**\n",
"- Tokyo, Japan\n",
"- Singapore\n",
"- Sydney, Australia\n",
"\n",
"**Currency Conversion Rates Applied:**\n",
"- JPY (¥) → USD: 0.0068 (as of 11/15/2024)\n",
"- SGD (S$) → USD: 0.74 (as of 11/28/2024)\n",
"- AUD (A$) → USD: 0.65 (as of 12/03/2024)\n"
]
}
],
"source": [
"print(withInstructionParsing[0].text)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"### Purchase Order Document \n",
"\n",
"Here we identify the PO number, vendor details, shipping terms, and an itemized list of products with quantities and unit prices."
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"<img src=\"purchase_order_document.png\" alt=\"Alt Text\" width=\"500\">"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id b8cb11c3-7dce-4e6a-94bb-1a4e50e45e55\n"
]
}
],
"source": [
"vanilaParsing = LlamaParse(result_type=\"markdown\").load_data(\n",
" \"./purchase_order_document.pdf\"\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# GLOBAL TECH SOLUTIONS, INC.\n",
"\n",
"# PURCHASE ORDER\n",
"\n",
"Document Reference: PO-2024-GT-9876/REV.2\n",
"\n",
"[Original: PO-2024-GT-9876]\n",
"\n",
"Amendment Date: 12/10/2024\n",
"\n",
"# VENDOR INFORMATION:\n",
"\n",
"Quantum Electronics Manufacturing\n",
"\n",
"DUNS: 78-456-7890\n",
"\n",
"Tax ID: EU8976543210\n",
"\n",
"Hoofdorp, Netherlands\n",
"\n",
"Vendor #: QEM-EU-2024-001\n",
"\n",
"# SHIP TO:\n",
"\n",
"Global Tech Solutions, Inc.\n",
"\n",
"Building 7A, Innovation Park\n",
"\n",
"2100 Technology Drive\n",
"\n",
"Austin, TX 78701\n",
"\n",
"USA\n",
"\n",
"Attn: Sarah Martinez, Receiving Manager\n",
"\n",
"Tel: +1 (512) 555-0123\n",
"\n",
"# PAYMENT TERMS:\n",
"\n",
"Net 45\n",
"\n",
"2% discount if paid within 15 days\n",
"\n",
"# SHIPPING TERMS:\n",
"\n",
"DDP (Delivered Duty Paid) - Incoterms 2020\n",
"\n",
"Insurance Required: Yes\n",
"\n",
"Preferred Carrier: DHL/FedEx\n",
"\n",
"Required Delivery Date: 01/15/2025\n",
"\n",
"# SPECIAL INSTRUCTIONS:\n",
"\n",
"1. All shipments must include Certificate of Conformance\n",
"2. ESD-sensitive items must be properly packaged\n",
"3. Temperature logging required for items marked with *\n",
"4. Partial shipments accepted with prior approval\n",
"5. Quote PO number on all correspondence\n",
"\n",
"# ITEM DETAILS:\n",
"\n",
"|Line|Part Number|Description|Qty|UOM|Unit Price|Total|\n",
"|---|---|---|---|---|---|---|\n",
"|1|QE-MCU-5590|Microcontroller Unit|500|EA|$12.50|$6,250.00|\n"
]
}
],
"source": [
"print(vanilaParsing[0].text)"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Started parsing the file under job_id d2731305-984d-4633-8a52-0493748cf10b\n"
]
}
],
"source": [
"parsingInstruction = \"\"\"You are provided with a purchase order. \n",
"Identify the PO number, vendor details, shipping terms, and itemized list of products with quantities and unit prices.\"\"\"\n",
"\n",
"withInstructionParsing = LlamaParse(\n",
" result_type=\"markdown\", parsing_instruction=parsingInstruction\n",
").load_data(\"./purchase_order_document.pdf\")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Here are the details extracted from the purchase order:\n",
"\n",
"**PO Number:** PO-2024-GT-9876/REV.2\n",
"\n",
"**Vendor Details:**\n",
"- **Vendor Name:** Quantum Electronics Manufacturing\n",
"- **DUNS:** 78-456-7890\n",
"- **Tax ID:** EU8976543210\n",
"- **Address:** Hoofdorp, Netherlands\n",
"- **Vendor Number:** QEM-EU-2024-001\n",
"- **Contact Person:** Sarah Martinez, Receiving Manager\n",
"- **Phone:** +1 (512) 555-0123\n",
"\n",
"**Shipping Terms:**\n",
"- **Terms:** DDP (Delivered Duty Paid) - Incoterms 2020\n",
"- **Insurance Required:** Yes\n",
"- **Preferred Carrier:** DHL/FedEx\n",
"- **Required Delivery Date:** 01/15/2025\n",
"\n",
"**Itemized List of Products:**\n",
"1. **Part Number:** QE-MCU-5590\n",
"- **Description:** Microcontroller Unit\n",
"- **Quantity:** 500 EA\n",
"- **Unit Price:** $12.50\n",
"- **Total:** $6,250.00\n",
"\n",
"**Payment Terms:**\n",
"- Net 45\n",
"- 2% discount if paid within 15 days\n",
"\n",
"**Special Instructions:**\n",
"1. All shipments must include Certificate of Conformance\n",
"2. ESD-sensitive items must be properly packaged\n",
"3. Temperature logging required for items marked with *\n",
"4. Partial shipments accepted with prior approval\n",
"5. Quote PO number on all correspondence\n"
]
}
],
"source": [
"print(withInstructionParsing[0].text)"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llamacloud",
"language": "python",
"name": "llamacloud"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
Binary file not shown.

Before

Width:  |  Height:  |  Size: 344 KiB

File diff suppressed because it is too large Load Diff
File diff suppressed because one or more lines are too long
Binary file not shown.

Before

Width:  |  Height:  |  Size: 2.3 MiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 96 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 828 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 626 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 100 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 464 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 410 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 444 KiB

Binary file not shown.

Before

Width:  |  Height:  |  Size: 610 KiB

-762
View File
@@ -1,762 +0,0 @@
{
"cells": [
{
"cell_type": "markdown",
"metadata": {},
"source": [
"# Report Generation with LlamaReport\n",
"\n",
"In this notebook, we'll walk through the basic process of generating a report with LlamaReport, and highlight some of the key features of the library.\n",
"\n",
"TLDR:\n",
"1. Download source data to use as knowledge base for the report\n",
"2. Kick off report generation with a template\n",
"3. Get the plan and review/accept/reject suggestions\n",
"4. Get the final report\n",
"5. Review/accept/reject suggestions to edit the final report\n",
"6. Print the final report"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"%pip install llama-cloud-services"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 1. Download Source Data\n",
"\n",
"Here, we download the `Attention is All You Need` paper as a PDF.\n",
"\n",
"LlamaReport currently supports up to 5 files as input, and essentially any file type that can be parsed by LlamaParse.\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"!wget \"https://arxiv.org/pdf/1706.03762.pdf\" -O \"./attention.pdf\""
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 2. Kick off Report Generation\n",
"\n",
"Here, we kick off report generation with a template.\n",
"\n",
"The template can either be a string or a file path, but here we'll use a string.\n",
"\n",
"In our experiments, anything works as a template, but some general guidelines:\n",
"\n",
"- Use markdown formatting + instructions in each section to guide the report generation\n",
"- If using an existing file as a template, provide extra instructions to guide the report generation\n",
"\n",
"**NOTE:** Since we are in a notebook, we will use async functions and `await` throughout. Synchronous methods that work without `await` are available by just removing the `a` from the method name and removing the `await` keyword."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"from llama_cloud_services import LlamaReport\n",
"\n",
"llama_report = LlamaReport(\n",
" api_key=\"llx-...\",\n",
")\n",
"\n",
"report_client = await llama_report.acreate_report(\n",
" name=\"my_cool_report_on_attention\",\n",
" # can pass in file paths or bytes\n",
" input_files=[\"./attention.pdf\"],\n",
" template_text=\"\"\"\\\n",
"# [Some title]\\n\\n\n",
"## TLDR\\n\n",
"A quick summary of the paper.\\n\\n\n",
"## Details\\n\n",
"More details about the paper, possibly more than one section here.\\n\n",
"\"\"\",\n",
" # optional additional instructions for the report generation\n",
" # template_instructions=None,\n",
" # optional file path to an existing template instead of template_text\n",
" # template_file=None,\n",
")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"The returned `ReportClient` object is used to interact with the report generation process for this specific report."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Report(id=0a394b33-1a3e-463c-b5cb-7ff8ab827d0a, name=my_cool_report_on_attention)\n"
]
}
],
"source": [
"print(report_client)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 3. Get the plan\n",
"\n",
"The first phases of report generation involve ingesting the source data and generating a plan.\n",
"\n",
"The plan is a list of instructions for the report generation, and can be reviewed/accepted/rejected by the user.\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"plan = await report_client.await_for_plan(\n",
" timeout=10000,\n",
" poll_interval=10,\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# {title}\n",
"[ReportQuery(field='title', prompt='Generate a clear and concise title for this paper about the Transformer model and attention mechanisms', context='The paper discusses the Transformer architecture for sequence transduction using attention mechanisms, focusing on machine translation applications')]\n",
"==================\n",
"## TLDR\n",
"\n",
"{tldr_content}\n",
"[ReportQuery(field='tldr_content', prompt='Write a brief, clear summary of the key points about the Transformer model', context='Focus on the main innovations: attention mechanisms, efficiency improvements, and state-of-the-art results in machine translation')]\n",
"==================\n",
"## Details\n",
"\n",
"{details_content}\n",
"[ReportQuery(field='details_content', prompt='Provide detailed information about the Transformer model architecture and its applications', context='Include information about:\\n- The attention mechanism implementation\\n- Advantages over recurrent and convolutional models\\n- Performance in machine translation tasks\\n- Training efficiency improvements')]\n",
"==================\n"
]
}
],
"source": [
"for plan_block in plan.blocks:\n",
" print(plan_block.block.template)\n",
" print(plan_block.queries)\n",
" print(\"==================\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"With the plan, we can either use it to kick off generation of the final report, or we can edit the plan and adjust it as needed.\n",
"\n",
"While we could manually edit the objects here and use `await report_client.aupdate_plan(action=\"edit\", updated_plan=plan)`, we can also use `LlamaReport` to agentically edit the plan."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"suggestions = await report_client.asuggest_edits(\n",
" \"Can you split the details section into two sections?\"\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Justification for change: \n",
"I'll help you break down the details section into two distinct parts - one focusing on the architecture and another on the practical applications and performance. This will make the content more organized and easier to follow. The original block at index 2 will be replaced with these two new sections.\n",
"\n",
"Proposed changes:\n",
"\n",
"## Architecture Details\n",
"\n",
"{architecture_content}\n",
"\n",
"[ReportQuery(field='architecture_content', prompt='Describe the technical details of the Transformer model architecture', context='Focus on:\\n- Core components of the Transformer architecture\\n- Self-attention mechanism implementation\\n- Multi-head attention details\\n- Position encoding approach\\n- Feed-forward network structure')]\n",
"==================\n",
"\n",
"## Performance and Applications\n",
"\n",
"{applications_content}\n",
"\n",
"[ReportQuery(field='applications_content', prompt='Explain the practical applications and performance advantages of the Transformer model', context='Cover:\\n- Comparison with RNN and CNN models\\n- Machine translation results and benchmarks\\n- Training efficiency improvements\\n- Real-world applications and use cases\\n- Scalability benefits')]\n",
"==================\n"
]
}
],
"source": [
"for suggestion in suggestions:\n",
" print(\"Justification for change:\", suggestion.justification)\n",
" print(\"Proposed changes:\")\n",
" for plan_block in suggestion.blocks:\n",
" print(plan_block.block.template)\n",
" print(plan_block.queries)\n",
" print(\"==================\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"This looks pretty good! We can also use the client to automatically accept and apply, or reject, these suggestions.\n",
"\n",
"This will (locally) keep track of the history of changes, so that future suggestions can be based on the previous changes."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"for suggestion in suggestions:\n",
" await report_client.aaccept_edit(suggestion)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"What effect did that have on the tracked local history? Let's see!"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"[EditAction(block_idx=2, old_content='## Details\\n\\n{details_content}\\n\\nField: details_content, Prompt: Provide detailed information about the Transformer model architecture and its applications, Context: Include information about:\\n- The attention mechanism implementation\\n- Advantages over recurrent and convolutional models\\n- Performance in machine translation tasks\\n- Training efficiency improvements\\nDepends on: none', new_content='\\n## Architecture Details\\n\\n{architecture_content}\\n\\n\\nField: architecture_content, Prompt: Describe the technical details of the Transformer model architecture, Context: Focus on:\\n- Core components of the Transformer architecture\\n- Self-attention mechanism implementation\\n- Multi-head attention details\\n- Position encoding approach\\n- Feed-forward network structure\\nDepends on: none', action='approved', timestamp=datetime.datetime(2025, 2, 4, 20, 59, 55, 773558)),\n",
" EditAction(block_idx=3, old_content='[No old content]', new_content='\\n## Performance and Applications\\n\\n{applications_content}\\n\\n\\nField: applications_content, Prompt: Explain the practical applications and performance advantages of the Transformer model, Context: Cover:\\n- Comparison with RNN and CNN models\\n- Machine translation results and benchmarks\\n- Training efficiency improvements\\n- Real-world applications and use cases\\n- Scalability benefits\\nDepends on: previous', action='approved', timestamp=datetime.datetime(2025, 2, 4, 20, 59, 55, 773687))]"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"report_client.edit_history"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"data": {
"text/plain": [
"[Message(role=<MessageRole.USER: 'user'>, content='Can you split the details section into two sections?', timestamp=datetime.datetime(2025, 2, 4, 20, 59, 47, 754848)),\n",
" Message(role=<MessageRole.ASSISTANT: 'assistant'>, content=\"\\nI'll help you break down the details section into two distinct parts - one focusing on the architecture and another on the practical applications and performance. This will make the content more organized and easier to follow. The original block at index 2 will be replaced with these two new sections.\\n\", timestamp=datetime.datetime(2025, 2, 4, 20, 59, 55, 482070))]"
]
},
"execution_count": null,
"metadata": {},
"output_type": "execute_result"
}
],
"source": [
"report_client.chat_history"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"These two items are used to provide context for future suggestions! You can always clear this, or provide your own history."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"# report_client.suggest_edits(\"....\", chat_history=[{\"role\": \"user\", \"content\": \"...\"}, ...])"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 4. Get the final report\n",
"\n",
"Now that we have a plan, we can kick off generation of the final report."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"# kicks off report generation\n",
"await report_client.aupdate_plan(action=\"approve\")\n",
"\n",
"# waits for report generation to complete\n",
"report = await report_client.await_completion(\n",
" timeout=10000,\n",
" poll_interval=10,\n",
")"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# Attention Is All You Need: A Pure Attention-Based Architecture for Neural Machine Translation\n",
"\n",
"## TLDR\n",
"\n",
"The Transformer introduced a revolutionary architecture that relies entirely on attention mechanisms, eliminating the need for recurrence or convolution in sequence processing. Its key innovations include multi-head self-attention for parallel processing of input sequences, scaled dot-product attention for efficient computation, and positional encodings for sequence order awareness. The model achieved breakthrough results in machine translation (28.4 BLEU on English-to-German, 41.8 BLEU on English-to-French) while requiring significantly less training time than previous approaches, training in 3.5 days on 8 GPUs. This architecture demonstrated that attention mechanisms alone are sufficient for state-of-the-art sequence modeling, setting a new direction for natural language processing.\n",
"\n",
"\n",
"## Architecture Details\n",
"\n",
"The Transformer architecture represents a groundbreaking approach to sequence processing, built entirely on attention mechanisms without recurrence or convolution. Here are its key technical details:\n",
"\n",
"Core Components:\n",
"- Encoder-decoder architecture with stacked self-attention and point-wise feed-forward layers\n",
"- Each layer contains two main sub-layers: multi-head self-attention mechanism and position-wise feed-forward network\n",
"- Layer normalization and residual connections between sub-layers\n",
"- No recurrent or convolutional elements, enabling parallel processing\n",
"\n",
"Self-Attention Mechanism:\n",
"- Processes relationships between all positions in a sequence simultaneously\n",
"- Computes attention weights using queries, keys, and values derived from input representations\n",
"- Implements scaled dot-product attention to prevent gradient issues with large input dimensions\n",
"- Allows direct modeling of dependencies regardless of positional distance\n",
"- Uses masking in decoder to prevent leftward information flow and maintain auto-regressive property\n",
"\n",
"Multi-Head Attention:\n",
"- Employs multiple attention heads operating in parallel\n",
"- Each head processes information in different representation subspaces\n",
"- Three types of attention applications:\n",
" 1. Encoder self-attention (all positions attend to each other)\n",
" 2. Decoder self-attention (each position attends to previous positions)\n",
" 3. Encoder-decoder attention (decoder queries attend to encoder outputs)\n",
"- Counteracts reduced resolution from attention averaging through parallel processing\n",
"\n",
"Position-wise Feed-Forward Network:\n",
"- Applied identically to each position separately\n",
"- Consists of two linear transformations with ReLU activation\n",
"- Structure: FFN(x) = max(0, xW1 + b1)W2 + b2\n",
"- Input and output dimensionality: dmodel = 512\n",
"- Inner-layer dimensionality: dff = 2048\n",
"- Parameters vary between layers but remain constant across positions\n",
"\n",
"Position Encoding:\n",
"- Adds positional information to input embeddings\n",
"- Enables the model to consider sequential order without recurrence\n",
"- Implements sinusoidal position encodings to allow model to attend to relative positions\n",
"- Maintains constant number of operations between any two positions, unlike convolutional approaches\n",
"- Allows effective modeling of both local and long-range dependencies\n",
"\n",
"\n",
"\n",
"## Performance and Applications\n",
"\n",
"The Transformer model demonstrates significant performance advantages and practical applications across multiple domains:\n",
"\n",
"Performance Advantages over RNN/CNN Models:\n",
"- Eliminates sequential computation constraints present in RNNs, enabling superior parallelization\n",
"- Reduces operations needed for relating distant positions to a constant number, compared to linear/logarithmic scaling in CNNs\n",
"- Processes all input and output positions simultaneously through self-attention mechanisms\n",
"- Achieves state-of-the-art results while requiring significantly less computational resources\n",
"\n",
"Machine Translation Benchmarks:\n",
"- WMT 2014 English-to-German: 28.4 BLEU score, exceeding previous best results by over 2 BLEU points\n",
"- WMT 2014 English-to-French: 41.8 BLEU score (single-model state-of-the-art)\n",
"- Surpasses performance of existing model ensembles in translation tasks\n",
"\n",
"Training Efficiency:\n",
"- Requires only 3.5 days of training on eight GPUs for state-of-the-art performance\n",
"- Achieves superior results at \"a small fraction of the training costs\" compared to previous models\n",
"- Enables significantly faster training through parallel processing of input/output sequences\n",
"- Can reach production-quality performance in as little as twelve hours on modern GPU hardware\n",
"\n",
"Real-world Applications:\n",
"- Machine translation systems\n",
"- Natural language understanding tasks\n",
"- Reading comprehension\n",
"- Abstractive summarization\n",
"- Text entailment analysis\n",
"- Constituency parsing (achieving 92.7 F1 score in semi-supervised settings)\n",
"- Adaptable to both large and limited training data scenarios\n",
"\n",
"Scalability Benefits:\n",
"- Highly parallelizable architecture enables efficient scaling across multiple GPUs\n",
"- Constant computational complexity for relating any input/output positions\n",
"- Effective handling of long-range dependencies in sequences\n",
"- Maintains performance quality while scaling to larger datasets and model sizes\n",
"- Generalizes well across different tasks and domains without architectural changes\n",
"- Supports efficient inference and deployment in production environments\n",
"\n"
]
}
],
"source": [
"report_text = \"\\n\\n\".join([block.template for block in report.blocks])\n",
"print(report_text)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 5. Edit the final report\n",
"\n",
"Now that we have a report, we can edit it.\n",
"\n",
"We can use the `asuggest_edits` method to get suggestions for edits, and then use the `aaccept_edit`/`areject_edit` methods to apply them.\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"Justification for change: \n",
"I'd suggest changing \"TLDR\" to \"Executive Summary\" which is more appropriate for a professional or academic report. This term is widely used in formal documents and better reflects the nature of this concise overview section while maintaining the same function of providing a quick summary of the key points.\n",
"\n",
"Proposed changes:\n",
"## Executive Summary\n",
"\n",
"The Transformer introduced a revolutionary architecture that relies entirely on attention mechanisms, eliminating the need for recurrence or convolution in sequence processing. Its key innovations include multi-head self-attention for parallel processing of input sequences, scaled dot-product attention for efficient computation, and positional encodings for sequence order awareness. The model achieved breakthrough results in machine translation (28.4 BLEU on English-to-German, 41.8 BLEU on English-to-French) while requiring significantly less training time than previous approaches, training in 3.5 days on 8 GPUs. This architecture demonstrated that attention mechanisms alone are sufficient for state-of-the-art sequence modeling, setting a new direction for natural language processing.\n",
"==================\n"
]
}
],
"source": [
"suggestions = await report_client.asuggest_edits(\n",
" \"Can you change the TLDR header to something more professional?\"\n",
")\n",
"for suggestion in suggestions:\n",
" print(\"Justification for change:\", suggestion.justification)\n",
" print(\"Proposed changes:\")\n",
" for block in suggestion.blocks:\n",
" print(block.template)\n",
" print(\"==================\")"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"Changing to \"Executive Summary\" sounds reasonable, lets accept that!\n"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"for suggestion in suggestions:\n",
" await report_client.aaccept_edit(suggestion)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"## 7. Print the final report\n",
"\n",
"Now that we have a report, we can print it."
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"# Attention Is All You Need: A Pure Attention-Based Architecture for Neural Machine Translation\n",
"\n",
"## Executive Summary\n",
"\n",
"The Transformer introduced a revolutionary architecture that relies entirely on attention mechanisms, eliminating the need for recurrence or convolution in sequence processing. Its key innovations include multi-head self-attention for parallel processing of input sequences, scaled dot-product attention for efficient computation, and positional encodings for sequence order awareness. The model achieved breakthrough results in machine translation (28.4 BLEU on English-to-German, 41.8 BLEU on English-to-French) while requiring significantly less training time than previous approaches, training in 3.5 days on 8 GPUs. This architecture demonstrated that attention mechanisms alone are sufficient for state-of-the-art sequence modeling, setting a new direction for natural language processing.\n",
"\n",
"\n",
"## Architecture Details\n",
"\n",
"The Transformer architecture represents a groundbreaking approach to sequence processing, built entirely on attention mechanisms without recurrence or convolution. Here are its key technical details:\n",
"\n",
"Core Components:\n",
"- Encoder-decoder architecture with stacked self-attention and point-wise feed-forward layers\n",
"- Each layer contains two main sub-layers: multi-head self-attention mechanism and position-wise feed-forward network\n",
"- Layer normalization and residual connections between sub-layers\n",
"- No recurrent or convolutional elements, enabling parallel processing\n",
"\n",
"Self-Attention Mechanism:\n",
"- Processes relationships between all positions in a sequence simultaneously\n",
"- Computes attention weights using queries, keys, and values derived from input representations\n",
"- Implements scaled dot-product attention to prevent gradient issues with large input dimensions\n",
"- Allows direct modeling of dependencies regardless of positional distance\n",
"- Uses masking in decoder to prevent leftward information flow and maintain auto-regressive property\n",
"\n",
"Multi-Head Attention:\n",
"- Employs multiple attention heads operating in parallel\n",
"- Each head processes information in different representation subspaces\n",
"- Three types of attention applications:\n",
" 1. Encoder self-attention (all positions attend to each other)\n",
" 2. Decoder self-attention (each position attends to previous positions)\n",
" 3. Encoder-decoder attention (decoder queries attend to encoder outputs)\n",
"- Counteracts reduced resolution from attention averaging through parallel processing\n",
"\n",
"Position-wise Feed-Forward Network:\n",
"- Applied identically to each position separately\n",
"- Consists of two linear transformations with ReLU activation\n",
"- Structure: FFN(x) = max(0, xW1 + b1)W2 + b2\n",
"- Input and output dimensionality: dmodel = 512\n",
"- Inner-layer dimensionality: dff = 2048\n",
"- Parameters vary between layers but remain constant across positions\n",
"\n",
"Position Encoding:\n",
"- Adds positional information to input embeddings\n",
"- Enables the model to consider sequential order without recurrence\n",
"- Implements sinusoidal position encodings to allow model to attend to relative positions\n",
"- Maintains constant number of operations between any two positions, unlike convolutional approaches\n",
"- Allows effective modeling of both local and long-range dependencies\n",
"\n",
"\n",
"\n",
"## Performance and Applications\n",
"\n",
"The Transformer model demonstrates significant performance advantages and practical applications across multiple domains:\n",
"\n",
"Performance Advantages over RNN/CNN Models:\n",
"- Eliminates sequential computation constraints present in RNNs, enabling superior parallelization\n",
"- Reduces operations needed for relating distant positions to a constant number, compared to linear/logarithmic scaling in CNNs\n",
"- Processes all input and output positions simultaneously through self-attention mechanisms\n",
"- Achieves state-of-the-art results while requiring significantly less computational resources\n",
"\n",
"Machine Translation Benchmarks:\n",
"- WMT 2014 English-to-German: 28.4 BLEU score, exceeding previous best results by over 2 BLEU points\n",
"- WMT 2014 English-to-French: 41.8 BLEU score (single-model state-of-the-art)\n",
"- Surpasses performance of existing model ensembles in translation tasks\n",
"\n",
"Training Efficiency:\n",
"- Requires only 3.5 days of training on eight GPUs for state-of-the-art performance\n",
"- Achieves superior results at \"a small fraction of the training costs\" compared to previous models\n",
"- Enables significantly faster training through parallel processing of input/output sequences\n",
"- Can reach production-quality performance in as little as twelve hours on modern GPU hardware\n",
"\n",
"Real-world Applications:\n",
"- Machine translation systems\n",
"- Natural language understanding tasks\n",
"- Reading comprehension\n",
"- Abstractive summarization\n",
"- Text entailment analysis\n",
"- Constituency parsing (achieving 92.7 F1 score in semi-supervised settings)\n",
"- Adaptable to both large and limited training data scenarios\n",
"\n",
"Scalability Benefits:\n",
"- Highly parallelizable architecture enables efficient scaling across multiple GPUs\n",
"- Constant computational complexity for relating any input/output positions\n",
"- Effective handling of long-range dependencies in sequences\n",
"- Maintains performance quality while scaling to larger datasets and model sizes\n",
"- Generalizes well across different tasks and domains without architectural changes\n",
"- Supports efficient inference and deployment in production environments\n",
"\n"
]
}
],
"source": [
"report_response = await report_client.aget()\n",
"report_text = \"\\n\\n\".join([block.template for block in report_response.report.blocks])\n",
"print(report_text)"
]
},
{
"cell_type": "markdown",
"metadata": {},
"source": [
"We can also see the sources for each block!"
]
},
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [
{
"name": "stdout",
"output_type": "stream",
"text": [
"0.99687636\n",
"# Abstract\n",
"\n",
"The dominant sequence transduction models are based on complex recurrent or convolutiona\n",
"==================\n",
"0.99591404\n",
"# 2 Background\n",
"\n",
"The goal of reducing sequential computation also forms the foundation of the Extende\n",
"==================\n",
"0.9951325\n",
"# 1 Introduction\n",
"\n",
"Recurrent neural networks, long short-term memory [13] and gated recurrent [7] neu\n",
"==================\n",
"0.99442345\n",
"# 7 Conclusion\n",
"\n",
"In this work, we presented the Transformer, the first sequence transduction model ba\n",
"==================\n",
"0.9967649\n",
"# 3.2.3 Applications of Attention in our Model\n",
"\n",
"The Transformer uses multi-head attention in three d\n",
"==================\n",
"0.99533635\n",
"# 2 Background\n",
"\n",
"The goal of reducing sequential computation also forms the foundation of the Extende\n",
"==================\n",
"0.9935868\n",
"# Abstract\n",
"\n",
"The dominant sequence transduction models are based on complex recurrent or convolutiona\n",
"==================\n",
"0.98780584\n",
"# Outputs\n",
"\n",
"(shifted right)\n",
"\n",
"Figure 1: The Transformer - model architecture.\n",
"\n",
"The Transformer follows\n",
"==================\n",
"0.9205043\n",
"# 3.3 Position-wise Feed-Forward Networks\n",
"\n",
"In addition to attention sub-layers, each of the layers i\n",
"==================\n",
"0.79581684\n",
"# 1 Introduction\n",
"\n",
"Recurrent neural networks, long short-term memory [13] and gated recurrent [7] neu\n",
"==================\n",
"0.9946774\n",
"# Abstract\n",
"\n",
"The dominant sequence transduction models are based on complex recurrent or convolutiona\n",
"==================\n",
"0.97079873\n",
"# 7 Conclusion\n",
"\n",
"In this work, we presented the Transformer, the first sequence transduction model ba\n",
"==================\n",
"0.9535353\n",
"# 6.3 English Constituency Parsing\n",
"\n",
"To evaluate if the Transformer can generalize to other tasks we \n",
"==================\n",
"0.9514138\n",
"# 2 Background\n",
"\n",
"The goal of reducing sequential computation also forms the foundation of the Extende\n",
"==================\n",
"0.9790758\n",
"# 1 Introduction\n",
"\n",
"Recurrent neural networks, long short-term memory [13] and gated recurrent [7] neu\n",
"==================\n",
"0.92262185\n",
"# Outputs\n",
"\n",
"(shifted right)\n",
"\n",
"Figure 1: The Transformer - model architecture.\n",
"\n",
"The Transformer follows\n",
"==================\n"
]
}
],
"source": [
"for block in report_response.report.blocks:\n",
" # Each block has a list of sources, which are the nodes that were used to generate the block\n",
" for source in block.sources:\n",
" print(source.score)\n",
" print(source.node.text[:100])\n",
" print(\"==================\")"
]
}
],
"metadata": {
"kernelspec": {
"display_name": "llama-parse-aNC435Vv-py3.10",
"language": "python",
"name": "python3"
},
"language_info": {
"codemirror_mode": {
"name": "ipython",
"version": 3
},
"file_extension": ".py",
"mimetype": "text/x-python",
"name": "python",
"nbconvert_exporter": "python",
"pygments_lexer": "ipython3"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
-10
View File
@@ -1,10 +0,0 @@
# LlamaCloud Services Examples - TypeScript
In this folder you will find two end-to-end examples on how to build applications with:
- [LlamaParse](./parse/)
- [LlamaCloud Index](./index)
In TypeScript.
Follow the instruction in each of the example sub-folders to get started!
-3
View File
@@ -1,3 +0,0 @@
LLAMA_CLOUD_API_KEY="llx-***"
OPENAI_API_KEY="sk-***"
PIPELINE_NAME="my-pipeline"
-131
View File
@@ -1,131 +0,0 @@
# LlamaCloud Index Demo
A TypeScript demo application showcasing the power of **LlamaCloud Index** - a fully automated document ingestion and retrieval serviced offered within [LlamaCloud](https://cloud.llamaindex.ai). This demo allows you to ask questions, retrieve relevant contextual information and generate AI-powered responses using OpenAI's GPT models.
## Table of Contents
- [Features](#features)
- [Prerequisites](#prerequisites)
- [Installation](#installation)
- [Usage](#usage)
- [Start the Demo](#start-the-demo)
- [Development Mode](#development-mode)
- [Build the Project](#build-the-project)
- [Code Quality](#code-quality)
- [Quick Commands Reference](#quick-commands-reference)
- [How It Works](#how-it-works)
- [API Dependencies](#api-dependencies)
- [Troubleshooting](#troubleshooting)
- [Common Issues](#common-issues)
- [License](#license)
- [Contributing](#contributing)
## Features
- 🤖 **RAG**: Simple-yet-effective Retrieval Augmented Generation pipeline built on top of LlamaCloud Index adn OpenAI
- 🎨 **Beautiful CLI**: Styled console interface with colors and ASCII art
-**Fast Development**: Hot reload support with watch mode
- 🛠️ **TypeScript**: Full TypeScript support with strict type checking
## Prerequisites
- Node.js (version 18 or higher)
- pnpm package manager
- OpenAI API key
- LlamaCloud API key
- An existing LlamaCloud Index pipeline
## Installation
1. Clone the repository:
```bash
git clone <repository-url>
cd llamaparse-demo
```
2. Install dependencies:
```bash
pnpm install
```
3. Set up your environment variables:
```bash
export OPENAI_API_KEY="your-openai-api-key"
export LLAMA_CLOUD_API_KEY="your-llamacloud-api-key"
export PIPELINE_NAME="your-pipeline-name"
```
4. Or write them into a `.env` file:
```env
OPENAI_API_KEY="your-openai-api-key"
LLAMA_CLOUD_API_KEY="your-llamacloud-api-key"
PIPELINE_NAME="your-pipeline-name"
```
## Usage
### Start the Demo
```bash
pnpm run start
```
The application will display a welcome screen and prompt you to start chatting!
### Development Mode
For development with hot reload:
```bash
pnpm run dev
```
### Build the Project
```bash
pnpm run build
```
### Code Quality
Format code:
```bash
pnpm run format
```
Lint code:
```bash
pnpm run lint
```
## How It Works
1. **Message Input**: Enter a message
2. **Retrieval**: Several nodes are retrieved from the LlamaCloud index you specified
3. **AI Response Generation**: The retrieved information is passed on to the AI model, along with its relevance score, and a reply to your original message is generated starting from that.
4. **Results**: View the AI-generated summary in your terminal
## Troubleshooting
### Common Issues
1. **Module Resolution Errors**: Ensure you're using Node.js 18+ and have all dependencies installed
2. **API Key Issues**: Verify your OpenAI and LlamaCloud API keys are correctly set
## License
MIT License - see the [LICENSE](../../../LICENSE) file for details.
## Contributing
1. Fork the repository
2. Create a feature branch
3. Make your changes
4. Run `pnpm format` and `pnpm lint`
5. Submit a pull request
-15
View File
@@ -1,15 +0,0 @@
import js from "@eslint/js";
import globals from "globals";
import tseslint from "typescript-eslint";
import { defineConfig } from "eslint/config";
export default defineConfig([
{
files: ["**/*.{js,mjs,cjs,ts,mts,cts}"],
plugins: { js },
extends: ["js/recommended"],
languageOptions: { globals: globals.browser },
},
{ files: ["**/*.js"], languageOptions: { sourceType: "script" } },
tseslint.configs.recommended,
]);
-48
View File
@@ -1,48 +0,0 @@
{
"name": "llama-chat",
"version": "0.1.0",
"description": "Demo for LlamaCloud Index in TypeScript",
"type": "module",
"main": "index.js",
"scripts": {
"test": "echo \"There are no tests\"",
"start": "pnpm exec tsx src/index.ts",
"lint": "eslint ./src/",
"format": "prettier --write ./src/",
"build": "tsc",
"dev": "pnpm exec tsx --watch src/index.ts"
},
"keywords": [
"ai",
"rag",
"retrieval",
"pipeline",
"llms",
"chatbot"
],
"author": "LlamaIndex",
"license": "MIT",
"packageManager": "pnpm@10.12.4",
"devDependencies": {
"@eslint/js": "^9.32.0",
"@types/figlet": "^1.7.0",
"@types/node": "^24.1.0",
"@typescript-eslint/eslint-plugin": "^8.38.0",
"@typescript-eslint/parser": "^8.38.0",
"eslint": "^9.32.0",
"globals": "^16.3.0",
"jiti": "^2.5.1",
"prettier": "^3.6.2",
"typescript": "^5.8.3",
"typescript-eslint": "^8.38.0"
},
"dependencies": {
"@ai-sdk/openai": "^1.3.23",
"ai": "^4.3.19",
"consola": "^3.4.2",
"dotenv": "^17.2.1",
"figlet": "^1.8.2",
"llama-cloud-services": "link:../../../ts/llama_cloud_services",
"picocolors": "^1.1.1"
}
}
File diff suppressed because it is too large Load Diff
-48
View File
@@ -1,48 +0,0 @@
import { LlamaCloudIndex } from "llama-cloud-services";
import { logger } from "./logger";
import pc from "picocolors";
import {
consoleInput,
retrievalAugmentedGeneration,
renderLogo,
} from "./utils";
import dotenv from "dotenv";
dotenv.config();
export async function main(): Promise<number> {
const index = new LlamaCloudIndex({
name: process.env.PIPELINE_NAME as string,
projectName: "Default",
apiKey: process.env.LLAMA_CLOUD_API_KEY, // can provide API-key in the constructor or in the env
});
const retriever = index.asRetriever({
similarityTopK: 5,
});
await renderLogo();
logger.log(
`Welcome to ${pc.bold(
pc.magentaBright("✨LlamaChat✨"),
)}, our demo for ${pc.bold(pc.green("Index🦙"))}, a ${pc.bold(
pc.cyan("LlamaCloud☁️"),
)} (https://cloud.llamaindex.ai) product!.\nType a question below, and you will get an answer!👇\nIf you wish to exit, just type ${pc.bold(
pc.gray("quit"),
)}.\n`,
);
while (true) {
const userInput = await consoleInput();
if (userInput.toLowerCase() == "quit") {
break;
}
try {
const nodes = await retriever.retrieve(userInput);
const summary = await retrievalAugmentedGeneration(nodes, userInput);
logger.log(`${pc.bold(pc.magentaBright("LlamaChat✨:"))}\n${summary}`);
} catch (error) {
logger.error(`Error processing your request: ${error}`);
}
}
return 0;
}
main().catch(console.error);
-8
View File
@@ -1,8 +0,0 @@
import { createConsola } from "consola";
import type { ConsolaInstance } from "consola";
export const logger: ConsolaInstance = createConsola({
formatOptions: {
date: false,
},
});
-56
View File
@@ -1,56 +0,0 @@
import { generateText } from "ai";
import { openai } from "@ai-sdk/openai";
import { NodeWithScore, MetadataMode } from "llamaindex";
import * as readline from "readline/promises";
import figlet from "figlet";
import pc from "picocolors";
export async function renderLogo(): Promise<void> {
const logoText = figlet.textSync("LlamaChat", {
font: "ANSI Shadow",
horizontalLayout: "default",
verticalLayout: "default",
width: 100,
whitespaceBreak: true,
});
// Add some styling with picocolors
const styledLogo = pc.bold(pc.yellowBright(logoText));
// Add some padding/margin
console.log("\n");
console.log(styledLogo);
console.log(pc.gray("─".repeat(60)));
console.log("\n");
}
export async function consoleInput(): Promise<string> {
const rl = readline.createInterface({
input: process.stdin,
output: process.stdout,
});
const answer = await rl.question(pc.cyanBright("You✨:"));
rl.close();
return answer;
}
export async function retrievalAugmentedGeneration(
nodes: NodeWithScore[],
prompt: string,
): Promise<string> {
let mainText: string = "";
for (const node of nodes) {
mainText += `\t{information: '${node.node.getContent(
MetadataMode.ALL,
)}', relevanceScore: '${node.score ?? "no score"}'}\n`;
}
const { text } = await generateText({
model: openai("gpt-4.1"),
prompt: `[\n${mainText}\n]\n\nBased on the information you are given and on the relevance score of that (where -1 means no score available), answer to this user prompt: '${prompt}'`,
});
return text;
}
-22
View File
@@ -1,22 +0,0 @@
{
"compilerOptions": {
"target": "ES2022",
"module": "ES2022",
"lib": ["ES2022"],
"outDir": "./dist",
"rootDir": "./src",
"strict": true,
"esModuleInterop": true,
"skipLibCheck": true,
"forceConsistentCasingInFileNames": true,
"declaration": true,
"declarationMap": true,
"sourceMap": true,
"types": ["node"],
"moduleResolution": "bundler",
"allowSyntheticDefaultImports": true,
"resolveJsonModule": true
},
"include": ["src/**/*"],
"exclude": ["node_modules", "dist"]
}
-124
View File
@@ -1,124 +0,0 @@
# LlamaParse Demo
A TypeScript demo application showcasing the power of **LlamaParse** - an intelligent document parsing service from [LlamaCloud](https://cloud.llamaindex.ai). This demo allows you to parse various document formats and generate AI-powered summaries using OpenAI's GPT models.
## Table of Contents
- [Features](#features)
- [Prerequisites](#prerequisites)
- [Installation](#installation)
- [Usage](#usage)
- [Start the Demo](#start-the-demo)
- [Development Mode](#development-mode)
- [Build the Project](#build-the-project)
- [Code Quality](#code-quality)
- [Quick Commands Reference](#quick-commands-reference)
- [How It Works](#how-it-works)
- [API Dependencies](#api-dependencies)
- [Troubleshooting](#troubleshooting)
- [Common Issues](#common-issues)
- [License](#license)
- [Contributing](#contributing)
## Features
- 📄 **Document Parsing**: Parse PDFs, Word docs, and other formats using LlamaParse
- 🤖 **AI Summaries**: Generate intelligent summaries using OpenAI GPT-4
- 🎨 **Beautiful CLI**: Styled console interface with colors and ASCII art
-**Fast Development**: Hot reload support with watch mode
- 🛠️ **TypeScript**: Full TypeScript support with strict type checking
## Prerequisites
- Node.js (version 18 or higher)
- pnpm package manager
- OpenAI API key
- LlamaCloud API key
## Installation
1. Clone the repository:
```bash
git clone <repository-url>
cd llamaparse-demo
```
2. Install dependencies:
```bash
pnpm install
```
3. Set up your environment variables:
```bash
# Add your API keys to your environment
export OPENAI_API_KEY="your-openai-api-key"
export LLAMA_CLOUD_API_KEY="your-llamacloud-api-key"
```
## Usage
### Start the Demo
```bash
pnpm run start
```
The application will display a welcome screen and prompt you to enter the path to a document you'd like to process.
### Development Mode
For development with hot reload:
```bash
pnpm run dev
```
### Build the Project
```bash
pnpm run build
```
### Code Quality
Format code:
```bash
pnpm run format
```
Lint code:
```bash
pnpm run lint
```
## How It Works
1. **Document Input**: Enter the path to your document when prompted
2. **Parsing**: LlamaParse processes the document and extracts structured content
3. **AI Summary**: The extracted content is sent to OpenAI GPT-4 for summarization
4. **Results**: View the AI-generated summary in your terminal
## Troubleshooting
### Common Issues
1. **Module Resolution Errors**: Ensure you're using Node.js 18+ and have all dependencies installed
2. **API Key Issues**: Verify your OpenAI and LlamaCloud API keys are correctly set
3. **File Path Errors**: Use absolute paths or ensure relative paths are correct from the project root
## License
MIT License - see the [LICENSE](../../../LICENSE) file for details.
## Contributing
1. Fork the repository
2. Create a feature branch
3. Make your changes
4. Run `pnpm format` and `pnpm lint`
5. Submit a pull request
Binary file not shown.
-15
View File
@@ -1,15 +0,0 @@
import js from "@eslint/js";
import globals from "globals";
import tseslint from "typescript-eslint";
import { defineConfig } from "eslint/config";
export default defineConfig([
{
files: ["**/*.{js,mjs,cjs,ts,mts,cts}"],
plugins: { js },
extends: ["js/recommended"],
languageOptions: { globals: globals.browser },
},
{ files: ["**/*.js"], languageOptions: { sourceType: "script" } },
tseslint.configs.recommended,
]);
-47
View File
@@ -1,47 +0,0 @@
{
"name": "llamaparse-demo",
"version": "0.1.0",
"description": "Demo for LlamaParse in TypeScript",
"type": "module",
"main": "index.js",
"scripts": {
"test": "echo \"There are no tests\"",
"start": "pnpm exec tsx src/index.ts",
"lint": "eslint ./src/",
"format": "prettier --write ./src/",
"build": "tsc",
"dev": "pnpm exec tsx --watch src/index.ts"
},
"keywords": [
"ai",
"ocr",
"parsing",
"intelligent-document-processing",
"pdf",
"llms"
],
"author": "LlamaIndex",
"license": "MIT",
"packageManager": "pnpm@10.12.4",
"devDependencies": {
"@eslint/js": "^9.32.0",
"@types/figlet": "^1.7.0",
"@types/node": "^24.1.0",
"@typescript-eslint/eslint-plugin": "^8.38.0",
"@typescript-eslint/parser": "^8.38.0",
"eslint": "^9.32.0",
"globals": "^16.3.0",
"jiti": "^2.5.1",
"prettier": "^3.6.2",
"typescript": "^5.8.3",
"typescript-eslint": "^8.38.0"
},
"dependencies": {
"@ai-sdk/openai": "^1.3.23",
"ai": "^4.3.19",
"consola": "^3.4.2",
"figlet": "^1.8.2",
"llama-cloud-services": "link:../../../ts/llama_cloud_services",
"picocolors": "^1.1.1"
}
}

Some files were not shown because too many files have changed in this diff Show More