Compare commits

..

1 Commits

Author SHA1 Message Date
Adrian Lyjak 4cce90aa02 Fix more bugs in publishing 2025-10-03 11:15:54 -04:00
7 changed files with 14 additions and 69 deletions
-5
View File
@@ -1,5 +0,0 @@
---
"llama-cloud-services-py": minor
---
Escaping dollar signs in markdown output in jupyter notebooks to prevent them being interpreted as equation delimiters
-5
View File
@@ -1,5 +0,0 @@
---
"llama-cloud-services-py": patch
---
Make markdown safe for jupyter notebooks
+1 -1
View File
@@ -8,7 +8,7 @@
"scripts": {
"pre-commit-version": "pnpm changeset",
"version": "./scripts/changeset-version.py version",
"publish": "./scripts/changeset-version.py publish --no-js --tag"
"publish": "./scripts/changeset-version.py publish"
},
"devDependencies": {
"prettier": "^3.6.2",
+6 -35
View File
@@ -4,10 +4,7 @@ import re
from pydantic import BaseModel, Field, SerializeAsAny
from typing import Dict, Any, List, Optional
from llama_cloud_services.parse.utils import (
make_api_request,
is_jupyter,
)
from llama_cloud_services.parse.utils import make_api_request
from llama_index.core.async_utils import asyncio_run
from llama_index.core.schema import Document, ImageDocument, ImageNode, TextNode
@@ -261,24 +258,6 @@ class JobResult(BaseModel):
documents = await self.aget_text_documents(split_by_page)
return [TextNode(text=doc.text, metadata=doc.metadata) for doc in documents]
def _format_markdown_for_notebook(self, text: Optional[str]) -> Optional[str]:
"""Format markdown text for Jupyter notebook display by escaping dollar signs."""
if text is None:
return None
def escape_dollar_signs(text: str) -> str:
"""Escape dollar signs in text to prevent Jupyter from interpreting them as LaTeX.
Args:
text: The text to escape
Returns:
Text with dollar signs escaped
"""
return text.replace("$", r"\$")
return escape_dollar_signs(text)
def get_markdown_documents(self, split_by_page: bool = False) -> List[Document]:
"""
Get the markdown documents from the job.
@@ -289,22 +268,17 @@ class JobResult(BaseModel):
if split_by_page:
return [
Document(
text=self._format_markdown_for_notebook(page.md)
if is_jupyter()
else page.md,
text=page.md,
metadata={"page_number": page.page, "file_name": self.file_name},
)
for page in self.pages
]
else:
text = self._page_separator.join(
[page.md if page.md is not None else "" for page in self.pages]
)
return [
Document(
text=self._format_markdown_for_notebook(text)
if is_jupyter()
else text,
text=self._page_separator.join(
[page.md if page.md is not None else "" for page in self.pages]
),
metadata={"file_name": self.file_name},
)
]
@@ -354,10 +328,7 @@ class JobResult(BaseModel):
"""
url = f"{self._base_url}/api/v1/parsing/job/{self.job_id}/result/raw/markdown"
response = await make_api_request(self._client, "GET", url)
markdown = response.content.decode("utf-8")
return (
self._format_markdown_for_notebook(markdown) if is_jupyter() else markdown
)
return response.content.decode("utf-8")
def get_text(self) -> str:
"""
-12
View File
@@ -1,4 +1,3 @@
import functools
import httpx
import itertools
import logging
@@ -357,17 +356,6 @@ def partition_pages(
return
@functools.lru_cache(maxsize=1)
def is_jupyter() -> bool:
"""Check if we're running in a Jupyter environment."""
try:
from IPython import get_ipython
return get_ipython().__class__.__name__ == "ZMQInteractiveShell"
except (ImportError, AttributeError):
return False
def extract_tables_from_json_results(
json_results: List[dict], download_path: str
) -> List[str]:
Generated
+2 -2
View File
@@ -1,5 +1,5 @@
version = 1
revision = 3
revision = 2
requires-python = ">=3.9, <4.0"
resolution-markers = [
"python_full_version >= '3.14'",
@@ -1596,7 +1596,7 @@ wheels = [
[[package]]
name = "llama-cloud-services"
version = "0.6.70"
version = "0.6.69"
source = { editable = "." }
dependencies = [
{ name = "click", version = "8.1.8", source = { registry = "https://pypi.org/simple" }, marker = "python_full_version < '3.10'" },
+5 -9
View File
@@ -109,9 +109,7 @@ def version() -> None:
@cli.command()
@click.option("--tag", is_flag=True, help="Tag the packages after publishing")
@click.option("--dry-run", is_flag=True, help="Dry run the publish")
@click.option("--js/--no-js", default=True, help="Publish the js package")
@click.option("--py/--no-py", default=True, help="Publish the py package")
def publish(tag: bool, dry_run: bool, js: bool, py: bool) -> None:
def publish(tag: bool, dry_run: bool) -> None:
"""Publish all packages."""
# move to the root
os.chdir(Path(__file__).parent.parent)
@@ -124,10 +122,8 @@ def publish(tag: bool, dry_run: bool, js: bool, py: bool) -> None:
raise click.Abort("No token set")
# not general script. Just checks each of the 2 packages to see if they need to be published.
if js:
maybe_publish_npm(dry_run)
if py:
maybe_publish_pypi(dry_run)
maybe_publish_ts_package(dry_run)
maybe_publish_py_packages(dry_run)
if tag:
if dry_run:
@@ -140,7 +136,7 @@ def publish(tag: bool, dry_run: bool, js: bool, py: bool) -> None:
_run_command(["git", "push", "--tags"])
def maybe_publish_npm(dry_run: bool) -> None:
def maybe_publish_ts_package(dry_run: bool) -> None:
"""Publish the ts package if it needs to be published."""
target_dir = Path("ts/llama_cloud_services")
ts_path_package = target_dir / "package.json"
@@ -173,7 +169,7 @@ def maybe_publish_npm(dry_run: bool) -> None:
_run_command(["pnpm", "publish"], cwd=target_dir)
def maybe_publish_pypi(dry_run: bool) -> None:
def maybe_publish_py_packages(dry_run: bool) -> None:
"""Publish the py packages if they need to be published."""
for pyproject in list(Path("py").glob("*/pyproject.toml")) + [
Path("py/pyproject.toml")