mirror of
https://github.com/Mintplex-Labs/langchain-python.git
synced 2026-07-25 04:26:41 -04:00
2ceb807da2
# Add PDF parser implementations
This PR separates the data loading from the parsing for a number of
existing PDF loaders.
Parser tests have been designed to help encourage developers to create a
consistent interface for parsing PDFs.
This interface can be made more consistent in the future by adding
information into the initializer on desired behavior with respect to splitting by
page etc.
This code is expected to be backwards compatible -- with the exception
of a bug fix with pymupdf parser which was returning `bytes` in the page
content rather than strings.
Also changing the lazy parser method of document loader to return an
Iterator rather than Iterable over documents.
## Before submitting
<!-- If you're adding a new integration, include an integration test and
an example notebook showing its use! -->
## Who can review?
Community members can review the PR once tests pass. Tag
maintainers/contributors who might be interested:
@
<!-- For a quicker response, figure out the right person to tag with @
@hwchase17 - project lead
Tracing / Callbacks
- @agola11
Async
- @agola11
DataLoader Abstractions
- @eyurtsev
LLM/Chat Wrappers
- @hwchase17
- @agola11
Tools / Toolkits
- @vowelparrot
-->
88 lines
2.7 KiB
Python
88 lines
2.7 KiB
Python
"""Abstract interface for document loader implementations."""
|
|
from abc import ABC, abstractmethod
|
|
from typing import Iterator, List, Optional
|
|
|
|
from langchain.document_loaders.blob_loaders import Blob
|
|
from langchain.schema import Document
|
|
from langchain.text_splitter import RecursiveCharacterTextSplitter, TextSplitter
|
|
|
|
|
|
class BaseLoader(ABC):
|
|
"""Interface for loading documents.
|
|
|
|
Implementations should implement the lazy-loading method using generators
|
|
to avoid loading all documents into memory at once.
|
|
|
|
The `load` method will remain as is for backwards compatibility, but it's
|
|
implementation should be just `list(self.lazy_load())`.
|
|
"""
|
|
|
|
# Sub-classes should implement this method
|
|
# as return list(self.lazy_load()).
|
|
# This method returns a List which is materialized in memory.
|
|
@abstractmethod
|
|
def load(self) -> List[Document]:
|
|
"""Load data into document objects."""
|
|
|
|
def load_and_split(
|
|
self, text_splitter: Optional[TextSplitter] = None
|
|
) -> List[Document]:
|
|
"""Load documents and split into chunks."""
|
|
if text_splitter is None:
|
|
_text_splitter: TextSplitter = RecursiveCharacterTextSplitter()
|
|
else:
|
|
_text_splitter = text_splitter
|
|
docs = self.load()
|
|
return _text_splitter.split_documents(docs)
|
|
|
|
# Attention: This method will be upgraded into an abstractmethod once it's
|
|
# implemented in all the existing subclasses.
|
|
def lazy_load(
|
|
self,
|
|
) -> Iterator[Document]:
|
|
"""A lazy loader for document content."""
|
|
raise NotImplementedError(
|
|
f"{self.__class__.__name__} does not implement lazy_load()"
|
|
)
|
|
|
|
|
|
class BaseBlobParser(ABC):
|
|
"""Abstract interface for blob parsers.
|
|
|
|
A blob parser is provides a way to parse raw data stored in a blob into one
|
|
or more documents.
|
|
|
|
The parser can be composed with blob loaders, making it easy to re-use
|
|
a parser independent of how the blob was originally loaded.
|
|
"""
|
|
|
|
@abstractmethod
|
|
def lazy_parse(self, blob: Blob) -> Iterator[Document]:
|
|
"""Lazy parsing interface.
|
|
|
|
Subclasses are required to implement this method.
|
|
|
|
Args:
|
|
blob: Blob instance
|
|
|
|
Returns:
|
|
Generator of documents
|
|
"""
|
|
|
|
def parse(self, blob: Blob) -> List[Document]:
|
|
"""Eagerly parse the blob into a document or documents.
|
|
|
|
This is a convenience method for interactive development environment.
|
|
|
|
Production applications should favor the lazy_parse method instead.
|
|
|
|
Subclasses should generally not over-ride this parse method.
|
|
|
|
Args:
|
|
blob: Blob instance
|
|
|
|
Returns:
|
|
List of documents
|
|
"""
|
|
return list(self.lazy_parse(blob))
|