mirror of
https://github.com/Mintplex-Labs/langchain-python.git
synced 2026-07-19 21:33:31 -04:00
Grobid parser for Scientific Articles from PDF (#6729)
### Scientific Article PDF Parsing via Grobid
`Description:`
This change adds the GrobidParser class, which uses the Grobid library
to parse scientific articles into a universal XML format containing the
article title, references, sections, section text etc. The GrobidParser
uses a local Grobid server to return PDFs document as XML and parses the
XML to optionally produce documents of individual sentences or of whole
paragraphs. Metadata includes the text, paragraph number, pdf relative
bboxes, pages (text may overlap over two pages), section title
(Introduction, Methodology etc), section_number (i.e 1.1, 2.3), the
title of the paper and finally the file path.
Grobid parsing is useful beyond standard pdf parsing as it accurately
outputs sections and paragraphs within them. This allows for
post-fitering of results for specific sections i.e. limiting results to
the methodology section or results. While sections are split via
headings, ideally they could be classified specifically into
introduction, methodology, results, discussion, conclusion. I'm
currently experimenting with chatgpt-3.5 for this function, which could
later be implemented as a textsplitter.
`Dependencies:`
For use, the grobid repo must be cloned and Java must be installed, for
colab this is:
```
!apt-get install -y openjdk-11-jdk -q
!update-alternatives --set java /usr/lib/jvm/java-11-openjdk-amd64/bin/java
!git clone https://github.com/kermitt2/grobid.git
os.environ["JAVA_HOME"] = "/usr/lib/jvm/java-11-openjdk-amd64"
os.chdir('grobid')
!./gradlew clean install
```
Once installed the server is ran on localhost:8070 via
```
get_ipython().system_raw('nohup ./gradlew run > grobid.log 2>&1 &')
```
@rlancemartin, @eyurtsev
Twitter Handle: @Corranmac
Grobid Demo Notebook is
[here](https://colab.research.google.com/drive/1X-St_mQRmmm8YWtct_tcJNtoktbdGBmd?usp=sharing).
---------
Co-authored-by: rlm <pexpresss31@gmail.com>
This commit is contained in:
@@ -1,4 +1,5 @@
|
||||
from langchain.document_loaders.parsers.audio import OpenAIWhisperParser
|
||||
from langchain.document_loaders.parsers.grobid import GrobidParser
|
||||
from langchain.document_loaders.parsers.html import BS4HTMLParser
|
||||
from langchain.document_loaders.parsers.language import LanguageParser
|
||||
from langchain.document_loaders.parsers.pdf import (
|
||||
@@ -11,6 +12,7 @@ from langchain.document_loaders.parsers.pdf import (
|
||||
|
||||
__all__ = [
|
||||
"BS4HTMLParser",
|
||||
"GrobidParser",
|
||||
"LanguageParser",
|
||||
"OpenAIWhisperParser",
|
||||
"PDFMinerParser",
|
||||
|
||||
@@ -0,0 +1,142 @@
|
||||
from typing import Dict, Iterator, List, Union
|
||||
|
||||
import requests
|
||||
|
||||
from langchain.docstore.document import Document
|
||||
from langchain.document_loaders.base import BaseBlobParser
|
||||
from langchain.document_loaders.blob_loaders import Blob
|
||||
|
||||
|
||||
class ServerUnavailableException(Exception):
|
||||
pass
|
||||
|
||||
|
||||
class GrobidParser(BaseBlobParser):
|
||||
"""Loader that uses Grobid to load article PDF files."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
segment_sentences: bool,
|
||||
grobid_server: str = "http://localhost:8070/api/processFulltextDocument",
|
||||
) -> None:
|
||||
self.segment_sentences = segment_sentences
|
||||
self.grobid_server = grobid_server
|
||||
try:
|
||||
requests.get(grobid_server)
|
||||
except requests.exceptions.RequestException:
|
||||
print(
|
||||
"GROBID server does not appear up and running, \
|
||||
please ensure Grobid is installed and the server is running"
|
||||
)
|
||||
raise ServerUnavailableException
|
||||
|
||||
def process_xml(
|
||||
self, file_path: str, xml_data: str, segment_sentences: bool
|
||||
) -> Iterator[Document]:
|
||||
"""Process the XML file from Grobin."""
|
||||
|
||||
try:
|
||||
from bs4 import BeautifulSoup
|
||||
except ImportError:
|
||||
raise ImportError(
|
||||
"`bs4` package not found, please install it with " "`pip install bs4`"
|
||||
)
|
||||
soup = BeautifulSoup(xml_data, "xml")
|
||||
sections = soup.find_all("div")
|
||||
title = soup.find_all("title")[0].text
|
||||
chunks = []
|
||||
for section in sections:
|
||||
sect = section.find("head")
|
||||
if sect is not None:
|
||||
for i, paragraph in enumerate(section.find_all("p")):
|
||||
chunk_bboxes = []
|
||||
paragraph_text = []
|
||||
for i, sentence in enumerate(paragraph.find_all("s")):
|
||||
paragraph_text.append(sentence.text)
|
||||
sbboxes = []
|
||||
for bbox in sentence.get("coords").split(";"):
|
||||
box = bbox.split(",")
|
||||
sbboxes.append(
|
||||
{
|
||||
"page": box[0],
|
||||
"x": box[1],
|
||||
"y": box[2],
|
||||
"h": box[3],
|
||||
"w": box[4],
|
||||
}
|
||||
)
|
||||
chunk_bboxes.append(sbboxes)
|
||||
if segment_sentences is True:
|
||||
fpage, lpage = sbboxes[0]["page"], sbboxes[-1]["page"]
|
||||
sentence_dict = {
|
||||
"text": sentence.text,
|
||||
"para": str(i),
|
||||
"bboxes": [sbboxes],
|
||||
"section_title": sect.text,
|
||||
"section_number": sect.get("n"),
|
||||
"pages": (fpage, lpage),
|
||||
}
|
||||
chunks.append(sentence_dict)
|
||||
if segment_sentences is not True:
|
||||
fpage, lpage = (
|
||||
chunk_bboxes[0][0]["page"],
|
||||
chunk_bboxes[-1][-1]["page"],
|
||||
)
|
||||
paragraph_dict = {
|
||||
"text": "".join(paragraph_text),
|
||||
"para": str(i),
|
||||
"bboxes": chunk_bboxes,
|
||||
"section_title": sect.text,
|
||||
"section_number": sect.get("n"),
|
||||
"pages": (fpage, lpage),
|
||||
}
|
||||
chunks.append(paragraph_dict)
|
||||
|
||||
yield from [
|
||||
Document(
|
||||
page_content=chunk["text"],
|
||||
metadata=dict(
|
||||
{
|
||||
"text": str(chunk["text"]),
|
||||
"para": str(chunk["para"]),
|
||||
"bboxes": str(chunk["bboxes"]),
|
||||
"pages": str(chunk["pages"]),
|
||||
"section_title": str(chunk["section_title"]),
|
||||
"section_number": str(chunk["section_number"]),
|
||||
"paper_title": str(title),
|
||||
"file_path": str(file_path),
|
||||
}
|
||||
),
|
||||
)
|
||||
for chunk in chunks
|
||||
]
|
||||
|
||||
def lazy_parse(self, blob: Blob) -> Iterator[Document]:
|
||||
file_path = blob.source
|
||||
if file_path is None:
|
||||
raise ValueError("blob.source cannot be None.")
|
||||
pdf = open(file_path, "rb")
|
||||
files = {"input": (file_path, pdf, "application/pdf", {"Expires": "0"})}
|
||||
try:
|
||||
data: Dict[str, Union[str, List[str]]] = {}
|
||||
for param in ["generateIDs", "consolidateHeader", "segmentSentences"]:
|
||||
data[param] = "1"
|
||||
data["teiCoordinates"] = ["head", "s"]
|
||||
files = files or {}
|
||||
r = requests.request(
|
||||
"POST",
|
||||
self.grobid_server,
|
||||
headers=None,
|
||||
params=None,
|
||||
files=files,
|
||||
data=data,
|
||||
timeout=60,
|
||||
)
|
||||
xml_data = r.text
|
||||
except requests.exceptions.ReadTimeout:
|
||||
xml_data = None
|
||||
|
||||
if xml_data is None:
|
||||
return iter([])
|
||||
else:
|
||||
return self.process_xml(file_path, xml_data, self.segment_sentences)
|
||||
Reference in New Issue
Block a user