mirror of
https://github.com/langchain-ai/langchain.git
synced 2026-10-09 19:35:20 +03:00
#9578 --------- Co-authored-by: Leonid Kuligin <kuligin@google.com> Co-authored-by: Eugene Yurtsev <eyurtsev@gmail.com>
6.8 KiB
6.8 KiB
In [1]:
from langchain.document_loaders.blob_loaders import Blob
from langchain.document_loaders.parsers import DocAIParserIn [2]:
PROJECT = "PUT_SOMETHING_HERE"
GCS_OUTPUT_PATH = "PUT_SOMETHING_HERE"
PROCESSOR_NAME = "PUT_SOMETHING_HERE"In [3]:
parser = DocAIParser(location="us", processor_name=PROCESSOR_NAME, gcs_output_path=GCS_OUTPUT_PATH)In [4]:
blob = Blob(path="gs://vertex-pgt/examples/goog-exhibit-99-1-q1-2023-19.pdf")In [5]:
docs = list(parser.lazy_parse(blob))In [8]:
print(len(docs))11
In [9]:
operations = parser.docai_parse([blob])
print([op.operation.name for op in operations])['projects/543079149601/locations/us/operations/16447136779727347991']
In [10]:
parser.is_running(operations)Out [10]:
True
In [11]:
parser.is_running(operations)Out [11]:
False
In [12]:
results = parser.get_results(operations)
print(results[0])DocAIParsingResults(source_path='gs://vertex-pgt/examples/goog-exhibit-99-1-q1-2023-19.pdf', parsed_path='gs://vertex-pgt/test/run1/16447136779727347991/0')
In [15]:
docs = list(parser.parse_from_results(results))In [16]:
print(len(docs))11