| 1 | import os |
| 2 | |
| 3 | from dotenv import load_dotenv |
| 4 | from openai import OpenAI |
| 5 | |
| 6 | load_dotenv() |
| 7 | |
| 8 | |
| 9 | pdf_files = [] |
| 10 | for root, dirs, files in os.walk(".data/leads"): |
| 11 | for file in files: |
| 12 | if file.endswith(".pdf"): |
| 13 | pdf_files.append(os.path.join(root, file)) |
| 14 | |
| 15 | |
| 16 | client = OpenAI(api_key=os.getenv("OPENAI_API_KEY")) |
| 17 | assistant = client.beta.assistants.create( |
| 18 | name="Lead Parser", |
| 19 | description="Parse leads from PDF files", |
| 20 | model="gpt-3.5-turbo", |
| 21 | tools=[{"type": "file_search"}], |
| 22 | ) |
| 23 | |
| 24 | |
| 25 | vector_store = client.beta.vector_stores.create(name="Leads") |
| 26 | file_streams = [open(file, "rb") for file in pdf_files] |
| 27 | file_batch = client.beta.vector_stores.file_batches.upload_and_poll( |
| 28 | vector_store_id=vector_store.id, files=file_streams |
| 29 | ) |
| 30 | print(file_batch.status) |
| 31 | print(file_batch.file_counts) |
| 32 | |
| 33 | |
| 34 | assistant = client.beta.assistants.update( |
| 35 | assistant_id=assistant.id, |
| 36 | tool_resources={"file_search": {"vector_store_ids": [vector_store.id]}}, |
| 37 | ) |
| 38 | |
| 39 | |
| 40 | thread = client.beta.threads.create() |
| 41 | message = client.beta.threads.messages.create( |
| 42 | thread_id=thread.id, |
| 43 | role="user", |
| 44 | content="Parse the leads from the PDF files into a json list.", |
| 45 | ) |
| 46 | |
| 47 | |
| 48 | run = client.beta.threads.runs.poll( |
| 49 | thread_id=thread.id, |
| 50 | assistant_id=assistant.id, |
| 51 | instructions="Parse the leads from the PDF files into a json list.", |
| 52 | ) |
| 53 | |
| 54 | if run.status == "completed": |
| 55 | messages = client.beta.threads.messages.list(thread_id=thread.id) |
| 56 | for message in messages: |
| 57 | print(message.content) |