irongit

Tools for Ozone roofing

ozone/to_csv.py
66 lines1.9 KBPython
1import os
2
3from dotenv import load_dotenv
4from openai import OpenAI
5
6load_dotenv()
7
8# Get the pdf filenames
9pdf_files = []
10file_streams = []
11for root, dirs, files in os.walk(".data/leads"):
12 for file in files:
13 if file.endswith(".pdf"):
14 pdf_files.append(os.path.join(root, file))
15 file_streams.append(open(os.path.join(root, file), "rb"))
16
17# Create the assistant
18client = OpenAI(api_key=os.getenv("OPENAI_API_KEY"))
19assistant = client.beta.assistants.create(
20 name="Lead Parser",
21 description="Parse leads from PDF files",
22 model="gpt-3.5-turbo",
23 tools=[{"type": "file_search"}],
24)
25
26# Create the vector store
27# vector_store = client.beta.vector_stores.create(name="Leads")
28# file_batch = client.beta.vector_stores.file_batches.upload_and_poll(
29# vector_store_id=vector_store.id, files=file_streams
30# )
31# print(file_batch.status)
32# print(file_batch.file_counts)
33
34# Update assistant to use the vector store
35# assistant = client.beta.assistants.update(
36# assistant_id=assistant.id,
37# tool_resources={"file_search": {"vector_store_ids": [vector_store.id]}},
38# )
39
40# Create a thread
41for file in file_streams:
42 response = client.beta.threads.create_and_run_poll(
43 assistant_id=assistant.id,
44 message={
45 "role": "user",
46 "content": "Parse the leads from the PDF files into a json list.",
47 },
48 )
49thread = client.beta.threads.create()
50message = client.beta.threads.messages.create(
51 thread_id=thread.id,
52 role="user",
53 content="Parse the leads from the PDF files into a json list.",
54)
55
56# Run the assistant
57run = client.beta.threads.runs.poll(
58 thread_id=thread.id,
59 assistant_id=assistant.id,
60 instructions="Parse the leads from the PDF files into a json list.",
61)
62
63if run.status == "completed":
64 messages = client.beta.threads.messages.list(thread_id=thread.id)
65 for message in messages:
66 print(message.content)