Tools for Ozone roofing
able to extract lead
4 files changed, +142 -11
+81-0main.py
| @@ -0,0 +1,81 @@ | ||
| 1 | +import json | |
| 2 | +import os | |
| 3 | +import re | |
| 4 | + | |
| 5 | +from dotenv import load_dotenv | |
| 6 | +from openai import OpenAI | |
| 7 | + | |
| 8 | +load_dotenv() | |
| 9 | +JSON_REGEX = re.compile(r"```json(.*)```", re.DOTALL) | |
| 10 | + | |
| 11 | +# Get the pdf filenames | |
| 12 | +pdf_files = [] | |
| 13 | +file_streams = [] | |
| 14 | +for root, dirs, files in os.walk(".data/leads"): | |
| 15 | + for file in files: | |
| 16 | + if file.endswith(".pdf"): | |
| 17 | + pdf_files.append(os.path.join(root, file)) | |
| 18 | + file_streams.append(open(os.path.join(root, file), "rb")) | |
| 19 | + | |
| 20 | +# Create the assistant | |
| 21 | +client = OpenAI(api_key=os.getenv("OPENAI_API_KEY")) | |
| 22 | +assistant = client.beta.assistants.create( | |
| 23 | + name="Lead Parser", | |
| 24 | + description="Parse leads from PDF files", | |
| 25 | + model="gpt-3.5-turbo", | |
| 26 | + tools=[{"type": "file_search"}], | |
| 27 | +) | |
| 28 | + | |
| 29 | +# Ask for the leads | |
| 30 | +file = file_streams[0] | |
| 31 | +message_file = client.files.create(file=file, purpose="assistants") | |
| 32 | +content = """ | |
| 33 | +Parse the leads from the PDF files into a json list in the following format. | |
| 34 | +[ | |
| 35 | + { | |
| 36 | + "todays_date": "", | |
| 37 | + "agent_name": "ANDREA", | |
| 38 | + "roofing_company": "CORE FOUR - AUSTIN", | |
| 39 | + "names": "BORAN ZHAO & TANIA BETANCOURT", | |
| 40 | + "appointment_date": "4/18/24", | |
| 41 | + "time": "2PM", | |
| 42 | + "phone": "979-218-4997", | |
| 43 | + "email": "TANIA@TXSTATE.EDU", | |
| 44 | + "address": "524 CARISMATIC LN", | |
| 45 | + "city": "AUSTIN", | |
| 46 | + "state": "TX", | |
| 47 | + "zip_code": "78748", | |
| 48 | + "additional_address": "", | |
| 49 | + "insurance_provider": "METROPOLITAN/FARMERS", | |
| 50 | + "age_of_roof": "3 years", | |
| 51 | + "animals_in_yard": "Yes", | |
| 52 | + "last_roof_inspection": "", | |
| 53 | + "notes": "", | |
| 54 | + "contact_number": "303-908-3193" | |
| 55 | + } | |
| 56 | +] | |
| 57 | +""" | |
| 58 | +thread = client.beta.threads.create( | |
| 59 | + messages=[ | |
| 60 | + { | |
| 61 | + "role": "user", | |
| 62 | + "content": content, | |
| 63 | + "attachments": [ | |
| 64 | + {"file_id": message_file.id, "tools": [{"type": "file_search"}]} | |
| 65 | + ], | |
| 66 | + } | |
| 67 | + ] | |
| 68 | +) | |
| 69 | +run = client.beta.threads.runs.create_and_poll( | |
| 70 | + thread_id=thread.id, assistant_id=assistant.id | |
| 71 | +) | |
| 72 | +messages = list(client.beta.threads.messages.list(thread_id=thread.id, run_id=run.id)) | |
| 73 | +message_content = messages[0].content[0].text.value | |
| 74 | +json_match = JSON_REGEX.search(message_content) | |
| 75 | +if json_match: | |
| 76 | + json_content = json_match.group(1) | |
| 77 | + parsed_json = json.loads(json_content) | |
| 78 | + print(json.dumps(parsed_json, indent=2)) | |
| 79 | +else: | |
| 80 | + print("No JSON content found.") | |
| 81 | + print(message_content) |
+0-0notes.ipynb
No content changes (mode or rename only).
+41-0test.py
| @@ -0,0 +1,41 @@ | ||
| 1 | +import re | |
| 2 | +import json | |
| 3 | + | |
| 4 | +text = """ | |
| 5 | +No JSON content found. | |
| 6 | +I have found a lead sheet from the file "BORAN ZHAO- CORE AUSTIN TX.pdf" that contains information about a potential client. Here is the extracted lead information in JSON format: | |
| 7 | + | |
| 8 | +```json | |
| 9 | +[ | |
| 10 | + { | |
| 11 | + "Agent's Name": "ANDREA", | |
| 12 | + "ROOFING COMPANY": "CORE FOUR - AUSTIN", | |
| 13 | + "Name(s)": "BORAN ZHAO & TANIA BETANCOURT", | |
| 14 | + "Appointment Date": "4/18/24", | |
| 15 | + "Time": "2PM", | |
| 16 | + "Phone": "979-218-4997", | |
| 17 | + "Email": "TANIA@TXSTATE.EDU", | |
| 18 | + "Address": "524 CARISMATIC LN", | |
| 19 | + "City": "AUSTIN", | |
| 20 | + "State": "TX", | |
| 21 | + "ZIP CODE": "78748", | |
| 22 | + "Insurance Provider": "METROPOLITAN/FARMERS", | |
| 23 | + "Age of Roof": "3", | |
| 24 | + "Any Animal in Yard?": "YES", | |
| 25 | + "When was the last time you had a Roof inspection?": "", | |
| 26 | + "Additional Address": "ALT # 512-291-2707" | |
| 27 | + } | |
| 28 | +] | |
| 29 | +``` | |
| 30 | + | |
| 31 | +This JSON list captures the lead details from the provided document. | |
| 32 | +""" | |
| 33 | + | |
| 34 | +regex = re.compile(r"```json(.*)```", re.DOTALL) | |
| 35 | +match = regex.search(text) | |
| 36 | +if match: | |
| 37 | + json_content = match.group(1) | |
| 38 | + parsed_json = json.loads(json_content) | |
| 39 | + print(json.dumps(parsed_json, indent=2)) | |
| 40 | +else: | |
| 41 | + print("No JSON content found.") |
+20-11to_csv.py
| @@ -7,10 +7,12 @@ load_dotenv() | ||
| 7 | 7 | |
| 8 | 8 | # Get the pdf filenames |
| 9 | 9 | pdf_files = [] |
| 10 | +file_streams = [] | |
| 10 | 11 | for root, dirs, files in os.walk(".data/leads"): |
| 11 | 12 | for file in files: |
| 12 | 13 | if file.endswith(".pdf"): |
| 13 | 14 | pdf_files.append(os.path.join(root, file)) |
| 15 | + file_streams.append(open(os.path.join(root, file), "rb")) | |
| 14 | 16 | |
| 15 | 17 | # Create the assistant |
| 16 | 18 | client = OpenAI(api_key=os.getenv("OPENAI_API_KEY")) |
| @@ -22,21 +24,28 @@ assistant = client.beta.assistants.create( | ||
| 22 | 24 | ) |
| 23 | 25 | |
| 24 | 26 | # Create the vector store |
| 25 | -vector_store = client.beta.vector_stores.create(name="Leads") | |
| 26 | -file_streams = [open(file, "rb") for file in pdf_files] | |
| 27 | -file_batch = client.beta.vector_stores.file_batches.upload_and_poll( | |
| 28 | - vector_store_id=vector_store.id, files=file_streams | |
| 29 | -) | |
| 30 | -print(file_batch.status) | |
| 31 | -print(file_batch.file_counts) | |
| 27 | +# vector_store = client.beta.vector_stores.create(name="Leads") | |
| 28 | +# file_batch = client.beta.vector_stores.file_batches.upload_and_poll( | |
| 29 | +# vector_store_id=vector_store.id, files=file_streams | |
| 30 | +# ) | |
| 31 | +# print(file_batch.status) | |
| 32 | +# print(file_batch.file_counts) | |
| 32 | 33 | |
| 33 | 34 | # Update assistant to use the vector store |
| 34 | -assistant = client.beta.assistants.update( | |
| 35 | - assistant_id=assistant.id, | |
| 36 | - tool_resources={"file_search": {"vector_store_ids": [vector_store.id]}}, | |
| 37 | -) | |
| 35 | +# assistant = client.beta.assistants.update( | |
| 36 | +# assistant_id=assistant.id, | |
| 37 | +# tool_resources={"file_search": {"vector_store_ids": [vector_store.id]}}, | |
| 38 | +# ) | |
| 38 | 39 | |
| 39 | 40 | # Create a thread |
| 41 | +for file in file_streams: | |
| 42 | + response = client.beta.threads.create_and_run_poll( | |
| 43 | + assistant_id=assistant.id, | |
| 44 | + message={ | |
| 45 | + "role": "user", | |
| 46 | + "content": "Parse the leads from the PDF files into a json list.", | |
| 47 | + }, | |
| 48 | + ) | |
| 40 | 49 | thread = client.beta.threads.create() |
| 41 | 50 | message = client.beta.threads.messages.create( |
| 42 | 51 | thread_id=thread.id, |