Tools for Ozone roofing
have the lead converter
4 files changed, +86 -173
+0-92main.py
| @@ -1,92 +0,0 @@ | ||
| 1 | -import json | |
| 2 | -import os | |
| 3 | -import re | |
| 4 | - | |
| 5 | -from dotenv import load_dotenv | |
| 6 | -from openai import OpenAI | |
| 7 | - | |
| 8 | -load_dotenv() | |
| 9 | -JSON_REGEX = re.compile(r"```json(.*)```", re.DOTALL) | |
| 10 | - | |
| 11 | - | |
| 12 | -def parse_lead_file(file): | |
| 13 | - """Uses the OpenAI API to parse the lead file and return raw content.""" | |
| 14 | - message_file = client.files.create(file=file, purpose="assistants") | |
| 15 | - content = """ | |
| 16 | - Parse the leads from the PDF files into a json list in the following format. | |
| 17 | - [ | |
| 18 | - { | |
| 19 | - "todays_date": "", | |
| 20 | - "agent_name": "ANDREA", | |
| 21 | - "roofing_company": "CORE FOUR - AUSTIN", | |
| 22 | - "names": "BORAN ZHAO & TANIA BETANCOURT", | |
| 23 | - "appointment_date": "4/18/24", | |
| 24 | - "time": "2PM", | |
| 25 | - "phone": "979-218-4997", | |
| 26 | - "email": "TANIA@TXSTATE.EDU", | |
| 27 | - "address": "524 CARISMATIC LN", | |
| 28 | - "city": "AUSTIN", | |
| 29 | - "state": "TX", | |
| 30 | - "zip_code": "78748", | |
| 31 | - "additional_address": "", | |
| 32 | - "insurance_provider": "METROPOLITAN/FARMERS", | |
| 33 | - "age_of_roof": "3 years", | |
| 34 | - "animals_in_yard": "Yes", | |
| 35 | - "last_roof_inspection": "", | |
| 36 | - "notes": "", | |
| 37 | - "contact_number": "303-908-3193" | |
| 38 | - } | |
| 39 | - ] | |
| 40 | - """ | |
| 41 | - thread = client.beta.threads.create( | |
| 42 | - messages=[ | |
| 43 | - { | |
| 44 | - "role": "user", | |
| 45 | - "content": content, | |
| 46 | - "attachments": [ | |
| 47 | - {"file_id": message_file.id, "tools": [{"type": "file_search"}]} | |
| 48 | - ], | |
| 49 | - } | |
| 50 | - ] | |
| 51 | - ) | |
| 52 | - run = client.beta.threads.runs.create_and_poll( | |
| 53 | - thread_id=thread.id, assistant_id=assistant.id | |
| 54 | - ) | |
| 55 | - messages = list( | |
| 56 | - client.beta.threads.messages.list(thread_id=thread.id, run_id=run.id) | |
| 57 | - ) | |
| 58 | - return messages[0].content[0].text.value | |
| 59 | - | |
| 60 | - | |
| 61 | -# Get the pdf filenames | |
| 62 | -pdf_files = [] | |
| 63 | -file_streams = [] | |
| 64 | -for root, dirs, files in os.walk(".data/leads"): | |
| 65 | - for file in files: | |
| 66 | - if file.endswith(".pdf"): | |
| 67 | - pdf_files.append(os.path.join(root, file)) | |
| 68 | - file_streams.append(open(os.path.join(root, file), "rb")) | |
| 69 | - | |
| 70 | -# Create the assistant | |
| 71 | -client = OpenAI(api_key=os.getenv("OPENAI_API_KEY")) | |
| 72 | -assistant = client.beta.assistants.create( | |
| 73 | - name="Lead Parser", | |
| 74 | - description="Parse leads from PDF files", | |
| 75 | - model="gpt-3.5-turbo", | |
| 76 | - tools=[{"type": "file_search"}], | |
| 77 | -) | |
| 78 | - | |
| 79 | -# Ask for the leads | |
| 80 | -file = file_streams[0] | |
| 81 | -json_leads = [] | |
| 82 | -for file in file_streams: | |
| 83 | - message = parse_lead_file(file) | |
| 84 | - json_match = JSON_REGEX.search(message) | |
| 85 | - json_content = json_match.group(1) | |
| 86 | - parsed_json = json.loads(json_content) | |
| 87 | - json_leads.extend(parsed_json) | |
| 88 | - | |
| 89 | -import pandas as pd | |
| 90 | - | |
| 91 | -df = pd.DataFrame(json_leads) | |
| 92 | -df.to_csv("leads.csv", index=False) |
+0-0notes.ipynb
No content changes (mode or rename only).
+0-41test.py
| @@ -1,41 +0,0 @@ | ||
| 1 | -import re | |
| 2 | -import json | |
| 3 | - | |
| 4 | -text = """ | |
| 5 | -No JSON content found. | |
| 6 | -I have found a lead sheet from the file "BORAN ZHAO- CORE AUSTIN TX.pdf" that contains information about a potential client. Here is the extracted lead information in JSON format: | |
| 7 | - | |
| 8 | -```json | |
| 9 | -[ | |
| 10 | - { | |
| 11 | - "Agent's Name": "ANDREA", | |
| 12 | - "ROOFING COMPANY": "CORE FOUR - AUSTIN", | |
| 13 | - "Name(s)": "BORAN ZHAO & TANIA BETANCOURT", | |
| 14 | - "Appointment Date": "4/18/24", | |
| 15 | - "Time": "2PM", | |
| 16 | - "Phone": "979-218-4997", | |
| 17 | - "Email": "TANIA@TXSTATE.EDU", | |
| 18 | - "Address": "524 CARISMATIC LN", | |
| 19 | - "City": "AUSTIN", | |
| 20 | - "State": "TX", | |
| 21 | - "ZIP CODE": "78748", | |
| 22 | - "Insurance Provider": "METROPOLITAN/FARMERS", | |
| 23 | - "Age of Roof": "3", | |
| 24 | - "Any Animal in Yard?": "YES", | |
| 25 | - "When was the last time you had a Roof inspection?": "", | |
| 26 | - "Additional Address": "ALT # 512-291-2707" | |
| 27 | - } | |
| 28 | -] | |
| 29 | -``` | |
| 30 | - | |
| 31 | -This JSON list captures the lead details from the provided document. | |
| 32 | -""" | |
| 33 | - | |
| 34 | -regex = re.compile(r"```json(.*)```", re.DOTALL) | |
| 35 | -match = regex.search(text) | |
| 36 | -if match: | |
| 37 | - json_content = match.group(1) | |
| 38 | - parsed_json = json.loads(json_content) | |
| 39 | - print(json.dumps(parsed_json, indent=2)) | |
| 40 | -else: | |
| 41 | - print("No JSON content found.") |
+86-40to_csv.py
| @@ -1,18 +1,89 @@ | ||
| 1 | +import json | |
| 1 | 2 | import os |
| 3 | +import re | |
| 2 | 4 | |
| 5 | +import pandas as pd | |
| 3 | 6 | from dotenv import load_dotenv |
| 4 | 7 | from openai import OpenAI |
| 5 | 8 | |
| 6 | 9 | load_dotenv() |
| 10 | +JSON_REGEX = re.compile(r"```json(.*)```", re.DOTALL) | |
| 11 | +API_KEY = os.getenv("OPENAI_API_KEY", None) | |
| 12 | +if API_KEY is None: | |
| 13 | + print("Please set the OPENAI_API_KEY environment variable.") | |
| 14 | + exit(1) | |
| 15 | + | |
| 16 | + | |
| 17 | +def parse_lead_file(filename) -> str: | |
| 18 | + """Uses the OpenAI API to parse the lead file and return raw content.""" | |
| 19 | + file = open(filename, "rb") | |
| 20 | + message_file = client.files.create(file=file, purpose="assistants") | |
| 21 | + content = """ | |
| 22 | + Parse the leads from the PDF files into a json list in the following format. | |
| 23 | + [ | |
| 24 | + { | |
| 25 | + "today_date": "", | |
| 26 | + "agent_name": "ANDREA", | |
| 27 | + "roofing_company": "CORE FOUR - AUSTIN", | |
| 28 | + "names": "BORAN ZHAO & TANIA BETANCOURT", | |
| 29 | + "appointment_date": "4/18/24", | |
| 30 | + "time": "2PM", | |
| 31 | + "phone": "979-218-4997", | |
| 32 | + "email": "TANIA@TXSTATE.EDU", | |
| 33 | + "address": "524 CARISMATIC LN", | |
| 34 | + "city": "AUSTIN", | |
| 35 | + "state": "TX", | |
| 36 | + "zip_code": "78748", | |
| 37 | + "additional_address": "", | |
| 38 | + "insurance_provider": "METROPOLITAN/FARMERS", | |
| 39 | + "age_of_roof": "3 years", | |
| 40 | + "animals_in_yard": "Yes", | |
| 41 | + "last_roof_inspection": "", | |
| 42 | + "notes": "", | |
| 43 | + } | |
| 44 | + ] | |
| 45 | + """ | |
| 46 | + thread = client.beta.threads.create( | |
| 47 | + messages=[ | |
| 48 | + { | |
| 49 | + "role": "user", | |
| 50 | + "content": content, | |
| 51 | + "attachments": [ | |
| 52 | + {"file_id": message_file.id, "tools": [{"type": "file_search"}]} | |
| 53 | + ], | |
| 54 | + } | |
| 55 | + ] | |
| 56 | + ) | |
| 57 | + run = client.beta.threads.runs.create_and_poll( | |
| 58 | + thread_id=thread.id, assistant_id=assistant.id | |
| 59 | + ) | |
| 60 | + messages = list( | |
| 61 | + client.beta.threads.messages.list(thread_id=thread.id, run_id=run.id) | |
| 62 | + ) | |
| 63 | + return messages[0].content[0].text.value | |
| 64 | + | |
| 65 | + | |
| 66 | +def pdf_to_json(leads: list, filename: str, max_retries=3, attempt=0): | |
| 67 | + """Parse the leads from a PDF file and append them to the leads list.""" | |
| 68 | + message = parse_lead_file(filename) | |
| 69 | + json_match = JSON_REGEX.search(message) | |
| 70 | + try: | |
| 71 | + json_content = json_match.group(1) | |
| 72 | + parsed_json = json.loads(json_content) | |
| 73 | + leads.extend(parsed_json) | |
| 74 | + except Exception as e: | |
| 75 | + print( | |
| 76 | + f"Error parsing JSON from {filename}:\n{e}\nCurrently on attempt {attempt + 1} of {max_retries}" | |
| 77 | + ) | |
| 78 | + pdf_to_json(leads, filename, max_retries, attempt + 1) | |
| 79 | + | |
| 7 | 80 | |
| 8 | 81 | # Get the pdf filenames |
| 9 | 82 | pdf_files = [] |
| 10 | -file_streams = [] | |
| 11 | 83 | for root, dirs, files in os.walk(".data/leads"): |
| 12 | 84 | for file in files: |
| 13 | 85 | if file.endswith(".pdf"): |
| 14 | 86 | pdf_files.append(os.path.join(root, file)) |
| 15 | - file_streams.append(open(os.path.join(root, file), "rb")) | |
| 16 | 87 | |
| 17 | 88 | # Create the assistant |
| 18 | 89 | client = OpenAI(api_key=os.getenv("OPENAI_API_KEY")) |
| @@ -23,44 +94,19 @@ assistant = client.beta.assistants.create( | ||
| 23 | 94 | tools=[{"type": "file_search"}], |
| 24 | 95 | ) |
| 25 | 96 | |
| 26 | -# Create the vector store | |
| 27 | -# vector_store = client.beta.vector_stores.create(name="Leads") | |
| 28 | -# file_batch = client.beta.vector_stores.file_batches.upload_and_poll( | |
| 29 | -# vector_store_id=vector_store.id, files=file_streams | |
| 30 | -# ) | |
| 31 | -# print(file_batch.status) | |
| 32 | -# print(file_batch.file_counts) | |
| 33 | - | |
| 34 | -# Update assistant to use the vector store | |
| 35 | -# assistant = client.beta.assistants.update( | |
| 36 | -# assistant_id=assistant.id, | |
| 37 | -# tool_resources={"file_search": {"vector_store_ids": [vector_store.id]}}, | |
| 38 | -# ) | |
| 97 | +# Ask for the leads | |
| 98 | +from threading import Thread | |
| 39 | 99 | |
| 40 | -# Create a thread | |
| 41 | -for file in file_streams: | |
| 42 | - response = client.beta.threads.create_and_run_poll( | |
| 43 | - assistant_id=assistant.id, | |
| 44 | - message={ | |
| 45 | - "role": "user", | |
| 46 | - "content": "Parse the leads from the PDF files into a json list.", | |
| 47 | - }, | |
| 48 | - ) | |
| 49 | -thread = client.beta.threads.create() | |
| 50 | -message = client.beta.threads.messages.create( | |
| 51 | - thread_id=thread.id, | |
| 52 | - role="user", | |
| 53 | - content="Parse the leads from the PDF files into a json list.", | |
| 54 | -) | |
| 100 | +json_leads = [] | |
| 101 | +threads = [ | |
| 102 | + Thread(target=pdf_to_json, args=(json_leads, filename), daemon=True) | |
| 103 | + for filename in pdf_files | |
| 104 | +] | |
| 105 | +for thread in threads: | |
| 106 | + thread.start() | |
| 107 | +for thread in threads: | |
| 108 | + thread.join() | |
| 55 | 109 | |
| 56 | -# Run the assistant | |
| 57 | -run = client.beta.threads.runs.poll( | |
| 58 | - thread_id=thread.id, | |
| 59 | - assistant_id=assistant.id, | |
| 60 | - instructions="Parse the leads from the PDF files into a json list.", | |
| 61 | -) | |
| 62 | 110 | |
| 63 | -if run.status == "completed": | |
| 64 | - messages = client.beta.threads.messages.list(thread_id=thread.id) | |
| 65 | - for message in messages: | |
| 66 | - print(message.content) | |
| 111 | +df = pd.DataFrame(json_leads) | |
| 112 | +df.to_csv(".data/leads.csv", index=False) |