irongit

Tools for Ozone roofing

able to extract lead

huncholanehuncholaneauthored
parent 0e278cecommit 3e29cf409c89ed51e6cd3104daa2974aa093d3b9Browse files

1 file changed, +60 -49

+60-49main.py
@@ -8,6 +8,56 @@ from openai import OpenAI
88 load_dotenv()
99 JSON_REGEX = re.compile(r"```json(.*)```", re.DOTALL)
1010
11+
12+def parse_lead_file(file):
13+ """Uses the OpenAI API to parse the lead file and return raw content."""
14+ message_file = client.files.create(file=file, purpose="assistants")
15+ content = """
16+ Parse the leads from the PDF files into a json list in the following format.
17+ [
18+ {
19+ "todays_date": "",
20+ "agent_name": "ANDREA",
21+ "roofing_company": "CORE FOUR - AUSTIN",
22+ "names": "BORAN ZHAO & TANIA BETANCOURT",
23+ "appointment_date": "4/18/24",
24+ "time": "2PM",
25+ "phone": "979-218-4997",
26+ "email": "TANIA@TXSTATE.EDU",
27+ "address": "524 CARISMATIC LN",
28+ "city": "AUSTIN",
29+ "state": "TX",
30+ "zip_code": "78748",
31+ "additional_address": "",
32+ "insurance_provider": "METROPOLITAN/FARMERS",
33+ "age_of_roof": "3 years",
34+ "animals_in_yard": "Yes",
35+ "last_roof_inspection": "",
36+ "notes": "",
37+ "contact_number": "303-908-3193"
38+ }
39+ ]
40+ """
41+ thread = client.beta.threads.create(
42+ messages=[
43+ {
44+ "role": "user",
45+ "content": content,
46+ "attachments": [
47+ {"file_id": message_file.id, "tools": [{"type": "file_search"}]}
48+ ],
49+ }
50+ ]
51+ )
52+ run = client.beta.threads.runs.create_and_poll(
53+ thread_id=thread.id, assistant_id=assistant.id
54+ )
55+ messages = list(
56+ client.beta.threads.messages.list(thread_id=thread.id, run_id=run.id)
57+ )
58+ return messages[0].content[0].text.value
59+
60+
1161 # Get the pdf filenames
1262 pdf_files = []
1363 file_streams = []
@@ -28,54 +78,15 @@ assistant = client.beta.assistants.create(
2878
2979 # Ask for the leads
3080 file = file_streams[0]
31-message_file = client.files.create(file=file, purpose="assistants")
32-content = """
33-Parse the leads from the PDF files into a json list in the following format.
34-[
35- {
36- "todays_date": "",
37- "agent_name": "ANDREA",
38- "roofing_company": "CORE FOUR - AUSTIN",
39- "names": "BORAN ZHAO & TANIA BETANCOURT",
40- "appointment_date": "4/18/24",
41- "time": "2PM",
42- "phone": "979-218-4997",
43- "email": "TANIA@TXSTATE.EDU",
44- "address": "524 CARISMATIC LN",
45- "city": "AUSTIN",
46- "state": "TX",
47- "zip_code": "78748",
48- "additional_address": "",
49- "insurance_provider": "METROPOLITAN/FARMERS",
50- "age_of_roof": "3 years",
51- "animals_in_yard": "Yes",
52- "last_roof_inspection": "",
53- "notes": "",
54- "contact_number": "303-908-3193"
55- }
56-]
57-"""
58-thread = client.beta.threads.create(
59- messages=[
60- {
61- "role": "user",
62- "content": content,
63- "attachments": [
64- {"file_id": message_file.id, "tools": [{"type": "file_search"}]}
65- ],
66- }
67- ]
68-)
69-run = client.beta.threads.runs.create_and_poll(
70- thread_id=thread.id, assistant_id=assistant.id
71-)
72-messages = list(client.beta.threads.messages.list(thread_id=thread.id, run_id=run.id))
73-message_content = messages[0].content[0].text.value
74-json_match = JSON_REGEX.search(message_content)
75-if json_match:
81+json_leads = []
82+for file in file_streams:
83+ message = parse_lead_file(file)
84+ json_match = JSON_REGEX.search(message)
7685 json_content = json_match.group(1)
7786 parsed_json = json.loads(json_content)
78- print(json.dumps(parsed_json, indent=2))
79-else:
80- print("No JSON content found.")
81- print(message_content)
87+ json_leads.extend(parsed_json)
88+
89+import pandas as pd
90+
91+df = pd.DataFrame(json_leads)
92+df.to_csv("leads.csv", index=False)