summaryrefslogtreecommitdiff
path: root/main.py
diff options
context:
space:
mode:
authorvin <git@vineetk.net>2025-07-19 01:16:12 -0400
committervin <git@vineetk.net>2025-07-19 01:16:12 -0400
commitbb8c6a7ed9b5b5c450c61408f432980be8e6b120 (patch)
treee1630cf3bcfa5d009c759c6e8b3a63e114457950 /main.py
parentd9fdeaf5af49f065c2a5041d6e6230a2f371f53a (diff)
remove prototyping llm prints, use tqdm, use arguments as inputs
Diffstat (limited to 'main.py')
-rw-r--r--main.py274
1 files changed, 134 insertions, 140 deletions
diff --git a/main.py b/main.py
index 27b588c..e344a59 100644
--- a/main.py
+++ b/main.py
@@ -1,7 +1,109 @@
1import os
1import re 2import re
2import sys 3import sys
3from openai import OpenAI 4from openai import OpenAI
4from datetime import datetime, timedelta 5from datetime import datetime, timedelta
6from tqdm import tqdm
7
8SYSTEM_PROMPT_TEMPLATE = """
9You are an expert podcast content analyzer. Your task is to identify and extract pre-recorded dynamic advertising segments from podcast transcripts. These ads typically have a distinct tone shift, often using more direct, persuasive language, and frequently include calls to action, product mentions, or specific brand names.
10
11You will be provided with segments of a podcast transcript, formatted with timestamps. Your output must be ONLY the start and end timestamp of the identified advertising segment, separated by a hyphen, in the format HH:MM:SS.mmm-HH:MM:SS.mmm. If there are multiple ad segments, output each on a new line. If no ad segments are found, output nothing.
12
13**Strict Rules:**
14- Only identify segments that are clearly pre-recorded dynamic ads with a noticeable tone shift.
15- Do NOT include host-read sponsorships that blend naturally with the content. Focus only on the "distracting" pre-recorded elements.
16- The end of an ad segment is marked by the return to the original podcast discussion, a clear transition to new content, or the end of the ad's persuasive language and calls to action.
17- Ensure the start and end times correspond precisely to the ad segment in the transcript.
18- Output ONLY the timestamp range(s) in the specified format. No other text, no JSON, no explanations.
19
20**Examples:**
21
22---
23**Example 1 Input:**
24[00:00:00.000 --> 00:00:02.480] This episode is brought to you by Apple Cash.
25[00:00:02.480 --> 00:00:07.040] Sending payments used to be clunky, unnecessarily difficult, and weirdly invasive at times,
26[00:00:07.040 --> 00:00:08.560] until I discovered Apple Cash.
27[00:00:08.560 --> 00:00:11.020] With Apple Cash, payments are private by design,
28[00:00:11.020 --> 00:00:14.900] so I don't have to deal with public feeds, awkward reactions, or other payment drama.
29[00:00:14.900 --> 00:00:18.780] I can send cash in messages right in the conversations I'm already having,
30[00:00:18.780 --> 00:00:20.060] which is super convenient.
31[00:00:20.060 --> 00:00:22.600] There's also a cool feature called Tap to Cash
32[00:00:22.600 --> 00:00:25.900] that lets you pay somebody nearby by holding your iPhone near theirs.
33[00:00:25.900 --> 00:00:28.560] Switch to Apple Cash and start sending privately.
34[00:00:28.620 --> 00:00:31.900] Apple Cash services are provided by Green Dot Bank, member FDIC.
35
36**Example 1 Output:**
3700:00:00.000-00:00:31.900
38
39---
40**Example 2 Input:**
41[00:38:07.480 --> 00:38:10.820] This podcast is brought to you by Carvana.
42[00:38:10.820 --> 00:38:13.660] Got a car to sell, but no time to waste?
43[00:38:13.660 --> 00:38:17.820] Hop on to Carvana.com to get a real offer for your car in seconds.
44[00:38:17.820 --> 00:38:23.060] All you have to do is enter your license plate, answer a few quick questions, and if you accept
45[00:38:23.060 --> 00:38:26.000] the offer, Carvana will pay you as soon as you hand the keys over.
46[00:38:26.360 --> 00:38:29.060] They even offer same day pickup in many cities.
47[00:38:29.060 --> 00:38:35.740] Save your time, score some cash, and sell your car the convenient way to Carvana.
48[00:38:36.360 --> 00:38:37.000] Pickup times vary.
49[00:38:37.000 --> 00:38:37.620] Fees may apply.
50
51**Example 2 Output:**
5200:38:07.480-00:38:37.620
53
54---
55**Example 3 Input:**
56[00:14:46.000 --> 00:14:50.860] This episode is brought to you by Diet Coke.
57[00:14:50.860 --> 00:14:53.800] You know that moment when you just need to hit pause and refresh?
58[00:14:54.380 --> 00:14:56.320] An ice-cold Diet Coke isn't just a break.
59[00:14:56.320 --> 00:14:59.800] It's your chance to catch your breath and savor a moment that's all about you.
60[00:15:00.000 --> 00:15:02.540] Always refreshing, still the same great taste.
61[00:15:02.540 --> 00:15:03.500] Diet Coke.
62[00:15:03.500 --> 00:15:04.860] Make time for you time.
63
64**Example 3 Output:**
6500:14:46.000-00:15:04.860
66
67---
68**Example 4 Input:**
69[00:15:00.000 --> 00:15:20.000] So what are your thoughts on the new policy changes that were just announced?
70[00:15:20.000 --> 00:15:45.000] Well, I think it's a step in the right direction, but there are definitely some areas that need more clarification.
71[00:15:45.000 --> 00:16:10.000] For instance, the section on renewable energy credits could be more explicit.
72[00:16:10.000 --> 00:16:15.000] This next part of our conversation might be a bit technical, but I think it's important to delve into the specifics.
73
74**Example 4 Output:**
75
76
77---
78**Example 5 Input:**
79[00:11:27.080 --> 00:11:28.180] You're going to be using the same thing.
80[00:11:28.180 --> 00:11:32.480] You've got to, like, get outside that bubble or you're just, you're going to bore yourself.
81[00:11:32.480 --> 00:11:40.820] This episode of Cortex is brought to you by Squarespace, the all-in-one website platform designed to help you stand out and succeed online.
82[00:11:40.820 --> 00:11:43.920] Whether you're just getting started or scaling your own business,
83[00:11:43.920 --> 00:11:46.840] Squarespace gives you everything you need to claim your domain,
84[00:11:46.840 --> 00:11:49.380] showcase your offerings of a professional website,
85[00:14:50.000 --> 00:14:52.220] grow your brand, and get paid all in one place.
86[00:14:52.220 --> 00:14:54.960] It's so easy to get started with Squarespace.
87[00:14:54.960 --> 00:14:59.380] In fact, they've made it easier than ever before with their new system, Blueprint AI.
88[00:14:59.380 --> 00:15:05.140] Squarespace's AI-enhanced website builder lets you quickly and easily build a site bespoke to your
89business.
90[00:15:05.140 --> 00:15:08.140] Just input some basic information about your industry and goals.
91[00:15:08.140 --> 00:15:09.480] This isn't the only way.
92[00:15:09.480 --> 00:15:12.200] In fact, you can go in and choose from one of their beautiful templates,
93[00:15:12.200 --> 00:15:14.620] their professionally designed beautiful templates,
94[00:15:14.940 --> 00:15:16.900] and customize it to your heart's content.
95[00:15:16.900 --> 00:15:20.800] One of the things that I've always loved about Squarespace is their drag-and-drop tools,
96[00:15:20.800 --> 00:15:26.440] buttons, sliders, selectors, where you can go in and customize your design for your website
97[00:15:26.440 --> 00:15:29.420] by actually looking at the design for your website while you're doing it.
98[00:15:29.460 --> 00:15:30.620] You don't need to know any code.
99[00:15:30.620 --> 00:15:31.880] I love this.
100[00:13:31.200 --> 00:13:35.220] I want to talk about the other devices that are in your working life.
101[00:13:35.220 --> 00:13:39.060] What are the other computers that you use to get things done?
102[00:13:39.060 --> 00:13:45.000] So there are three computers that live in my life.
103
104**Example 5 Output:**
10500:11:32.480-00:13:31.200
106"""
5 107
6def parse_whisper_transcript(transcript_text: str) -> list[dict]: 108def parse_whisper_transcript(transcript_text: str) -> list[dict]:
7 """ 109 """
@@ -10,7 +112,6 @@ def parse_whisper_transcript(transcript_text: str) -> list[dict]:
10 """ 112 """
11 segments = [] 113 segments = []
12 # Regex to match timestamp format [HH:MM:SS.mmm --> HH:MM:SS.mmm] and the following text 114 # Regex to match timestamp format [HH:MM:SS.mmm --> HH:MM:SS.mmm] and the following text
13 # It now correctly handles the leading space before text
14 pattern = re.compile(r"^\[(\d{2}:\d{2}:\d{2}\.\d{3}) --> (\d{2}:\d{2}:\d{2}\.\d{3})\]\s*(.*)$", re.MULTILINE) 115 pattern = re.compile(r"^\[(\d{2}:\d{2}:\d{2}\.\d{3}) --> (\d{2}:\d{2}:\d{2}\.\d{3})\]\s*(.*)$", re.MULTILINE)
15 116
16 for line in transcript_text.strip().split('\n'): 117 for line in transcript_text.strip().split('\n'):
@@ -43,7 +144,6 @@ def timestamp_to_seconds(ts_str: str) -> float:
43 144
44def seconds_to_timestamp(total_seconds: float) -> str: 145def seconds_to_timestamp(total_seconds: float) -> str:
45 """Converts total seconds to HH:MM:SS.mmm string.""" 146 """Converts total seconds to HH:MM:SS.mmm string."""
46 # Split total_seconds into integer seconds and fractional milliseconds
47 integer_seconds = int(total_seconds) 147 integer_seconds = int(total_seconds)
48 milliseconds = int((total_seconds - integer_seconds) * 1000) 148 milliseconds = int((total_seconds - integer_seconds) * 1000)
49 149
@@ -56,7 +156,7 @@ def seconds_to_timestamp(total_seconds: float) -> str:
56def chunk_transcript( 156def chunk_transcript(
57 segments: list[dict], 157 segments: list[dict],
58 trigger_phrase: str = "brought to you by", 158 trigger_phrase: str = "brought to you by",
59 max_chunk_duration_minutes: int = 4 # New parameter for max duration 159 max_chunk_duration_minutes: int = 4
60) -> list[list[dict]]: 160) -> list[list[dict]]:
61 """ 161 """
62 Chunks the transcript segments based on occurrences of a trigger phrase, 162 Chunks the transcript segments based on occurrences of a trigger phrase,
@@ -68,20 +168,14 @@ def chunk_transcript(
68 current_chunk_start_idx = 0 168 current_chunk_start_idx = 0
69 max_chunk_duration_seconds = max_chunk_duration_minutes * 60 169 max_chunk_duration_seconds = max_chunk_duration_minutes * 60
70 170
71 # Ensure there are segments to process
72 if not segments: 171 if not segments:
73 return [] 172 return []
74 173
75 for i, segment in enumerate(segments): 174 for i, segment in enumerate(segments):
76 is_trigger_segment = trigger_phrase.lower() in segment["text"].lower() 175 is_trigger_segment = trigger_phrase.lower() in segment["text"].lower()
77 176
78 # Get the start time of the current logical chunk being built
79 current_chunk_start_time_seconds = timestamp_to_seconds(segments[current_chunk_start_idx]['start_time']) 177 current_chunk_start_time_seconds = timestamp_to_seconds(segments[current_chunk_start_idx]['start_time'])
80
81 # Get the end time of the current segment being considered for inclusion
82 current_segment_end_time_seconds = timestamp_to_seconds(segment['end_time']) 178 current_segment_end_time_seconds = timestamp_to_seconds(segment['end_time'])
83
84 # Calculate the duration if this segment were to be the end of the current chunk
85 current_chunk_duration = current_segment_end_time_seconds - current_chunk_start_time_seconds 179 current_chunk_duration = current_segment_end_time_seconds - current_chunk_start_time_seconds
86 180
87 # Condition 1: New trigger phrase found (and it's not the start of the very first chunk) 181 # Condition 1: New trigger phrase found (and it's not the start of the very first chunk)
@@ -123,138 +217,23 @@ def call_openai_api(client: OpenAI, chunk_text: str, system_prompt: str) -> str:
123 ], 217 ],
124 temperature=0.0, 218 temperature=0.0,
125 ) 219 )
126 print(response) 220 #print(response)
127 # Extract content from the first choice's message
128 return response.choices[0].message.content.strip() 221 return response.choices[0].message.content.strip()
129 except Exception as e: 222 except Exception as e:
130 print(f"Error calling OpenAI API: {e}") 223 print(f"Error calling OpenAI API: {e}")
131 return "" 224 return ""
132 225
133def main(): 226def process_file(input_file: str, output_file: str, client: OpenAI):
134 # --- Configuration --- 227 with open(input_file, "r", encoding="utf-8") as f:
135 client = OpenAI(
136 #api_key="none",
137 api_key="REDACTED",
138 #base_url="http://localhost:8080/v1",
139 base_url="https://api.openai.com/v1",
140 )
141
142 with open(sys.argv[1], "r", encoding="utf-8") as f:
143 full_transcript_content = f.read() 228 full_transcript_content = f.read()
144 229
145 # The system prompt you provided earlier, including examples
146 SYSTEM_PROMPT_TEMPLATE = """
147You are an expert podcast content analyzer. Your task is to identify and extract pre-recorded dynamic advertising segments from podcast transcripts. These ads typically have a distinct tone shift, often using more direct, persuasive language, and frequently include calls to action, product mentions, or specific brand names.
148
149You will be provided with segments of a podcast transcript, formatted with timestamps. Your output must be ONLY the start and end timestamp of the identified advertising segment, separated by a hyphen, in the format HH:MM:SS.mmm-HH:MM:SS.mmm. If there are multiple ad segments, output each on a new line. If no ad segments are found, output nothing.
150
151**Strict Rules:**
152- Only identify segments that are clearly pre-recorded dynamic ads with a noticeable tone shift.
153- Do NOT include host-read sponsorships that blend naturally with the content. Focus only on the "distracting" pre-recorded elements.
154- The end of an ad segment is marked by the return to the original podcast discussion, a clear transition to new content, or the end of the ad's persuasive language and calls to action.
155- Ensure the start and end times correspond precisely to the ad segment in the transcript.
156- Output ONLY the timestamp range(s) in the specified format. No other text, no JSON, no explanations.
157
158**Examples:**
159
160---
161**Example 1 Input:**
162[00:00:00.000 --> 00:00:02.480] This episode is brought to you by Apple Cash.
163[00:00:02.480 --> 00:00:07.040] Sending payments used to be clunky, unnecessarily difficult, and weirdly invasive at times,
164[00:00:07.040 --> 00:00:08.560] until I discovered Apple Cash.
165[00:00:08.560 --> 00:00:11.020] With Apple Cash, payments are private by design,
166[00:00:11.020 --> 00:00:14.900] so I don't have to deal with public feeds, awkward reactions, or other payment drama.
167[00:00:14.900 --> 00:00:18.780] I can send cash in messages right in the conversations I'm already having,
168[00:00:18.780 --> 00:00:20.060] which is super convenient.
169[00:00:20.060 --> 00:00:22.600] There's also a cool feature called Tap to Cash
170[00:00:22.600 --> 00:00:25.900] that lets you pay somebody nearby by holding your iPhone near theirs.
171[00:00:25.900 --> 00:00:28.560] Switch to Apple Cash and start sending privately.
172[00:00:28.620 --> 00:00:31.900] Apple Cash services are provided by Green Dot Bank, member FDIC.
173
174**Example 1 Output:**
17500:00:00.000-00:00:31.900
176
177---
178**Example 2 Input:**
179[00:38:07.480 --> 00:38:10.820] This podcast is brought to you by Carvana.
180[00:38:10.820 --> 00:38:13.660] Got a car to sell, but no time to waste?
181[00:38:13.660 --> 00:38:17.820] Hop on to Carvana.com to get a real offer for your car in seconds.
182[00:38:17.820 --> 00:38:23.060] All you have to do is enter your license plate, answer a few quick questions, and if you accept
183[00:38:23.060 --> 00:38:26.000] the offer, Carvana will pay you as soon as you hand the keys over.
184[00:38:26.360 --> 00:38:29.060] They even offer same day pickup in many cities.
185[00:38:29.060 --> 00:38:35.740] Save your time, score some cash, and sell your car the convenient way to Carvana.
186[00:38:36.360 --> 00:38:37.000] Pickup times vary.
187[00:38:37.000 --> 00:38:37.620] Fees may apply.
188
189**Example 2 Output:**
19000:38:07.480-00:38:37.620
191
192---
193**Example 3 Input:**
194[00:14:46.000 --> 00:14:50.860] This episode is brought to you by Diet Coke.
195[00:14:50.860 --> 00:14:53.800] You know that moment when you just need to hit pause and refresh?
196[00:14:54.380 --> 00:14:56.320] An ice-cold Diet Coke isn't just a break.
197[00:14:56.320 --> 00:14:59.800] It's your chance to catch your breath and savor a moment that's all about you.
198[00:15:00.000 --> 00:15:02.540] Always refreshing, still the same great taste.
199[00:15:02.540 --> 00:15:03.500] Diet Coke.
200[00:15:03.500 --> 00:15:04.860] Make time for you time.
201
202**Example 3 Output:**
20300:14:46.000-00:15:04.860
204
205---
206**Example 4 Input:**
207[00:15:00.000 --> 00:15:20.000] So what are your thoughts on the new policy changes that were just announced?
208[00:15:20.000 --> 00:15:45.000] Well, I think it's a step in the right direction, but there are definitely some areas that need more clarification.
209[00:15:45.000 --> 00:16:10.000] For instance, the section on renewable energy credits could be more explicit.
210[00:16:10.000 --> 00:16:15.000] This next part of our conversation might be a bit technical, but I think it's important to delve into the specifics.
211
212**Example 4 Output:**
213
214
215---
216**Example 5 Input:**
217[00:11:27.080 --> 00:11:28.180] You're going to be using the same thing.
218[00:11:28.180 --> 00:11:32.480] You've got to, like, get outside that bubble or you're just, you're going to bore yourself.
219[00:11:32.480 --> 00:11:40.820] This episode of Cortex is brought to you by Squarespace, the all-in-one website platform designed to help you stand out and succeed online.
220[00:11:40.820 --> 00:11:43.920] Whether you're just getting started or scaling your own business,
221[00:11:43.920 --> 00:11:46.840] Squarespace gives you everything you need to claim your domain,
222[00:11:46.840 --> 00:11:49.380] showcase your offerings of a professional website,
223[00:14:50.000 --> 00:14:52.220] grow your brand, and get paid all in one place.
224[00:14:52.220 --> 00:14:54.960] It's so easy to get started with Squarespace.
225[00:14:54.960 --> 00:14:59.380] In fact, they've made it easier than ever before with their new system, Blueprint AI.
226[00:14:59.380 --> 00:15:05.140] Squarespace's AI-enhanced website builder lets you quickly and easily build a site bespoke to your
227business.
228[00:15:05.140 --> 00:15:08.140] Just input some basic information about your industry and goals.
229[00:15:08.140 --> 00:15:09.480] This isn't the only way.
230[00:15:09.480 --> 00:15:12.200] In fact, you can go in and choose from one of their beautiful templates,
231[00:15:12.200 --> 00:15:14.620] their professionally designed beautiful templates,
232[00:15:14.940 --> 00:15:16.900] and customize it to your heart's content.
233[00:15:16.900 --> 00:15:20.800] One of the things that I've always loved about Squarespace is their drag-and-drop tools,
234[00:15:20.800 --> 00:15:26.440] buttons, sliders, selectors, where you can go in and customize your design for your website
235[00:15:26.440 --> 00:15:29.420] by actually looking at the design for your website while you're doing it.
236[00:15:29.460 --> 00:15:30.620] You don't need to know any code.
237[00:15:30.620 --> 00:15:31.880] I love this.
238[00:13:31.200 --> 00:13:35.220] I want to talk about the other devices that are in your working life.
239[00:13:35.220 --> 00:13:39.060] What are the other computers that you use to get things done?
240[00:13:39.060 --> 00:13:45.000] So there are three computers that live in my life.
241
242**Example 5 Output:**
24300:11:32.480-00:13:31.200
244"""
245 # --- Processing --- 230 # --- Processing ---
246 print("Parsing full transcript...")
247 parsed_segments = parse_whisper_transcript(full_transcript_content) 231 parsed_segments = parse_whisper_transcript(full_transcript_content)
248 print(f"Parsed {len(parsed_segments)} segments.")
249 232
250 print("Chunking transcript...")
251 # Adjust the trigger phrase if "brought to you by" isn't consistently sufficient 233 # Adjust the trigger phrase if "brought to you by" isn't consistently sufficient
252 # For more complex cases, you might use a list of trigger phrases: 234 # For more complex cases, you might use a list of trigger phrases:
253 # ["brought to you by", "a quick word from our sponsor", "we'll be right back"] 235 # ["brought to you by", "a quick word from our sponsor", "we'll be right back"]
254 transcript_chunks = chunk_transcript(parsed_segments, trigger_phrase="brought to you by") 236 transcript_chunks = chunk_transcript(parsed_segments, trigger_phrase="brought to you by")
255 print(f"Created {len(transcript_chunks)} chunks.")
256
257 print(transcript_chunks)
258 237
259 all_ad_timestamps = [] 238 all_ad_timestamps = []
260 239
@@ -264,29 +243,44 @@ business.
264 continue 243 continue
265 244
266 chunk_text_for_api = format_transcript_segment(chunk) 245 chunk_text_for_api = format_transcript_segment(chunk)
267 print(f"\nProcessing Chunk {i+1}/{len(transcript_chunks)} (starting at {chunk[0]['start_time']})...")
268 246
269 # Add the "Your Turn" header before the chunk content 247 # Add the "Your Turn" header before the chunk content
270 user_prompt_content = f"**Your Turn - Input Transcript:**\n{chunk_text_for_api}" 248 user_prompt_content = f"**Your Turn - Input Transcript:**\n{chunk_text_for_api}"
271 249
272 print(user_prompt_content)
273
274 ad_timestamps_str = call_openai_api(client, user_prompt_content, SYSTEM_PROMPT_TEMPLATE) 250 ad_timestamps_str = call_openai_api(client, user_prompt_content, SYSTEM_PROMPT_TEMPLATE)
275 251
276 if ad_timestamps_str: 252 if ad_timestamps_str:
277 # Split by newlines in case the model returns multiple segments 253 # Split by newlines in case the model returns multiple segments
278 found_timestamps = [ts.strip() for ts in ad_timestamps_str.split('\n') if ts.strip()] 254 found_timestamps = [ts.strip() for ts in ad_timestamps_str.split('\n') if ts.strip()]
279 all_ad_timestamps.extend(found_timestamps) 255 all_ad_timestamps.extend(found_timestamps)
280 print(f"Found ad timestamps in chunk {i+1}: {', '.join(found_timestamps)}") 256
281 else: 257 with open(output_file, "w") as f:
282 print(f"No ad timestamps found in chunk {i+1}.") 258 f.write("\n".join(all_ad_timestamps))
283 259
284 print("\n--- All Identified Ad Timestamps ---") 260def main():
285 if all_ad_timestamps: 261 # --- Configuration ---
286 for ts in all_ad_timestamps: 262 api_key = os.getenv("OPENAI_KEY")
287 print(ts) 263 if not api_key:
288 else: 264 print("OPENAI_KEY empty")
289 print("No ad segments found in the entire transcript.") 265 sys.exit(1)
266
267 client = OpenAI(
268 #api_key="none",
269 api_key=api_key,
270 #base_url="http://localhost:8080/v1",
271 base_url="https://api.openai.com/v1",
272 )
273
274 paths = sys.argv[1:]
275 for i, p in enumerate(paths):
276 out = "".join(p.split(".")[:-1]) + "_gpt.txt"
277 if os.path.isfile(out)
278 paths.remove(i)
279 print(paths)
280
281 for p in tqdm(paths):
282 out = "".join(p.split(".")[:-1]) + "_gpt.txt"
283 process_file(p, out, client)
290 284
291if __name__ == "__main__": 285if __name__ == "__main__":
292 main() 286 main()