#!/usr/bin/env python3 """Extract a Fireflies.ai (or similar) meeting transcript from a saved .mhtml page. Usage: python3 extract_transcript.py Writes .md next to the input file (same directory, same basename, .md extension), containing the meeting title/date if found and the speaker-by-speaker transcript with timestamps. """ import email import re import sys from pathlib import Path try: from bs4 import BeautifulSoup except ImportError: sys.exit( "Missing dependency 'beautifulsoup4'. Install it with:\n" " pip3 install beautifulsoup4" ) MONTHS = "Jan|Feb|Mar|Apr|May|Jun|Jul|Aug|Sep|Oct|Nov|Dec" DATE_RE = re.compile(rf"({MONTHS})[a-z]* \d{{1,2}},? \d{{4}}(?:, \d{{1,2}}:\d{{2}} ?[AP]M)?") # Fireflies virtualizes the transcript list - if the user didn't scroll # through the whole thing before saving the page, only the visible portion # ends up in the DOM/mhtml, leaving silent gaps. A gap this long between # consecutive timestamps is a strong signal that's what happened. GAP_WARNING_SECONDS = 90 def parse_timestamp(ts): if not ts: return None parts = ts.split(":") try: parts = [int(p) for p in parts] except ValueError: return None seconds = 0 for p in parts: seconds = seconds * 60 + p return seconds def find_large_gaps(lines): gaps = [] prev_ts = None prev_idx = None for idx, (ts, _, _) in enumerate(lines): secs = parse_timestamp(ts) if secs is None: continue if prev_ts is not None and secs - prev_ts >= GAP_WARNING_SECONDS: gaps.append((prev_idx, idx, secs - prev_ts)) prev_ts, prev_idx = secs, idx return gaps def load_html_parts(mhtml_path): with open(mhtml_path, "rb") as f: msg = email.message_from_binary_file(f) for part in msg.walk(): if part.get_content_type() != "text/html": continue payload = part.get_payload(decode=True) if not payload: continue charset = part.get_content_charset() or "utf-8" yield payload.decode(charset, errors="replace") def find_transcript_container(html): soup = BeautifulSoup(html, "html.parser") container = soup.find(id=lambda i: i and i.endswith("content-transcript")) if container is not None: return soup, container # Fallback: some captures use the ScrollArea class directly without the # radix id being present in this particular MIME part. candidates = soup.find_all( class_=lambda c: c and any("ScrollArea-styled__Root" in x for x in (c if isinstance(c, list) else [c])) ) if candidates: # The transcript panel is reliably the largest ScrollArea on the page. best = max(candidates, key=lambda el: len(el.get_text())) if len(best.get_text(strip=True)) > 0: return soup, best return soup, None def extract_paragraphs(container): paragraphs = container.find_all( "div", id=lambda i: i and i.startswith("transcript-paragraph-") ) lines = [] last_name = None for p in paragraphs: name_span = p.find("span", class_="name") name = name_span.get_text(strip=True) if name_span else (last_name or "Unknown speaker") last_name = name ts_span = p.find("span", attrs={"text-decoration": "underline"}) ts = ts_span.get_text(strip=True) if ts_span else "" content_div = p.find("div", class_=lambda c: c and "ContentPost-styled__Content" in c) text = content_div.get_text(" ", strip=True) if content_div else p.get_text(" ", strip=True) if not text: continue lines.append((ts, name, text)) return lines def find_inactive_transcript_tab(html_parts): """Detect the common failure case: the page was saved with a different tab (usually Notes) active, so Fireflies never mounted the transcript panel into the DOM at all - there's nothing to extract, not a selector mismatch.""" for html in html_parts: soup = BeautifulSoup(html, "html.parser") tab = soup.find( attrs={"role": "tab"}, id=lambda i: i and i.endswith("trigger-transcript"), ) if tab is not None: active_tab = soup.find(attrs={"role": "tab", "data-state": "active"}) active_name = active_tab.get_text(strip=True) if active_tab else "another tab" return tab.get("data-state") != "active", active_name return False, None def extract_title_and_date(html_parts): title = None date = None for html in html_parts: soup = BeautifulSoup(html, "html.parser") if title is None and soup.title and soup.title.get_text(strip=True): title = soup.title.get_text(strip=True) if date is None: m = DATE_RE.search(soup.get_text(" ", strip=True)) if m: date = m.group(0) if title and date: break return title, date def main(): if len(sys.argv) != 2: sys.exit(f"Usage: {sys.argv[0]} ") src = Path(sys.argv[1]).expanduser() if not src.is_file(): sys.exit(f"File not found: {src}") html_parts = list(load_html_parts(src)) if not html_parts: sys.exit("No text/html part found in this .mhtml file - is it a valid MIME HTML capture?") lines = [] for html in html_parts: _, container = find_transcript_container(html) if container is None: continue lines = extract_paragraphs(container) if lines: break if not lines: tab_inactive, active_name = find_inactive_transcript_tab(html_parts) if tab_inactive: sys.exit( f"No transcript found - this page was saved with the " f"'{active_name}' tab open, not 'Transcript'. Fireflies only " f"renders the active tab's content into the page, so the " f"transcript panel isn't in this file at all. Reopen the " f"meeting, click the 'Transcript' tab, wait for it to load " f"and scroll to the end (so the full transcript renders), " f"then re-save the page and try again." ) sys.exit( "Could not find a transcript panel in this file. This script targets " "Fireflies.ai-style pages (a ScrollArea containing " "'#*-content-transcript'). If the page structure differs, the " "selectors in extract_transcript.py need updating." ) title, date = extract_title_and_date(html_parts) speakers = sorted(set(name for _, name, _ in lines)) out_path = src.with_suffix(".md") body = [] body.append(f"# {title or src.stem}") body.append("") if date: body.append(f"- **Date:** {date}") body.append(f"- **Attendees (speakers detected):** {', '.join(speakers)}") body.append(f"- **Source:** `{src.name}`") body.append("") body.append("## Transcript") body.append("") # Blank line between entries (not a single "\n") so each timestamp # starts its own paragraph in Markdown preview - a lone newline is a # soft break that most renderers collapse into one running paragraph. entries = [] for ts, name, text in lines: prefix = f"[{ts}] " if ts else "" entries.append(f"{prefix}{name}: {text}") body.append("\n\n".join(entries)) out_path.write_text("\n".join(body) + "\n", encoding="utf-8") print(f"Wrote {len(lines)} transcript lines to {out_path}") gaps = find_large_gaps(lines) if gaps: print( f"WARNING: {len(gaps)} gap(s) of {GAP_WARNING_SECONDS}s or more " f"between consecutive lines - Fireflies virtualizes the transcript " f"list, so this usually means the page was saved before scrolling " f"through the whole transcript, and content in between is simply " f"missing (not silence). Re-open the meeting, scroll the " f"Transcript tab all the way to the end first, then re-save and " f"re-run this script:", file=sys.stderr, ) for start_idx, end_idx, gap_secs in gaps: start_ts = lines[start_idx][0] end_ts = lines[end_idx][0] print(f" - {start_ts} -> {end_ts} ({gap_secs}s gap)", file=sys.stderr) if __name__ == "__main__": main()