#!/usr/bin/env python3 """ Phase 2 only: Download matching UIDs and convert to PDF. Reads UIDs from .export_progress (or scans fresh if not found). Reconnects for each email to avoid Gmail IMAP drops. """ import os import sys import re import time import logging from pathlib import Path from datetime import datetime, date from imap_tools import MailBox, AND from weasyprint import HTML from fpdf import FPDF import fitz # PyMuPDF # === CONFIG === IMAP_SERVER = "imap.gmail.com" GMAIL_FOLDER = "[Gmail]/Вся почта" MUTT_CONF = Path(__file__).parent / ".muttrc" if not MUTT_CONF.exists(): MUTT_CONF = Path.home() / ".muttrc" DOMAINS = [ "gosuslugi.ru", "dom.gosuslugi.ru", "gov.ru", "duma.gov.ru", "mvd.gov.ru", "mvd.ru", "rkn.gov.ru", "nalog.gov.ru", "nalog.ru", "genproc.gov.ru", "pfr.gov.ru", "fss.ru", "rospotrebnadzor.ru", "russianpost.ru", "mailop.ru", "73.mailop.ru", "uk-service.ru", "ulgss.ru", "ulgov.ru", "ulgkh.ru", "dgi.ru", "mos.ru", "pfrf.ru", "fsin.gov.ru", "skrf.ru", "epp.genproc.gov.ru", "ugpr.ru", "fas.gov.ru", "minstroyrf.gov.ru", "gkh.ru", "sudrf.ru", ] SINCE_DATE = date(2025, 12, 20) BASE_OUTPUT = Path(__file__).parent / "exported_emails" ALL_FOLDER = BASE_OUTPUT / "all" PROGRESS_FILE = Path(__file__).parent / ".export_progress" PHASE2_PROGRESS = Path(__file__).parent / ".phase2_progress" logging.basicConfig( level=logging.INFO, format="%(asctime)s [%(levelname)s] %(message)s", handlers=[ logging.FileHandler(Path(__file__).parent / "export.log", encoding="utf-8"), logging.StreamHandler(sys.stdout), ], ) log = logging.getLogger(__name__) def parse_mutt_config(): config = {} with open(MUTT_CONF) as f: for line in f: line = line.strip() if line.startswith('#') or not line: continue match = re.match(r'set\s+(\w+)\s*=\s*"([^"]*)"', line) if not match: match = re.match(r'set\s+(\w+)\s*=\s*(\S+)', line) if match: config[match.group(1)] = match.group(2) return config def sanitize_filename(name, max_len=80): name = re.sub(r'[<>:"/\\|?*\x00-\x1f]', '_', name) name = re.sub(r'\s+', ' ', name).strip() return name[:max_len] def matches_domain(addr, domains): if not addr: return None addr_lower = addr.lower() for domain in domains: if domain.lower() in addr_lower: return domain return None def html_to_pdf_stream(html_content): try: return HTML(string=html_content).write_pdf() except Exception as e: log.warning(f"weasyprint failed: {e}") return html_to_pdf_fpdf(html_content) def html_to_pdf_fpdf(html_content): from bs4 import BeautifulSoup soup = BeautifulSoup(html_content, 'lxml') text = soup.get_text(separator='\n') pdf = FPDF() pdf.add_page() pdf.add_font('DejaVu', '', '/usr/share/fonts/truetype/dejavu/DejaVuSans.ttf', uni=True) pdf.add_font('DejaVu', 'B', '/usr/share/fonts/truetype/dejavu/DejaVuSans-Bold.ttf', uni=True) pdf.set_font('DejaVu', '', 10) lines = [line for line in text.split('\n') if line.strip()] for line in lines: pdf.multi_cell(0, 6, line) pdf.ln(1) return pdf.output() def convert_attachment_to_pdf(data, filename, content_type): ext = Path(filename).suffix.lower() try: if ext in ('.jpg', '.jpeg', '.png', '.gif', '.bmp', '.tiff', '.webp', '.svg'): pdf = FPDF() pdf.add_page() import io from PIL import Image img = Image.open(io.BytesIO(data)) w_px, h_px = img.size w_mm = w_px * 25.4 / 96 h_mm = h_px * 25.4 / 96 page_w = pdf.w - 20 scale = min(page_w / w_mm, (pdf.h - 20) / h_mm, 1.0) import tempfile with tempfile.NamedTemporaryFile(suffix=ext, delete=False) as tmp: tmp.write(data) tmp_path = tmp.name try: pdf.image(tmp_path, x=10, y=10, w=w_mm*scale, h=h_mm*scale) finally: os.unlink(tmp_path) return pdf.output() elif ext in ('.doc', '.docx', '.xls', '.xlsx'): import tempfile, subprocess with tempfile.TemporaryDirectory() as tmpdir: src = Path(tmpdir) / filename src.write_bytes(data) subprocess.run( ['libreoffice', '--headless', '--convert-to', 'pdf', '--outdir', tmpdir, str(src)], capture_output=True, timeout=120 ) pdf_path = src.with_suffix('.pdf') if pdf_path.exists(): return pdf_path.read_bytes() elif ext == '.pdf': return data elif ext in ('.txt', '.csv', '.log', '.xml', '.html', '.htm'): try: text = data.decode('utf-8', errors='replace') except: text = data.decode('latin-1', errors='replace') return html_to_pdf_fpdf(text) else: info = f"Attachment: {filename}\nType: {content_type}\nSize: {len(data)} bytes\n\n[Binary content - not converted]" return html_to_pdf_fpdf(info) except Exception as e: log.warning(f"Failed to convert {filename}: {e}") info = f"Attachment: {filename}\nType: {content_type}\nSize: {len(data)} bytes\n\nConversion failed: {e}" return html_to_pdf_fpdf(info) def merge_pdfs(pdf_list): if not pdf_list: return None if len(pdf_list) == 1: return pdf_list[0] merger = fitz.open() for pdf_bytes in pdf_list: try: doc = fitz.open(stream=pdf_bytes, filetype="pdf") merger.insert_pdf(doc) except Exception as e: log.warning(f"Failed to merge PDF section: {e}") output = merger.tobytes() merger.close() return output def build_email_html(msg): subject = msg.subject or "(Без темы)" from_addr = msg.from_ or "unknown" to_addr = ", ".join(msg.to) if msg.to else "unknown" cc_addr = ", ".join(msg.cc) if msg.cc else "" date_str = msg.date_str or str(msg.date) if msg.date else "unknown" body_html = msg.html if not body_html and msg.text: text = msg.text text_escaped = text.replace('&', '&').replace('<', '<').replace('>', '>') body_html = text_escaped.replace('\n', '
\n') elif not body_html: body_html = "

(Нет тела письма)

" att_info = "" if msg.attachments: items = [f'
  • {a.filename} ({a.content_type}, {a.size} bytes)
  • ' for a in msg.attachments] att_info = '

    Вложения:

    ' return f"""

    {subject}

    {f'' if cc_addr else ''}
    От:{from_addr}
    Кому:{to_addr}
    Копия:{cc_addr}
    Дата:{date_str}
    {body_html}
    {f'
    {att_info}
    ' if att_info else ''} """ def load_phase2_progress(): """Load set of UIDs already converted to PDF.""" if PHASE2_PROGRESS.exists(): return set(PHASE2_PROGRESS.read_text().strip().splitlines()) return set() def save_phase2_progress(uid, processed_set): processed_set.add(str(uid)) PHASE2_PROGRESS.write_text("\n".join(processed_set)) def main(): log.info("=" * 60) log.info("Phase 2: Download & Convert to PDF") log.info("=" * 60) config = parse_mutt_config() imap_user = config.get('imap_user', '') imap_pass = config.get('imap_pass', '') if not imap_user or imap_user == 'your.email@gmail.com': log.error("No credentials in .muttrc!") sys.exit(1) BASE_OUTPUT.mkdir(parents=True, exist_ok=True) ALL_FOLDER.mkdir(parents=True, exist_ok=True) # Load UIDs from phase1 progress if not PROGRESS_FILE.exists(): log.error("No .export_progress found. Run phase 1 first.") sys.exit(1) all_uids = [line.strip() for line in PROGRESS_FILE.read_text().strip().splitlines() if line.strip()] log.info(f"Total UIDs from phase 1: {len(all_uids)}") # Load phase 2 progress (already converted) phase2_done = load_phase2_progress() log.info(f"Already converted: {len(phase2_done)}") remaining = [uid for uid in all_uids if uid not in phase2_done] log.info(f"Remaining to convert: {len(remaining)}") if not remaining: log.info("All done!") return start_time = time.time() total_processed = 0 total_errors = 0 for idx, uid in enumerate(remaining, 1): elapsed = time.time() - start_time if idx > 1: eta = (elapsed / (idx - 1)) * (len(remaining) - idx + 1) else: eta = 0 log.info(f"[{idx}/{len(remaining)}] UID={uid} | ETA: {eta/60:.1f}m") # Reconnect for each email mailbox = None for attempt in range(3): try: mailbox = MailBox(IMAP_SERVER).login(imap_user, imap_pass, initial_folder=GMAIL_FOLDER) break except Exception as e: log.warning(f"Connect attempt {attempt+1} failed: {e}") time.sleep(5 * (attempt + 1)) mailbox = None if not mailbox: log.error(f"Failed to connect for UID={uid}") total_errors += 1 continue try: msgs = list(mailbox.fetch( AND(uid=uid), headers_only=False, mark_seen=False, )) if not msgs: log.warning(f" Could not fetch UID={uid}") total_errors += 1 save_phase2_progress(uid, phase2_done) continue msg = msgs[0] subject = msg.subject or "(Без темы)" from_addr = msg.from_ or "unknown" match_domain = matches_domain(from_addr, DOMAINS) if not match_domain: for to_addr in (msg.to or []): match_domain = matches_domain(to_addr, DOMAINS) if match_domain: break if not match_domain: match_domain = "other" safe_subject = sanitize_filename(subject) safe_from = sanitize_filename(from_addr.split('<')[-1].rstrip('>').split('@')[0], 30) date_part = msg.date.strftime('%Y-%m-%d') if msg.date else 'unknown' filename = f"{date_part}_{safe_from}_{safe_subject}.pdf" if len(filename) > 150: filename = filename[:150] + ".pdf" html_content = build_email_html(msg) pdf_parts = [] body_pdf = html_to_pdf_stream(html_content) if body_pdf: pdf_parts.append(body_pdf) if msg.attachments: log.info(f" {len(msg.attachments)} attachment(s)") for att in msg.attachments: try: att_pdf = convert_attachment_to_pdf( att.payload, att.filename, att.content_type ) if att_pdf: pdf_parts.append(att_pdf) except Exception as e: log.warning(f" Attachment convert failed {att.filename}: {e}") if pdf_parts: merged_pdf = merge_pdfs(pdf_parts) if merged_pdf: domain_folder = BASE_OUTPUT / sanitize_filename(match_domain, 50) domain_folder.mkdir(parents=True, exist_ok=True) (domain_folder / filename).write_bytes(merged_pdf) (ALL_FOLDER / f"{match_domain}_{filename}").write_bytes(merged_pdf) log.info(f" SAVED: {filename}") total_processed += 1 else: log.warning(f" Failed to merge PDF") total_errors += 1 else: log.warning(f" No content") total_errors += 1 save_phase2_progress(uid, phase2_done) except Exception as e: log.error(f" ERROR UID={uid}: {e}") total_errors += 1 finally: if mailbox: try: mailbox.logout() except: pass # Small delay to avoid Gmail rate limiting time.sleep(0.5) total_time = time.time() - start_time log.info("\n" + "=" * 60) log.info("PHASE 2 COMPLETE!") log.info(f"Processed: {total_processed}") log.info(f"Errors: {total_errors}") log.info(f"Time: {total_time/60:.1f} min") log.info(f"Total PDFs: {len(list(ALL_FOLDER.glob('*.pdf')))}") log.info("=" * 60) if __name__ == '__main__': main()