Last active
August 28, 2026 09:07
-
-
Save scateu/6daefaabcfb9ed4066b6e1455ed83800 to your computer and use it in GitHub Desktop.
imap2markdown: markdown, txt, org file bi-directional sync to IMAP
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env python3 | |
| import os | |
| import sys | |
| import argparse | |
| import imaplib | |
| import email | |
| import re | |
| import html | |
| from email.header import decode_header | |
| from email.message import EmailMessage | |
| from datetime import datetime, timezone | |
| import getpass | |
| # 允许同步的文件配置:支持的后缀或精准完整文件名 | |
| ALLOWED_EXTENSIONS = ('.md', '.txt', '.org') | |
| ALLOWED_EXACT_NAMES = ('Makefile', '.imapsync.py') | |
| def is_allowed_file(filename): | |
| """检查文件名是否在允许列表中(匹配后缀或精准文件名)""" | |
| if filename in ALLOWED_EXACT_NAMES: | |
| return True | |
| ext = os.path.splitext(filename)[1].lower() | |
| if ext in ALLOWED_EXTENSIONS: | |
| return True | |
| return False | |
| def parse_args(): | |
| parser = argparse.ArgumentParser(description="Sync local files with an IMAP folder using Subject as filename.") | |
| parser.add_argument("--host", required=True, help="IMAP server host") | |
| parser.add_argument("--port", type=int, default=993, help="IMAP server port (default: 993)") | |
| parser.add_argument("--user", required=True, help="IMAP username") | |
| parser.add_argument("--folder", default="Notes", help="Remote IMAP folder/mailbox (default: 'Notes')") | |
| # Password arguments | |
| parser.add_argument("--password", help="Plaintext password (unsafe, use with caution)") | |
| parser.add_argument("--ask-pass", action="store_true", help="Prompt securely for password") | |
| # Execution modes | |
| parser.add_argument("--dry-run", action="store_true", help="Show what would happen without making changes") | |
| # Sync strategy overrides | |
| group = parser.add_mutually_exclusive_group() | |
| group.add_argument("--pull", action="store_true", help="Force pull from remote (overwrite local changes)") | |
| group.add_argument("--push", action="store_true", help="Force push from local (overwrite remote changes)") | |
| parser.add_argument("local_dir", help="Path to local folder containing note files") | |
| return parser.parse_args() | |
| def get_local_notes(local_dir): | |
| """Reads local supported files, returns metadata and content.""" | |
| notes = {} | |
| if not os.path.exists(local_dir): | |
| os.makedirs(local_dir, exist_ok=True) | |
| return notes | |
| for filename in os.listdir(local_dir): | |
| # 使用全新的允许列表机制进行过滤 | |
| if is_allowed_file(filename): | |
| filepath = os.path.join(local_dir, filename) | |
| stat = os.stat(filepath) | |
| mtime = datetime.fromtimestamp(stat.st_mtime, tz=timezone.utc) | |
| # 使用 utf-8-sig 自动剥离 Vim 的 BOM 头 <feff> | |
| with open(filepath, "r", encoding="utf-8-sig") as f: | |
| content = f.read() | |
| # 现在的策略:Subject 直接等于完整的 filename | |
| subject = filename | |
| ext = os.path.splitext(filename)[1].lower() | |
| notes[filename] = { | |
| "filename": filename, | |
| "filepath": filepath, | |
| "mtime": mtime, | |
| "subject": subject, | |
| "ext": ext, | |
| "content": content | |
| } | |
| return notes | |
| def parse_imap_date(date_str): | |
| """Parses email Date header into a timezone-aware datetime object.""" | |
| try: | |
| return email.utils.parsedate_to_datetime(date_str).astimezone(timezone.utc) | |
| except Exception: | |
| return datetime.min.replace(tzinfo=timezone.utc) | |
| def get_remote_notes(mail, folder): | |
| """Fetches notes existing in the IMAP folder. If duplicates exist, keeps only the newest.""" | |
| status, _ = mail.select(folder) | |
| if status != 'OK': | |
| mail.create(folder) | |
| mail.select(folder) | |
| status, messages = mail.search(None, 'ALL') | |
| remote_notes = {} | |
| if status != 'OK' or not messages or not messages[0]: | |
| return remote_notes | |
| for msg_id in messages[0].split(): | |
| status, data = mail.fetch(msg_id, '(BODY[HEADER])') | |
| if status != 'OK': | |
| continue | |
| header_bytes = None | |
| for response_part in data: | |
| if isinstance(response_part, tuple): | |
| header_bytes = response_part[1] | |
| break | |
| if not header_bytes: | |
| continue | |
| msg = email.message_from_bytes(header_bytes) | |
| # Decode Subject | |
| subject_header = msg.get('Subject', '') | |
| decoded_fragments = decode_header(subject_header) | |
| subject_parts = [] | |
| for text, encoding in decoded_fragments: | |
| if isinstance(text, bytes): | |
| subject_parts.append(text.decode(encoding or 'utf-8', errors='replace')) | |
| else: | |
| subject_parts.append(text) | |
| subject = "".join(subject_parts).strip() | |
| if not subject: | |
| continue | |
| filename = subject | |
| if not is_allowed_file(filename): | |
| continue | |
| ext = os.path.splitext(filename)[1].lower() | |
| mail_date_str = msg.get('Date') | |
| mtime = parse_imap_date(mail_date_str) if mail_date_str else datetime.min.replace(tzinfo=timezone.utc) | |
| # 🌟 强化版的头部提取器:完全保留 Apple 赖以生存的所有身份追踪标记 | |
| apple_headers = {} | |
| for k, v in msg.items(): | |
| k_lower = k.lower() | |
| if (k_lower == 'x-uniform-type-identifier' or | |
| k_lower == 'x-universally-unique-identifier' or | |
| k_lower == 'message-id' or | |
| k_lower == 'from' or | |
| k_lower == 'mime-version' or | |
| k_lower.startswith('x-apple-')): | |
| apple_headers[k] = v | |
| if filename in remote_notes: | |
| remote_notes[filename]['all_msg_ids'].append(msg_id) | |
| if mtime > remote_notes[filename]['mtime']: | |
| remote_notes[filename]['msg_id'] = msg_id | |
| remote_notes[filename]['mtime'] = mtime | |
| remote_notes[filename]['apple_headers'] = apple_headers | |
| else: | |
| remote_notes[filename] = { | |
| "msg_id": msg_id, | |
| "all_msg_ids": [msg_id], | |
| "subject": subject, | |
| "mtime": mtime, | |
| "ext": ext, | |
| "filename": filename, | |
| "apple_headers": apple_headers | |
| } | |
| return remote_notes | |
| def strip_html_tags(html_content): | |
| """Strips HTML tags and restores basic entities to plaintext, preserving newlines.""" | |
| # 1. 移除 <style> 和 <script> 块及其内容,防止样式代码混入正文 | |
| clean_text = re.sub(r'<(style|script)[^>]*?>.*?</\1>', '', html_content, flags=re.DOTALL | re.IGNORECASE) | |
| # 2. 将闭合的段落标签 </p>, 块标签 </div> 以及换行标签 <br> 统一替换为换行符 \n | |
| # 可以使用 [ \t]*\n 避免产生过多的连续空行,这里保留标准换行语义 | |
| clean_text = re.sub(r'</?(p|div|br)[^>]*?>', '\n', clean_text, flags=re.IGNORECASE) | |
| # 3. 移除其他普通的 HTML 标签(如 <span>, <strong> 等不影响换行的标签) | |
| clean_text = re.sub(r'<[^>]+>', '', clean_text) | |
| # 4. 转换常见的 HTML 实体字符(如 -> 空格, < -> <) | |
| clean_text = html.unescape(clean_text) | |
| # 5. 优化多余的连续空行:将 3 个及以上的连续换行压缩为 2 个(保留一个空行隔开段落),并去掉首尾空白 | |
| clean_text = re.sub(r'\n{3,}', '\n\n', clean_text) | |
| return clean_text.strip() | |
| def fetch_email_body(mail, msg_id): | |
| """Fetches the full body and properly decodes Quoted-Printable, Text, and HTML fallbacks.""" | |
| status, data = mail.fetch(msg_id, '(RFC822)') | |
| if status != 'OK': | |
| return "" | |
| raw_email = None | |
| for response_part in data: | |
| if isinstance(response_part, tuple): | |
| raw_email = response_part[1] # 确保安全解包 | |
| break | |
| if not raw_email: | |
| return "" | |
| msg = email.message_from_bytes(raw_email) | |
| plain_bytes = b"" | |
| html_bytes = b"" | |
| plain_charset = "utf-8" | |
| html_charset = "utf-8" | |
| if msg.is_multipart(): | |
| for part in msg.walk(): | |
| content_type = part.get_content_type() | |
| if content_type == "text/plain": | |
| plain_bytes = part.get_payload(decode=True) | |
| plain_charset = part.get_content_charset() or "utf-8" | |
| elif content_type == "text/html": | |
| html_bytes = part.get_payload(decode=True) | |
| html_charset = part.get_content_charset() or "utf-8" | |
| else: | |
| plain_bytes = msg.get_payload(decode=True) | |
| plain_charset = msg.get_content_charset() or "utf-8" | |
| if plain_bytes and plain_bytes.strip(): | |
| try: | |
| content = plain_bytes.decode(plain_charset, errors='replace') | |
| if content.startswith('\ufeff'): | |
| content = content[1:] | |
| return content | |
| except Exception: | |
| return plain_bytes.decode('utf-8', errors='replace') | |
| elif html_bytes and html_bytes.strip(): | |
| try: | |
| html_content = html_bytes.decode(html_charset, errors='replace') | |
| content = strip_html_tags(html_content) | |
| if content.startswith('\ufeff'): | |
| content = content[1:] | |
| return content | |
| except Exception: | |
| return strip_html_tags(html_bytes.decode('utf-8', errors='replace')) | |
| return "" | |
| def upload_note(mail, folder, filename, note_data, dry_run=False): | |
| """Pushes a local note up to the IMAP server, preserving original Apple client headers if present.""" | |
| print(f"[PUSH] Uploading '{filename}' -> Remote Subject: '{note_data['subject']}'") | |
| if dry_run: | |
| return | |
| msg = EmailMessage() | |
| msg['Subject'] = note_data['subject'] | |
| msg['Date'] = email.utils.format_datetime(note_data['mtime']) | |
| # 🌟 核心修复:按原样透传包括 UUID、From、Message-Id 在内的核心头部 | |
| apple_headers = note_data.get('apple_headers', {}) | |
| for k, v in apple_headers.items(): | |
| # 如果原始邮件包含了 Mime-Version(例如带 Mac OS X Mail 16.0 备注), | |
| # 我们可以直接覆盖 EmailMessage 的默认 Version 头 | |
| if k in msg: | |
| del msg[k] | |
| msg[k] = v | |
| msg.set_content(note_data['content'], charset='utf-8') | |
| mail.append(folder, None, imaplib.Time2Internaldate(note_data['mtime']), msg.as_bytes()) | |
| def delete_remote_note_batch(mail, msg_id_list, dry_run=False): | |
| """Marks multiple remote email note as Deleted.""" | |
| msg_id_string = ','.join(msg_id_list) | |
| if dry_run: | |
| return | |
| mail.store(msg_id_string, '+FLAGS', '\\Deleted') | |
| def download_note(mail, remote_note, local_dir, dry_run=False): | |
| """Pulls a remote note down to a local file.""" | |
| filename = remote_note['filename'] | |
| filepath = os.path.join(local_dir, filename) | |
| print(f"[PULL] Downloading Remote Note -> '{filename}'") | |
| if dry_run: | |
| return | |
| content = fetch_email_body(mail, remote_note['msg_id']) | |
| with open(filepath, "w", encoding="utf-8") as f: | |
| f.write(content) | |
| epoch_time = remote_note['mtime'].timestamp() | |
| os.utime(filepath, (epoch_time, epoch_time)) | |
| def main(): | |
| args = parse_args() | |
| password = None | |
| if args.password: | |
| password = args.password | |
| elif args.ask_pass: | |
| password = getpass.getpass("Enter IMAP Password: ") | |
| if not password: | |
| print("Error: Password must be provided via --password or --ask-pass", file=sys.stderr) | |
| sys.exit(1) | |
| print(f"Connecting to {args.host}:{args.port}...") | |
| try: | |
| mail = imaplib.IMAP4_SSL(args.host, args.port) | |
| mail.login(args.user, password) | |
| except Exception as e: | |
| print(f"Authentication/Connection failed: {e}", file=sys.stderr) | |
| sys.exit(1) | |
| try: | |
| print("Scanning local directory...") | |
| local_notes = get_local_notes(args.local_dir) | |
| print(f"Scanning remote IMAP folder '{args.folder}'...") | |
| # 修改:get_remote_notes 内部已调整,遇到同名邮件只会返回最新的一封 | |
| remote_notes = get_remote_notes(mail, args.folder) | |
| all_filenames = set(local_notes.keys()).union(set(remote_notes.keys())) | |
| pending_delete_note_msg_id_list = [] | |
| pending_upload_notes = [] | |
| for filename in all_filenames: | |
| local = local_notes.get(filename) | |
| remote = remote_notes.get(filename) | |
| if local and not remote: | |
| if args.pull: | |
| if not args.dry_run: | |
| print(f"[PULL-DROP] Removing local file '{filename}' (not on remote)") | |
| os.remove(local['filepath']) | |
| else: | |
| print(f"[DRY-RUN] Would remove local file '{filename}'") | |
| else: | |
| upload_note(mail, args.folder, filename, local, args.dry_run) | |
| elif remote and not local: | |
| if args.push: | |
| print(f"[PUSH-DROP] Marking remote note '{filename}' for deletion (missing locally)") | |
| # 支持将该文件关联的所有历史旧草稿(包括重复产生的旧件)一并加入删除列表 | |
| for old_id in remote['all_msg_ids']: | |
| pending_delete_note_msg_id_list.append(old_id.decode('utf-8')) | |
| else: | |
| download_note(mail, remote, args.local_dir, args.dry_run) | |
| elif local and remote: | |
| time_diff = (local['mtime'] - remote['mtime']).total_seconds() | |
| if args.pull or (time_diff < -2 and not args.push): | |
| download_note(mail, remote, args.local_dir, args.dry_run) | |
| elif args.push or (time_diff > 2 and not args.pull): | |
| print(f"[UPDATE] Local version of '{filename}' is newer or forced.") | |
| # 将远程现存的最新副本及可能存在的重复旧草稿,全部标记删除,随后推入新的 | |
| for old_id in remote['all_msg_ids']: | |
| pending_delete_note_msg_id_list.append(old_id.decode('utf-8')) | |
| # 提取并透传可能从远程获取到的 Apple 专属头部信息,避免破坏 Apple Mail 的草稿追踪链 | |
| apple_headers = remote.get('apple_headers', {}) | |
| local_with_headers = local.copy() | |
| local_with_headers['apple_headers'] = apple_headers | |
| pending_upload_notes.append((args.folder, filename, local_with_headers)) | |
| else: | |
| continue | |
| # 执行批量删除旧的/重复的远程邮件 | |
| if pending_delete_note_msg_id_list: | |
| delete_remote_note_batch(mail, pending_delete_note_msg_id_list, args.dry_run) | |
| if not args.dry_run: | |
| mail.expunge() | |
| pending_delete_note_msg_id_list = [] | |
| # 执行批量上传更新 | |
| if pending_upload_notes: | |
| for item in pending_upload_notes: | |
| # 正确解包三元组:item[0] 是 folder, item[1] 是 filename, item[2] 是 note_data | |
| upload_note(mail, item[0], item[1], item[2], args.dry_run) | |
| pending_upload_notes = [] | |
| print("Sync complete.") | |
| finally: | |
| try: | |
| mail.logout() | |
| except: | |
| pass | |
| if __name__ == "__main__": | |
| main() |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| IMAP_REMOTE_FOLDER := Drafts | |
| IMAP_HOST := imap.example.com | |
| IMAP_USER := rss | |
| IMAP_LOCAL_FOLDER := . | |
| edit: | |
| make sync | |
| ranger . | |
| make sync | |
| sync: | |
| python3 .imapsync.py --host ${IMAP_HOST} --port 993 --user ${IMAP_USER} --folder ${IMAP_REMOTE_FOLDER} --password `cat ~/.imappass` ${IMAP_LOCAL_FOLDER} | |
| -git status | |
| -git add --all | |
| push: | |
| python3 .imapsync.py --host ${IMAP_HOST} --port 993 --user ${IMAP_USER} --folder ${IMAP_REMOTE_FOLDER} --password `cat ~/.imappass` --push ${IMAP_LOCAL_FOLDER} | |
| -git status | |
| -git add --all | |
| pull: | |
| python3 .imapsync.py --host ${IMAP_HOST} --port 993 --user ${IMAP_USER} --folder ${IMAP_REMOTE_FOLDER} --password `cat ~/.imappass` --pull ${IMAP_LOCAL_FOLDER} | |
| -git status | |
| -git add --all |
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| #!/usr/bin/env python3 | |
| import os | |
| import glob | |
| import argparse | |
| from datetime import datetime | |
| def parse_arguments(): | |
| parser = argparse.ArgumentParser( | |
| description="Export Vimwiki plain text files (wiki/markdown) to an mbox email archive.", | |
| formatter_class=argparse.ArgumentDefaultsHelpFormatter | |
| ) | |
| parser.add_argument( | |
| "-i", "--input", | |
| default=os.path.expanduser("~/vimwiki"), | |
| help="Path to your Vimwiki directory" | |
| ) | |
| parser.add_argument( | |
| "-o", "--output", | |
| default=os.path.expanduser("~/vimwiki.mbox"), | |
| help="Path to the output .mbox file" | |
| ) | |
| parser.add_argument( | |
| "-m", "--markdown", | |
| action="store_true", | |
| help="Scan for Markdown files (.md) instead of default Vimwiki files (.wiki)" | |
| ) | |
| return parser.parse_args() | |
| def main(): | |
| args = parse_arguments() | |
| # Set file extension based on flag | |
| ext = "*.md" if args.markdown else "*.wiki" | |
| search_path = os.path.join(args.input, ext) | |
| # Find all files matching the extension | |
| files = sorted(glob.glob(search_path)) | |
| if not files: | |
| print(f"No {ext} files found in {args.input}") | |
| return | |
| print(f"Processing {len(files)} files...") | |
| with open(args.output, "w", encoding="utf-8") as mbox: | |
| for filepath in files: | |
| filename = os.path.basename(filepath) | |
| title, _ = os.path.splitext(filename) | |
| # Use file modification time for email Date header | |
| mtime = os.path.getmtime(filepath) | |
| date_str = datetime.fromtimestamp(mtime).strftime("%a %b %d %H:%M:%S %Y") | |
| with open(filepath, "r", encoding="utf-8") as f: | |
| content = f.read() | |
| # Write standard mbox envelope and basic headers | |
| mbox.write(f"From - {date_str}\n") | |
| mbox.write(f"Subject: {title}\n") | |
| mbox.write(f"Date: {date_str}\n") | |
| mbox.write("Content-Type: text/plain; charset=UTF-8\n\n") | |
| mbox.write(content) | |
| mbox.write("\n\n") # Standard spacing between mbox entries | |
| print(f"Successfully exported to {args.output}") | |
| if __name__ == "__main__": | |
| main() | |
Sign up for free
to join this conversation on GitHub.
Already have an account?
Sign in to comment