#!/usr/bin/env python3 """ EPUB to Obsidian Converter - Reusable Tool Pipeline: 1. Extract metadata from original EPUB (preserves correct UTF-8 encoding) 2. Calibre normalizes EPUB (handles OEBPS/anchor format, splits monolithic HTML) 3. Extract TOC from original EPUB 4. Parse HTML content from normalized EPUB 5. Split into per-chapter Obsidian notes with frontmatter Usage: python -m tools.epub_converter [--author "Author"] [--tags "tag1,tag2"] Or as module: from tools.epub_converter import EpubConverter converter = EpubConverter() converter.convert("input.epub", "output_dir") """ import argparse import io import os import re import shutil import subprocess import sys import tempfile import zipfile from xml.dom import minidom sys.stdout = io.TextIOWrapper(sys.stdout.buffer, encoding="utf-8", errors="replace") sys.stderr = io.TextIOWrapper(sys.stderr.buffer, encoding="utf-8", errors="replace") SECTION_KEYWORDS = ["编", "附录", "结束语", "修订版说明"] FRONT_MATTER_LABELS = {"致读者", "前言"} class EpubConverter: def __init__(self): self.metadata = {} self.toc = [] self.html_files = {} def convert( self, input_epub: str, output_dir: str, author: str = None, tags: list = None, keep_temp: bool = False, ) -> list[str]: input_epub = os.path.abspath(input_epub) output_dir = os.path.abspath(output_dir) if not os.path.exists(input_epub): raise FileNotFoundError(f"Input file not found: {input_epub}") os.makedirs(output_dir, exist_ok=True) with tempfile.TemporaryDirectory() as tmpdir: normalized_epub = self._normalize_epub(input_epub, tmpdir) self._extract_metadata(input_epub) self._extract_toc(input_epub) self._extract_html_content(normalized_epub) if author: self.metadata["creator"] = author files = self._split_and_write(output_dir, tags or []) return files def _run_calibre(self, input_epub: str, output_epub: str) -> bool: try: subprocess.run( ["ebook-convert", input_epub, output_epub], capture_output=True, timeout=120, encoding="utf-8", errors="replace", ) return os.path.exists(output_epub) except Exception: return False def _normalize_epub(self, input_epub: str, tmpdir: str) -> str: normalized_epub = os.path.join(tmpdir, "normalized.epub") if not self._run_calibre(input_epub, normalized_epub): return input_epub return normalized_epub def _extract_metadata(self, epub_path: str): z = zipfile.ZipFile(epub_path, "r") opf_files = [f for f in z.namelist() if f.endswith(".opf")] if not opf_files: z.close() return opf_raw = z.read(opf_files[0]) z.close() opf_text = opf_raw.decode("utf-8", errors="replace") doc = minidom.parseString(opf_text) for tag in ["title", "creator", "language", "publisher"]: elements = doc.getElementsByTagName(f"dc:{tag}") if elements and elements[0].firstChild: self.metadata[tag] = elements[0].firstChild.nodeValue.strip() date_els = doc.getElementsByTagName("dc:date") if date_els and date_els[0].firstChild: self.metadata["date"] = date_els[0].firstChild.nodeValue.strip()[:10] def _extract_toc(self, epub_path: str): z = zipfile.ZipFile(epub_path, "r") ncx_files = [f for f in z.namelist() if f.endswith(".ncx")] if not ncx_files: z.close() return toc_xml = z.read(ncx_files[0]).decode("utf-8", errors="replace") z.close() doc = minidom.parseString(toc_xml) nav_points = doc.getElementsByTagName("navPoint") for np in nav_points: text_el = np.getElementsByTagName("text") content_el = np.getElementsByTagName("content") if not text_el or not content_el: continue label = ( text_el[0].firstChild.nodeValue.strip() if text_el[0].firstChild else "" ) src = content_el[0].getAttribute("src") is_section = ( any(kw in label for kw in SECTION_KEYWORDS) or label in FRONT_MATTER_LABELS or label.endswith("(代序)") ) self.toc.append({"label": label, "src": src, "is_section": is_section}) def _extract_html_content(self, epub_path: str): z = zipfile.ZipFile(epub_path, "r") html_files = sorted( f for f in z.namelist() if f.endswith((".html", ".xhtml")) and "text" in f.lower() ) for hf in html_files: raw = z.read(hf) try: self.html_files[hf] = raw.decode("utf-8") except UnicodeDecodeError: self.html_files[hf] = raw.decode("gb18030", errors="replace") z.close() def _split_and_write(self, output_dir: str, tags: list) -> list[str]: created_files = [] chapter_map = self._build_chapter_map() if not chapter_map: return [] grouped = self._group_by_html_file(chapter_map) all_chapters = [] for html_file, chapters in grouped.items(): html = self.html_files.get(html_file, "") if not html: continue anchors = [c["anchor"] for c in chapters if c["anchor"]] if anchors: chunks = self._split_by_anchors(html, anchors) else: chunks = [("", html)] for chapter, (anchor_id, chunk) in zip(chapters, chunks): md = self._html_to_markdown(chunk) if chunk else "" md = self._clean_markdown(md) if md.strip(): all_chapters.append( { "label": chapter["label"], "is_section": chapter["is_section"], "content": md, } ) current_section = "" for ch in all_chapters: if ch["is_section"]: current_section = ch["label"] filename = self._sanitize_filename(ch["label"]) + ".md" filepath = os.path.join(output_dir, filename) counter = 1 base = filepath while os.path.exists(filepath): name, ext = os.path.splitext(base) filepath = f"{name}_{counter}{ext}" counter += 1 fm = self._generate_frontmatter( ch["label"], current_section, ch["is_section"], tags ) with open(filepath, "w", encoding="utf-8") as f: f.write(fm) f.write(ch["content"]) created_files.append(filepath) return created_files def _build_chapter_map(self): chapter_map = [] for i, entry in enumerate(self.toc): src = entry["src"] anchor = "" if "#" in src: path, anchor = src.split("#", 1) else: path = src matching_file = None for hf in self.html_files: if path in hf or hf.endswith(path): matching_file = hf break if matching_file: chapter_map.append( { "label": entry["label"], "is_section": entry["is_section"], "html_file": matching_file, "anchor": anchor, "toc_index": i, } ) return chapter_map def _group_by_html_file(self, chapter_map): grouped = {} for ch in chapter_map: hf = ch["html_file"] if hf not in grouped: grouped[hf] = [] grouped[hf].append(ch) return grouped def _split_by_anchors(self, html: str, anchors: list[str]): chunks = [] for i, anchor in enumerate(anchors): pattern = re.compile( rf'<(?:a|span|div|p|h[1-6])[^>]*(?:id|name)=["\']?{re.escape(anchor)}["\']?[^>]*>', re.IGNORECASE, ) match = pattern.search(html) if match: start = match.start() if i + 1 < len(anchors): next_pattern = re.compile( rf'<(?:a|span|div|p|h[1-6])[^>]*(?:id|name)=["\']?{re.escape(anchors[i + 1])}["\']?[^>]*>', re.IGNORECASE, ) next_match = next_pattern.search(html, start + 1) end = next_match.start() if next_match else len(html) else: end = len(html) chunks.append((anchor, html[start:end])) else: chunks.append((anchor, "")) return chunks def _html_to_markdown(self, html_content: str) -> str: try: import html2text h = html2text.HTML2Text() h.body_width = 0 h.unicode_snob = True h.protect_links = True h.wrap_links = False return h.handle(html_content) except ImportError: try: from bs4 import BeautifulSoup soup = BeautifulSoup(html_content, "html.parser") return soup.get_text("\n") except ImportError: return re.sub(r"<[^>]+>", "", html_content) def _clean_markdown(self, md: str) -> str: md = re.sub(r"\[([^\]]*)\]\([^\)]*\.html[^\)]*\)", r"\1", md) md = re.sub(r"\n{4,}", "\n\n\n", md) return md.strip() def _sanitize_filename(self, name: str) -> str: name = re.sub(r'[<>:"/\\|?*]', "", name) name = re.sub(r"\s+", " ", name).strip() if len(name) > 80: name = name[:80] return name def _generate_frontmatter( self, title: str, section: str, is_section: bool, tags: list ) -> str: book_title = self.metadata.get("title", "Unknown") author = self.metadata.get("creator", "") date = self.metadata.get("date", "") tag_list = ["book", "epub"] + tags if is_section: tag_list.append("section") tags_str = "\n - ".join(tag_list) return f"""--- title: "{title}" categories: - "[[LLM Wiki]]" - "[[Books]]" tags: - {tags_str} book: "[[{book_title}]]" author: "[[{author}]]" section: "{section}" created: "{date}" type: book-chapter --- """ def main(): parser = argparse.ArgumentParser(description="Convert EPUB to Obsidian notes") parser.add_argument("input", help="Input EPUB file path") parser.add_argument("output_dir", help="Output directory for chapter files") parser.add_argument("--author", help="Override author name") parser.add_argument("--tags", help="Comma-separated tags") args = parser.parse_args() converter = EpubConverter() files = converter.convert( args.input, args.output_dir, author=args.author, tags=args.tags.split(",") if args.tags else [], ) print(f"Done! Created {len(files)} files in {args.output_dir}") if __name__ == "__main__": main()