From bccd7ad1737dba72814a5768a241bbceadb129a6 Mon Sep 17 00:00:00 2001 From: Kritoooo Date: Sun, 12 Apr 2026 16:08:46 +0800 Subject: [PATCH] feat: add EPUB support for content processing and upload --- .github/workflows/convert.yml | 5 +- .github/workflows/test.yml | 2 +- README.md | 16 +- README.zh-CN.md | 16 +- cli/commands/reconvert.js | 2 +- cli/github.js | 27 ++- cli/index.js | 4 +- cli/mcp-server.mjs | 12 +- scripts/generate_structure.py | 13 +- scripts/process.py | 362 +++++++++++++++++++++++++++++++++- src/components/AdminView.jsx | 15 +- src/components/HomeView.jsx | 4 +- src/lib/github-api.js | 31 ++- tests/cli/commands.test.js | 2 +- tests/scripts/test_process.py | 146 ++++++++++++++ 15 files changed, 595 insertions(+), 62 deletions(-) create mode 100644 tests/scripts/test_process.py diff --git a/.github/workflows/convert.yml b/.github/workflows/convert.yml index 85ff185..a400155 100644 --- a/.github/workflows/convert.yml +++ b/.github/workflows/convert.yml @@ -7,7 +7,7 @@ on: workflow_dispatch: inputs: filename: - description: "Specific filename in input/ (e.g. book.pdf, notes.md, site.zip)" + description: "Specific filename in input/ (e.g. book.pdf, book.epub, notes.md, site.zip)" required: false type: string @@ -38,6 +38,9 @@ jobs: - name: Install dependencies run: pip install -r requirements.txt + - name: Install pandoc + run: sudo apt-get update && sudo apt-get install -y pandoc + - name: Process content run: python scripts/process.py env: diff --git a/.github/workflows/test.yml b/.github/workflows/test.yml index 3d85058..c040893 100644 --- a/.github/workflows/test.yml +++ b/.github/workflows/test.yml @@ -44,7 +44,7 @@ jobs: run: pip install -r requirements.txt - name: Validate Python scripts - run: python3 -m py_compile scripts/build_manifest.py scripts/convert.py + run: python3 -m py_compile scripts/build_manifest.py scripts/convert.py scripts/process.py - name: Run Python tests run: python3 -m unittest discover -s tests/scripts -v diff --git a/README.md b/README.md index 517a867..b2ee590 100644 --- a/README.md +++ b/README.md @@ -4,7 +4,7 @@ GitHub-hosted content shelf. Fork, upload, done. -> Fork this repo to get your own content platform on GitHub Pages. Upload PDFs (auto-converted to books), Markdown documents (rendered directly), or ZIP archives (deployed as static sites). Zero server cost. +> Fork this repo to get your own content platform on GitHub Pages. Upload PDFs or EPUBs (auto-converted to books), Markdown documents (rendered directly), or ZIP archives (deployed as static sites). Zero server cost. ## Quick Start @@ -24,7 +24,7 @@ Your site is now live at `https://.github.io/gitshelf/` 3. In your fork, go to **Settings > Secrets and variables > Actions** 4. Click **New repository secret**, name it `MINERU_TOKEN`, paste the token -> Only needed if you want to upload PDFs. Markdown and ZIP uploads work without this. +> Only needed if you want to upload PDFs. EPUB, Markdown, and ZIP uploads work without this. ### 3. Password Protection (Optional) @@ -41,6 +41,7 @@ Your site is now live at `https://.github.io/gitshelf/` ([Create one here](https://github.com/settings/tokens/new?scopes=repo&description=GitShelf)) 3. Upload a file: - **`.pdf`** — Converted to a multi-chapter book via MinerU API + - **`.epub`** — Converted to a multi-chapter book via pandoc - **`.md`** — Rendered directly as a document - **`.zip`** — Extracted as a static site (must contain `index.html`) 4. Wait for GitHub Actions to process (progress shown in Actions tab) @@ -50,14 +51,14 @@ Your site is now live at `https://.github.io/gitshelf/` | Type | Upload | Display | |------|--------|---------| -| **Book** | `.pdf` file | Chapter reader with TOC sidebar, keyboard navigation | +| **Book** | `.pdf` or `.epub` file | Chapter reader with TOC sidebar, keyboard navigation | | **Document** | `.md` file | Single-page Markdown rendering with syntax highlighting | | **Site** | `.zip` file | Static site served directly (clicks open in new tab) | ## Features - **Reader** — Dark/light theme, chapter sidebar, keyboard navigation, code highlighting (Shiki), math rendering (KaTeX), responsive layout -- **Admin** — Upload PDFs/Markdown/ZIPs from browser, catalog management (edit, publish, hide, archive, delete), search & filter +- **Admin** — Upload PDFs/EPUBs/Markdown/ZIPs from browser, catalog management (edit, publish, hide, archive, delete), search & filter - **Pipeline** — GitHub Actions processes uploads automatically, handles large PDFs by auto-chunking - **Homepage** — Tab-based filtering: All / Books / Documents / Sites @@ -66,9 +67,10 @@ Your site is now live at `https://.github.io/gitshelf/` ``` Upload content (browser → GitHub API → input/) → GitHub Actions runs scripts/process.py - → .pdf: MinerU API → Markdown → Split chapters → docs/books/{id}/ - → .md: Copy to docs/articles/{id}/content.md - → .zip: Extract to docs/sites/{id}/ + → .pdf: MinerU API → Markdown → Split chapters → docs/books/{id}/ + → .epub: pandoc → Markdown + media → Split chapters → docs/books/{id}/ + → .md: Copy to docs/articles/{id}/content.md + → .zip: Extract to docs/sites/{id}/ → Build manifest → GitHub Pages deploys ``` diff --git a/README.zh-CN.md b/README.zh-CN.md index 3ac30f7..0280420 100644 --- a/README.zh-CN.md +++ b/README.zh-CN.md @@ -4,7 +4,7 @@ 基于 GitHub 的内容托管平台。Fork,上传,搞定。 -> Fork 本仓库即可拥有自己的内容平台。上传 PDF(自动转换为书籍)、Markdown 文档(直接渲染)、ZIP 压缩包(部署为静态站点),全部托管在 GitHub Pages 上。零服务器成本。 +> Fork 本仓库即可拥有自己的内容平台。上传 PDF 或 EPUB(自动转换为书籍)、Markdown 文档(直接渲染)、ZIP 压缩包(部署为静态站点),全部托管在 GitHub Pages 上。零服务器成本。 ## 快速开始 @@ -24,7 +24,7 @@ 3. 在你的 Fork 中,进入 **Settings > Secrets and variables > Actions** 4. 点击 **New repository secret**,名称填 `MINERU_TOKEN`,粘贴 Token -> 仅上传 PDF 时需要。Markdown 和 ZIP 上传无需此配置。 +> 仅上传 PDF 时需要。EPUB、Markdown 和 ZIP 上传无需此配置。 ### 3. 密码保护(可选) @@ -41,6 +41,7 @@ ([点此创建](https://github.com/settings/tokens/new?scopes=repo&description=GitShelf)) 3. 上传文件: - **`.pdf`** — 通过 MinerU API 转换为多章节书籍 + - **`.epub`** — 通过 pandoc 转换为多章节书籍 - **`.md`** — 直接作为文档渲染展示 - **`.zip`** — 解压为静态站点(需包含 `index.html`) 4. 等待 GitHub Actions 处理完成 @@ -50,14 +51,14 @@ | 类型 | 上传格式 | 展示方式 | |------|----------|----------| -| **书籍** | `.pdf` | 章节阅读器 + TOC 侧栏 + 键盘导航 | +| **书籍** | `.pdf` 或 `.epub` | 章节阅读器 + TOC 侧栏 + 键盘导航 | | **文档** | `.md` | 单页 Markdown 渲染,支持代码高亮和数学公式 | | **站点** | `.zip` | 静态站点直接托管,点击新窗口打开 | ## 功能 - **阅读器** — 明暗主题、章节侧边栏、键盘导航、代码高亮(Shiki)、数学公式(KaTeX)、响应式布局 -- **管理面板** — 上传 PDF/Markdown/ZIP、目录管理(编辑、发布、隐藏、归档、删除)、搜索和筛选 +- **管理面板** — 上传 PDF/EPUB/Markdown/ZIP、目录管理(编辑、发布、隐藏、归档、删除)、搜索和筛选 - **处理流水线** — GitHub Actions 自动处理上传内容,大 PDF 自动分块转换 - **首页** — 标签页筛选:全部 / 书籍 / 文档 / 站点 @@ -66,9 +67,10 @@ ``` 上传内容(浏览器 → GitHub API → input/) → GitHub Actions 运行 scripts/process.py - → .pdf: MinerU API → Markdown → 拆分章节 → docs/books/{id}/ - → .md: 复制到 docs/articles/{id}/content.md - → .zip: 解压到 docs/sites/{id}/ + → .pdf: MinerU API → Markdown → 拆分章节 → docs/books/{id}/ + → .epub: pandoc → Markdown + 媒体资源 → 拆分章节 → docs/books/{id}/ + → .md: 复制到 docs/articles/{id}/content.md + → .zip: 解压到 docs/sites/{id}/ → 构建 manifest → GitHub Pages 部署 ``` diff --git a/cli/commands/reconvert.js b/cli/commands/reconvert.js index bc93be6..dae49ac 100644 --- a/cli/commands/reconvert.js +++ b/cli/commands/reconvert.js @@ -8,7 +8,7 @@ async function run(argv) { const { token, repo } = loadConfig(argv); const { items } = await fetchCatalog(repo, token); const { item } = selectCatalogItem(items, id); - if (item.type !== 'book') die('Only PDF books can be re-processed. Re-upload Markdown or ZIP sources instead.'); + if (item.type !== 'book') die('Only books can be re-processed. Re-upload Markdown or ZIP sources instead.'); const clearCache = hasFlag(argv, '--clear-cache'); await triggerReconvert(item, repo, token, { clearCache }); diff --git a/cli/github.js b/cli/github.js index 2c1fab6..8f54b2e 100644 --- a/cli/github.js +++ b/cli/github.js @@ -17,7 +17,7 @@ const FAILURES_PATH = 'docs/failures.json'; const CONTENT_TYPE_DIRS = { book: 'docs/books', doc: 'docs/articles', site: 'docs/sites' }; const VISIBILITY_VALUES = ['published', 'hidden', 'archived']; const MAX_FILE_SIZE = 100 * 1024 * 1024; -const ACCEPTED_EXTENSIONS = ['pdf', 'md', 'zip']; +const ACCEPTED_EXTENSIONS = ['pdf', 'epub', 'md', 'zip']; let catalogSourcePath = CATALOG_DEFAULT_PATH; @@ -172,17 +172,30 @@ async function listRepoTree(repo, path, token) { } async function getCacheDeleteOps(repo, itemId, token) { - let md5; + let pdfMd5; + let epubMd5; try { const f = await readJson(repo, `docs/books/${itemId}/meta.json`, token); - md5 = f.data?.pdf_md5; + pdfMd5 = f.data?.pdf_md5; + epubMd5 = f.data?.epub_md5; } catch { /* skip */ } - if (!md5) { - return []; + const ops = []; + + if (pdfMd5) { + const cacheFiles = await listRepoTree(repo, 'cache/markdown', token); + ops.push(...cacheFiles + .filter((f) => f.name.startsWith(pdfMd5)) + .map((f) => ({ path: f.path, delete: true }))); + } + + if (epubMd5) { + const cacheFiles = await listRepoTree(repo, 'cache/epub', token); + ops.push(...cacheFiles + .filter((f) => f.name.startsWith(epubMd5)) + .map((f) => ({ path: f.path, delete: true }))); } - const cacheFiles = await listRepoTree(repo, 'cache/markdown', token); - return cacheFiles.filter((f) => f.name.startsWith(md5)).map((f) => ({ path: f.path, delete: true })); + return ops; } function isSameCatalogItem(left, right) { diff --git a/cli/index.js b/cli/index.js index 103c0d8..99e4804 100755 --- a/cli/index.js +++ b/cli/index.js @@ -23,12 +23,12 @@ Usage: gitshelf [options] Commands: - upload Upload .pdf, .md, or .zip to GitShelf + upload Upload .pdf, .epub, .md, or .zip to GitShelf list [--type TYPE] List all content items info Show details for one item edit [...] Edit item metadata delete [--yes] Delete an item permanently - reconvert Trigger re-processing for a PDF book + reconvert Trigger re-processing for a book source failures List processing failures failures dismiss Dismiss a failure failures retry Retry a failed conversion diff --git a/cli/mcp-server.mjs b/cli/mcp-server.mjs index 6384142..ffc068a 100644 --- a/cli/mcp-server.mjs +++ b/cli/mcp-server.mjs @@ -3,10 +3,10 @@ /** * GitShelf MCP Server * - * GitShelf is a zero-cost GitHub Pages content platform. Upload PDFs - * (auto-converted to chapter books), Markdown (rendered as documents), - * or ZIP archives (deployed as static sites). Everything is processed - * by GitHub Actions and served from GitHub Pages. + * GitShelf is a zero-cost GitHub Pages content platform. Upload PDFs or + * EPUBs (auto-converted to chapter books), Markdown (rendered as + * documents), or ZIP archives (deployed as static sites). Everything is + * processed by GitHub Actions and served from GitHub Pages. * * This MCP server exposes GitShelf content reading and management tools. * @@ -42,7 +42,7 @@ const server = new McpServer( version: '0.1.3', }, { - instructions: 'GitShelf is a zero-cost GitHub Pages content platform. Upload PDFs (auto-converted to multi-chapter books with TOC and reader UI), Markdown files (rendered as documents), or ZIP archives (deployed as static sites). Everything is processed by GitHub Actions and served from GitHub Pages. Use these tools to browse, read, and manage GitShelf content.', + instructions: 'GitShelf is a zero-cost GitHub Pages content platform. Upload PDFs or EPUBs (auto-converted to multi-chapter books with TOC and reader UI), Markdown files (rendered as documents), or ZIP archives (deployed as static sites). Everything is processed by GitHub Actions and served from GitHub Pages. Use these tools to browse, read, and manage GitShelf content.', }, ); @@ -233,7 +233,7 @@ server.registerTool( 'upload', { title: 'Upload Content', - description: 'Upload a local file (.pdf, .md, or .zip) to GitShelf for processing.', + description: 'Upload a local file (.pdf, .epub, .md, or .zip) to GitShelf for processing.', inputSchema: z.object({ file_path: z.string().describe('Absolute path to the file to upload'), }), diff --git a/scripts/generate_structure.py b/scripts/generate_structure.py index 2d71ca8..d0efb07 100644 --- a/scripts/generate_structure.py +++ b/scripts/generate_structure.py @@ -30,9 +30,9 @@ def _slugify_anchor(text: str) -> str: return anchor.strip("-") -def _extract_subheadings(content: str) -> list[dict[str, str]]: - """Find H2 headings within a chapter to produce sub-children entries with anchors.""" - sub_level = 2 +def _extract_subheadings(content: str, chapter_level: int = 1) -> list[dict[str, str]]: + """Find sub-headings immediately below the chapter level for TOC children.""" + sub_level = chapter_level + 1 pattern = re.compile(rf"^{'#' * sub_level}(?!#)\s+(.+)$", re.MULTILINE) subheadings: list[dict[str, str]] = [] for match in pattern.finditer(content): @@ -47,6 +47,7 @@ def _extract_subheadings(content: str) -> list[dict[str, str]]: def _build_toc( title: str, chapters: list[Chapter], + chapter_level: int = 1, ) -> dict: """Build the toc.json structure from a list of chapters. @@ -56,7 +57,7 @@ def _build_toc( children: list[dict] = [] for chapter in chapters: entry: dict = {"title": chapter.title, "slug": chapter.slug} - subheadings = _extract_subheadings(chapter.content) + subheadings = _extract_subheadings(chapter.content, chapter_level=chapter_level) if subheadings: entry["children"] = [ { @@ -93,6 +94,8 @@ def generate_book_structure( title: str, chapters: list[Chapter], output_dir: Path = Path("docs/books"), + *, + chapter_level: int = 1, ) -> Path: """Create book directory, write chapter files, generate toc.json. @@ -123,7 +126,7 @@ def generate_book_structure( chapter_path = chapters_dir / f"{chapter.slug}.md" chapter_path.write_text(chapter.content, encoding="utf-8") - toc = _build_toc(title, chapters) + toc = _build_toc(title, chapters, chapter_level=chapter_level) toc_path = book_dir / "toc.json" toc_path.write_text(json.dumps(toc, indent=2, ensure_ascii=False) + "\n", encoding="utf-8") diff --git a/scripts/process.py b/scripts/process.py index 2af61cb..0a102d3 100644 --- a/scripts/process.py +++ b/scripts/process.py @@ -1,20 +1,24 @@ #!/usr/bin/env python3 """Unified content processing pipeline. -Handles three content types from input/: - - .pdf → book (chapters via MinerU API) - - .md → article (single markdown document) - - .zip → site (static site extraction) +Handles four content types from input/: + - .pdf → book (chapters via MinerU API) + - .epub → book (chapters via pandoc) + - .md → article (single markdown document) + - .zip → site (static site extraction) Usage: python scripts/process.py [--input-dir INPUT] [--output-dir OUTPUT] """ import argparse +import hashlib import json import os import re import shutil +import subprocess import sys +import tempfile import zipfile from datetime import datetime, timezone from pathlib import Path @@ -24,6 +28,13 @@ except ImportError: from build_manifest import build_manifest +try: + from .split_markdown import split_by_headings + from .generate_structure import generate_book_structure +except ImportError: + from split_markdown import split_by_headings + from generate_structure import generate_book_structure + # Reuse PDF pipeline from convert.py try: from .convert import ( @@ -32,6 +43,7 @@ ensure_unique_content_id, generate_book_id, reconvert_from_cache, + _rewrite_chapter_image_paths, _write_failures, _remove_failure, ) @@ -42,11 +54,14 @@ ensure_unique_content_id, generate_book_id, reconvert_from_cache, + _rewrite_chapter_image_paths, _write_failures, _remove_failure, ) FAILURES_FILENAME = "failures.json" +BOOK_METADATA_FILENAME = "meta.json" +EPUB_CACHE_DIR = Path("cache/epub") def _utc_now_iso() -> str: @@ -58,6 +73,306 @@ def _generate_id(path: Path) -> str: return generate_book_id(path) +def _file_md5(path: Path) -> str: + """Compute MD5 hex digest for an input file.""" + h = hashlib.md5() + with open(path, "rb") as f: + for chunk in iter(lambda: f.read(8192), b""): + h.update(chunk) + return h.hexdigest() + + +def detect_new_epubs(input_dir: Path) -> list[Path]: + """Find .epub files in input_dir.""" + return sorted(input_dir.glob("*.epub")) + + +def _read_existing_created_at(book_dir: Path, fallback: str) -> str: + meta_path = book_dir / BOOK_METADATA_FILENAME + if not meta_path.exists(): + return fallback + + try: + existing = json.loads(meta_path.read_text(encoding="utf-8")) + except json.JSONDecodeError: + return fallback + + return str(existing.get("created_at", "")).strip() or fallback + + +def _write_epub_metadata( + book_dir: Path, + *, + book_id: str, + source_epub: str, + epub_md5: str, + updated_at: str, + created_at: str | None = None, +) -> None: + normalized_created_at = created_at or _read_existing_created_at(book_dir, updated_at) + data = { + "id": book_id, + "type": "book", + "source": source_epub, + "source_format": "epub", + "epub_md5": epub_md5, + "created_at": normalized_created_at, + "updated_at": updated_at, + } + (book_dir / BOOK_METADATA_FILENAME).write_text( + json.dumps(data, indent=2, ensure_ascii=False) + "\n", + encoding="utf-8", + ) + + +def _write_epub_cache(md5: str, epub_path: Path) -> None: + EPUB_CACHE_DIR.mkdir(parents=True, exist_ok=True) + shutil.copy2(epub_path, EPUB_CACHE_DIR / f"{md5}.epub") + + +def _find_cached_epub(output_dir: Path, source_epub: str) -> tuple[str, str] | None: + """Return ``(book_id, md5)`` for a cached EPUB source.""" + for meta_file in output_dir.glob(f"*/{BOOK_METADATA_FILENAME}"): + try: + meta = json.loads(meta_file.read_text(encoding="utf-8")) + except json.JSONDecodeError: + continue + + if meta.get("source") == source_epub and meta.get("epub_md5"): + return meta_file.parent.name, str(meta["epub_md5"]) + return None + + +def _sanitize_pandoc_markdown(markdown: str) -> str: + """Remove pandoc EPUB wrapper elements that add noise to rendered chapters.""" + sanitized = re.sub( + r"^\s*]*>\s*\s*$\n?", + "", + markdown, + flags=re.MULTILINE, + ) + sanitized = re.sub( + r"^\s*]*class=[\"'][^\"']*\bsection\b[^\"']*[\"'][^>]*>\s*$\n?", + "", + sanitized, + flags=re.MULTILINE, + ) + sanitized = re.sub(r"^\s*\s*$\n?", "", sanitized, flags=re.MULTILINE) + sanitized = re.sub(r"\n{3,}", "\n\n", sanitized).strip() + return sanitized + "\n" if sanitized else "" + + +def _rewrite_book_asset_paths( + markdown: str, + source_prefixes: tuple[str, ...], + dest_prefix: str, +) -> str: + """Rewrite relative markdown assets from one local prefix to another.""" + + def _normalize_path(raw: str) -> str: + value = str(raw or "").strip() + if ( + not value + or value.startswith(("/", "#", "//", "data:")) + or re.match(r"^[a-z][a-z0-9+.-]*:", value, flags=re.IGNORECASE) + ): + return value + + for prefix in source_prefixes: + if value.startswith(prefix): + return f"{dest_prefix}{value[len(prefix):]}" + + return value + + markdown = re.sub( + r"(!\[[^\]]*\]\()([^)]+)(\))", + lambda m: f"{m.group(1)}{_normalize_path(m.group(2))}{m.group(3)}", + markdown, + ) + markdown = re.sub( + r'(]*\bsrc=["\'])([^"\']+)(["\'][^>]*>)', + lambda m: f"{m.group(1)}{_normalize_path(m.group(2))}{m.group(3)}", + markdown, + flags=re.IGNORECASE, + ) + return markdown + + +def _select_chapter_level(markdown: str) -> tuple[list, int]: + """Pick a usable heading level for book chapters, preferring multi-chapter splits.""" + fallback: tuple[list, int] | None = None + for level in (1, 2, 3): + try: + chapters = split_by_headings(markdown, level=level) + except ValueError: + continue + + if fallback is None: + fallback = (chapters, level) + + chapter_count = sum(1 for chapter in chapters if chapter.slug != "00-preface") + if chapter_count >= 2: + return chapters, level + + if fallback is None: + raise ValueError("No headings found in EPUB-derived markdown.") + + return fallback + + +def _copy_epub_media(work_dir: Path, book_dir: Path) -> int: + """Copy pandoc-extracted media into the book's images directory.""" + media_dir = work_dir / "media" + if not media_dir.is_dir(): + return 0 + + images_dir = book_dir / "images" + copied = 0 + for source in media_dir.rglob("*"): + if not source.is_file(): + continue + destination = images_dir / source.relative_to(media_dir) + destination.parent.mkdir(parents=True, exist_ok=True) + shutil.copy2(source, destination) + copied += 1 + return copied + + +def _run_pandoc_epub(epub_path: Path, work_dir: Path) -> str: + """Convert EPUB to Markdown using pandoc within a temporary work directory.""" + output_name = "book.md" + command = [ + "pandoc", + str(epub_path.resolve()), + "-t", + "gfm", + "--wrap=none", + "--extract-media=.", + "-o", + output_name, + ] + + try: + subprocess.run( + command, + check=True, + cwd=work_dir, + capture_output=True, + text=True, + ) + except FileNotFoundError as exc: + raise RuntimeError( + "pandoc is required to process EPUB files. Install pandoc locally " + "and in GitHub Actions." + ) from exc + except subprocess.CalledProcessError as exc: + detail = (exc.stderr or exc.stdout or "").strip() + raise RuntimeError( + f"pandoc failed to convert {epub_path.name}: {detail[:200]}" + ) from exc + + output_path = work_dir / output_name + if not output_path.exists(): + raise RuntimeError(f"pandoc did not produce {output_name} for {epub_path.name}") + + return output_path.read_text(encoding="utf-8") + + +def _build_epub_book(epub_path: Path, output_dir: Path, book_id: str, title: str) -> int: + """Convert an EPUB source into the standard book directory structure.""" + book_dir = output_dir / book_id + if book_dir.exists(): + shutil.rmtree(book_dir) + + with tempfile.TemporaryDirectory(prefix="gitshelf_epub_") as tmp_dir: + work_dir = Path(tmp_dir) + markdown = _run_pandoc_epub(epub_path, work_dir) + markdown = _sanitize_pandoc_markdown(markdown) + markdown = _rewrite_book_asset_paths( + markdown, + source_prefixes=("media/", "./media/"), + dest_prefix="images/", + ) + markdown = _rewrite_chapter_image_paths(markdown, book_id) + + chapters, chapter_level = _select_chapter_level(markdown) + generate_book_structure( + book_id, + title, + chapters, + chapter_level=chapter_level, + output_dir=output_dir, + ) + copied = _copy_epub_media(work_dir, book_dir) + + return copied + + +def process_epub(epub_path: Path, output_dir: Path) -> None: + """Process a single .epub file into a multi-chapter book.""" + book_id = ensure_unique_content_id(_generate_id(epub_path), output_dir.parent, "book") + title = epub_path.stem + md5 = _file_md5(epub_path) + book_dir = output_dir / book_id + timestamp = _utc_now_iso() + created_at = _read_existing_created_at(book_dir, timestamp) + + print(f"Processing EPUB: {epub_path.name} -> {book_id}") + + copied_assets = _build_epub_book(epub_path, output_dir, book_id, title) + if copied_assets: + print(f" Extracted {copied_assets} EPUB assets") + + _write_epub_cache(md5, epub_path) + + _write_epub_metadata( + book_dir, + book_id=book_id, + source_epub=epub_path.name, + epub_md5=md5, + created_at=created_at, + updated_at=timestamp, + ) + + epub_path.unlink(missing_ok=True) + print(f" Deleted source: {epub_path.name}") + + +def reconvert_epub_from_cache(source_epub: str, output_dir: Path) -> None: + """Rebuild an EPUB-backed book from cached source bytes.""" + cached = _find_cached_epub(output_dir, source_epub) + if not cached: + raise FileNotFoundError( + f"No cached EPUB conversion found for {source_epub}. Re-upload the EPUB." + ) + + book_id, md5 = cached + cached_epub = EPUB_CACHE_DIR / f"{md5}.epub" + if not cached_epub.exists(): + raise FileNotFoundError( + f"Cached EPUB missing for MD5 {md5}. Re-upload the EPUB." + ) + + title = Path(source_epub).stem + book_dir = output_dir / book_id + timestamp = _utc_now_iso() + created_at = _read_existing_created_at(book_dir, timestamp) + print(f"Reconverting EPUB from cache: {source_epub} -> {book_id} (md5={md5})") + copied_assets = _build_epub_book(cached_epub, output_dir, book_id, title) + if copied_assets: + print(f" Extracted {copied_assets} EPUB assets") + + _write_epub_metadata( + book_dir, + book_id=book_id, + source_epub=source_epub, + epub_md5=md5, + created_at=created_at, + updated_at=timestamp, + ) + print(" Reconversion complete.") + + # --- Markdown processing --- def _count_words(text: str) -> int: @@ -295,7 +610,7 @@ def _resolve_input_file(input_dir: Path, filename: str) -> Path: def main() -> None: - parser = argparse.ArgumentParser(description="Process content (PDF, Markdown, ZIP).") + parser = argparse.ArgumentParser(description="Process content (PDF, EPUB, Markdown, ZIP).") parser.add_argument("--input-dir", type=Path, default=Path("input")) parser.add_argument("--output-dir", type=Path, default=Path("docs")) args = parser.parse_args() @@ -311,6 +626,7 @@ def main() -> None: # Collect jobs by type pdf_jobs: list[Path] = [] + epub_jobs: list[Path] = [] md_jobs: list[Path] = [] zip_jobs: list[Path] = [] @@ -320,6 +636,8 @@ def main() -> None: ext = path.suffix.lower() if ext == ".pdf": pdf_jobs = [path] + elif ext == ".epub": + epub_jobs = [path] elif ext == ".md": md_jobs = [path] elif ext == ".zip": @@ -346,20 +664,41 @@ def main() -> None: except FileNotFoundError as exc: print(str(exc), file=sys.stderr) sys.exit(1) + elif input_filename.lower().endswith(".epub"): + print(f"EPUB not found, attempting reconvert from cache: {input_filename}") + try: + reconvert_epub_from_cache(input_filename, books_dir) + build_manifest( + books_dir=books_dir, + output_path=manifest_path, + catalog_metadata_path=metadata_path, + catalog_output_path=catalog_path, + articles_dir=articles_dir, + sites_dir=sites_dir, + ) + print("Manifest rebuilt.") + return + except FileNotFoundError as exc: + print(str(exc), file=sys.stderr) + sys.exit(1) else: print(f"File not found: {input_filename}", file=sys.stderr) sys.exit(1) else: pdf_jobs = detect_new_pdfs(args.input_dir) + epub_jobs = detect_new_epubs(args.input_dir) md_jobs = sorted(args.input_dir.glob("*.md")) zip_jobs = sorted(args.input_dir.glob("*.zip")) - total = len(pdf_jobs) + len(md_jobs) + len(zip_jobs) + total = len(pdf_jobs) + len(epub_jobs) + len(md_jobs) + len(zip_jobs) if total == 0: print("No new content found in input/. Nothing to do.") return - print(f"Found {total} item(s) to process: {len(pdf_jobs)} PDF, {len(md_jobs)} MD, {len(zip_jobs)} ZIP") + print( + f"Found {total} item(s) to process: {len(pdf_jobs)} PDF, " + f"{len(epub_jobs)} EPUB, {len(md_jobs)} MD, {len(zip_jobs)} ZIP" + ) failures: list[tuple[Path, Exception]] = [] @@ -371,6 +710,15 @@ def main() -> None: print(f" FAILED: {pdf_path.name}: {exc}", file=sys.stderr) failures.append((pdf_path, exc)) + # Process EPUB files + for epub_path in epub_jobs: + try: + process_epub(epub_path, books_dir) + _remove_failure(epub_path.name, args.output_dir) + except Exception as exc: + print(f" FAILED: {epub_path.name}: {exc}", file=sys.stderr) + failures.append((epub_path, exc)) + # Process Markdown files for md_path in md_jobs: try: diff --git a/src/components/AdminView.jsx b/src/components/AdminView.jsx index cb9299b..f364f93 100644 --- a/src/components/AdminView.jsx +++ b/src/components/AdminView.jsx @@ -8,7 +8,7 @@ import { loadHistory, fetchFailures, dismissFailure, retryFailure, getDisplayTitle, normalizeVisibility, normalizeTags, parseTagsInput, formatBytes, dateFormatter, dateTimeFormatter, - numberFormatter, MAX_FILE_SIZE, VISIBILITY_VALUES, toIsoNow, + numberFormatter, ACCEPTED_EXTENSIONS, MAX_FILE_SIZE, VISIBILITY_VALUES, toIsoNow, } from '../lib/github-api'; const CONTENT_TYPE_LABELS = { @@ -125,12 +125,17 @@ function UploadSection({ repo, disabled }) { const [error, setError] = useState(''); const [dragOver, setDragOver] = useState(false); let dragCounter = 0; + const acceptedExtensionsLabel = ACCEPTED_EXTENSIONS.map((ext) => `.${ext}`).join(','); + const acceptedExtensionsText = ACCEPTED_EXTENSIONS.map((ext) => `.${ext}`).join(', '); const handleFile = async (file) => { if (!file || disabled) return; setError(''); const ext = file.name.split('.').pop().toLowerCase(); - if (!['pdf', 'md', 'zip'].includes(ext)) { setError('Only .pdf, .md, and .zip files are accepted.'); return; } + if (!ACCEPTED_EXTENSIONS.includes(ext)) { + setError(`Only ${acceptedExtensionsText} files are accepted.`); + return; + } if (file.size > MAX_FILE_SIZE) { setError(`File too large (${formatBytes(file.size)}). Max 100 MB.`); return; } try { await apiUploadContent(file, repo, (stage, msg) => setProgress({ stage, msg })); @@ -158,7 +163,7 @@ function UploadSection({ repo, disabled }) { role="button" tabIndex="0" aria-label="Upload content file" - onClick={() => { const i = document.createElement('input'); i.type = 'file'; i.accept = '.pdf,.md,.zip'; i.onchange = () => handleFile(i.files[0]); i.click(); }} + onClick={() => { const i = document.createElement('input'); i.type = 'file'; i.accept = acceptedExtensionsLabel; i.onchange = () => handleFile(i.files[0]); i.click(); }} onKeyDown={(e) => { if (e.key === 'Enter' || e.key === ' ') { e.preventDefault(); e.currentTarget.click(); }}} onDragEnter={(e) => { e.preventDefault(); dragCounter++; setDragOver(true); }} onDragOver={(e) => e.preventDefault()} @@ -168,7 +173,7 @@ function UploadSection({ repo, disabled }) { -
Drop PDF, Markdown, or ZIP here or click to browse
+
Drop PDF, EPUB, Markdown, or ZIP here or click to browse
Maximum 100 MB
)} @@ -571,7 +576,7 @@ function CatalogSection({ repo }) { -

No content yet. Upload a PDF, Markdown file, or ZIP to get started.

+

No content yet. Upload a PDF, EPUB, Markdown file, or ZIP to get started.

) : ( <> diff --git a/src/components/HomeView.jsx b/src/components/HomeView.jsx index 68891a3..e906f2f 100644 --- a/src/components/HomeView.jsx +++ b/src/components/HomeView.jsx @@ -31,8 +31,8 @@ const TABS = [ ]; const EMPTY_MESSAGES = { - all: { title: 'No content yet', hint: 'Upload a PDF, Markdown file, or ZIP archive via the {admin} to get started.' }, - book: { title: 'No books yet', hint: 'Upload a PDF via the {admin} to add your first book.' }, + all: { title: 'No content yet', hint: 'Upload a PDF, EPUB, Markdown file, or ZIP archive via the {admin} to get started.' }, + book: { title: 'No books yet', hint: 'Upload a PDF or EPUB via the {admin} to add your first book.' }, doc: { title: 'No documents yet', hint: 'Upload a Markdown file via the {admin} to add a document.' }, site: { title: 'No sites yet', hint: 'Upload a ZIP archive via the {admin} to publish a static site.' }, }; diff --git a/src/lib/github-api.js b/src/lib/github-api.js index f4a6cca..f268dca 100644 --- a/src/lib/github-api.js +++ b/src/lib/github-api.js @@ -10,7 +10,7 @@ const CATALOG_DEFAULT_PATH = 'docs/catalog.json'; const CATALOG_METADATA_PATH = 'docs/catalog-metadata.json'; const VISIBILITY_VALUES = ['published', 'hidden', 'archived']; const MAX_FILE_SIZE = 100 * 1024 * 1024; -const ACCEPTED_EXTENSIONS = ['pdf', 'md', 'zip']; +const ACCEPTED_EXTENSIONS = ['pdf', 'epub', 'md', 'zip']; const CONTENT_TYPE_DIRS = { book: 'docs/books', @@ -298,19 +298,30 @@ async function listRepoTree(repo, path) { } async function getCacheDeleteOps(repo, itemId) { - let md5; + let pdfMd5; + let epubMd5; try { const f = await readRepositoryJson(repo, `docs/books/${itemId}/meta.json`); - md5 = f.data?.pdf_md5; + pdfMd5 = f.data?.pdf_md5; + epubMd5 = f.data?.epub_md5; } catch { /* not found */ } - if (!md5) { - return []; + const ops = []; + + if (pdfMd5) { + const cacheFiles = await listRepoTree(repo, 'cache/markdown'); + ops.push(...cacheFiles + .filter(f => f.name.startsWith(pdfMd5)) + .map(f => ({ path: f.path, delete: true }))); + } + + if (epubMd5) { + const cacheFiles = await listRepoTree(repo, 'cache/epub'); + ops.push(...cacheFiles + .filter(f => f.name.startsWith(epubMd5)) + .map(f => ({ path: f.path, delete: true }))); } - const cacheFiles = await listRepoTree(repo, 'cache/markdown'); - return cacheFiles - .filter(f => f.name.startsWith(md5)) - .map(f => ({ path: f.path, delete: true })); + return ops; } export async function deleteItemPermanently(item, repo, catalog) { @@ -375,4 +386,4 @@ export const dateFormatter = new Intl.DateTimeFormat(undefined, { year: 'numeric export const dateTimeFormatter = new Intl.DateTimeFormat(undefined, { year: 'numeric', month: 'short', day: 'numeric', hour: '2-digit', minute: '2-digit' }); export const numberFormatter = new Intl.NumberFormat(undefined); -export { MAX_FILE_SIZE, VISIBILITY_VALUES, toIsoNow }; +export { ACCEPTED_EXTENSIONS, MAX_FILE_SIZE, VISIBILITY_VALUES, toIsoNow }; diff --git a/tests/cli/commands.test.js b/tests/cli/commands.test.js index ff44fc9..c8c7c3b 100644 --- a/tests/cli/commands.test.js +++ b/tests/cli/commands.test.js @@ -129,7 +129,7 @@ describe('cli commands', () => { }); await expect(run(['doc-1'])).rejects.toThrow( - 'Only PDF books can be re-processed. Re-upload Markdown or ZIP sources instead.', + 'Only books can be re-processed. Re-upload Markdown or ZIP sources instead.', ); expect(triggerReconvert).not.toHaveBeenCalled(); }); diff --git a/tests/scripts/test_process.py b/tests/scripts/test_process.py new file mode 100644 index 0000000..6b15ec1 --- /dev/null +++ b/tests/scripts/test_process.py @@ -0,0 +1,146 @@ +import json +import subprocess +import sys +import tempfile +import unittest +import hashlib +from pathlib import Path +from unittest.mock import patch + +REPO_ROOT = Path(__file__).resolve().parents[2] +sys.path.insert(0, str(REPO_ROOT / "scripts")) + +import process + + +class EpubProcessingTest(unittest.TestCase): + def test_process_epub_builds_book_structure_and_cache(self) -> None: + with tempfile.TemporaryDirectory() as tmp_dir: + root = Path(tmp_dir) + books_dir = root / "docs" / "books" + books_dir.mkdir(parents=True, exist_ok=True) + + epub_path = root / "input" / "sample.epub" + epub_path.parent.mkdir(parents=True, exist_ok=True) + epub_path.write_bytes(b"epub-bytes") + + markdown = """ + + +
+## Chapter One + +![Cover](media/cover.png) + +### Section One + +Hello world. +
+ +
+## Chapter Two + +### Section Two + +More text. +
+""".strip() + + def fake_run(cmd, check, cwd, capture_output, text): + work_dir = Path(cwd) + (work_dir / "book.md").write_text(markdown, encoding="utf-8") + media_dir = work_dir / "media" + media_dir.mkdir(parents=True, exist_ok=True) + (media_dir / "cover.png").write_bytes(b"png") + return subprocess.CompletedProcess(cmd, 0, "", "") + + cache_dir = root / "cache" / "epub" + with patch.object(process, "EPUB_CACHE_DIR", cache_dir), patch.object( + process, "_utc_now_iso", return_value="2026-04-12T12:00:00Z" + ), patch.object(process.subprocess, "run", side_effect=fake_run): + process.process_epub(epub_path, books_dir) + + book_dir = books_dir / "sample" + expected_md5 = hashlib.md5(b"epub-bytes").hexdigest() + self.assertFalse(epub_path.exists()) + self.assertTrue((book_dir / "images" / "cover.png").exists()) + self.assertTrue((cache_dir / f"{expected_md5}.epub").exists()) + + toc = json.loads((book_dir / "toc.json").read_text(encoding="utf-8")) + self.assertEqual(toc["title"], "sample") + self.assertEqual(toc["children"][0]["title"], "Chapter One") + self.assertEqual(toc["children"][0]["children"][0]["title"], "Section One") + self.assertEqual(toc["children"][1]["title"], "Chapter Two") + + chapter_one = (book_dir / "chapters" / "01-chapter-one.md").read_text(encoding="utf-8") + self.assertIn("![Cover](../images/cover.png)", chapter_one) + self.assertNotIn(" None: + with tempfile.TemporaryDirectory() as tmp_dir: + root = Path(tmp_dir) + books_dir = root / "docs" / "books" + book_dir = books_dir / "cached-book" + book_dir.mkdir(parents=True, exist_ok=True) + (book_dir / "meta.json").write_text( + json.dumps( + { + "id": "cached-book", + "type": "book", + "source": "cached.epub", + "source_format": "epub", + "epub_md5": "abc123", + "created_at": "2026-04-10T08:00:00Z", + "updated_at": "2026-04-10T08:00:00Z", + }, + indent=2, + ) + + "\n", + encoding="utf-8", + ) + + cache_dir = root / "cache" / "epub" + cache_dir.mkdir(parents=True, exist_ok=True) + (cache_dir / "abc123.epub").write_bytes(b"cached-epub") + + markdown = """ +
+# Chapter One + +Body text. +
+ +
+# Chapter Two + +More text. +
+""".strip() + + def fake_run(cmd, check, cwd, capture_output, text): + (Path(cwd) / "book.md").write_text(markdown, encoding="utf-8") + return subprocess.CompletedProcess(cmd, 0, "", "") + + with patch.object(process, "EPUB_CACHE_DIR", cache_dir), patch.object( + process, "_utc_now_iso", return_value="2026-04-12T12:00:00Z" + ), patch.object(process.subprocess, "run", side_effect=fake_run): + process.reconvert_epub_from_cache("cached.epub", books_dir) + + toc = json.loads((book_dir / "toc.json").read_text(encoding="utf-8")) + self.assertEqual([entry["title"] for entry in toc["children"]], ["Chapter One", "Chapter Two"]) + + meta = json.loads((book_dir / "meta.json").read_text(encoding="utf-8")) + self.assertEqual(meta["created_at"], "2026-04-10T08:00:00Z") + self.assertEqual(meta["updated_at"], "2026-04-12T12:00:00Z") + + +if __name__ == "__main__": + unittest.main()