|
| 1 | +#!/usr/bin/env python3 |
| 2 | +import argparse |
| 3 | +import subprocess |
| 4 | +import sys |
| 5 | +from pathlib import Path |
| 6 | +from urllib.parse import urlparse |
| 7 | + |
| 8 | +API_BASE_URL = "https://api.liujiacai.net/ai/markdown?url=" |
| 9 | + |
| 10 | + |
| 11 | +def extract_slug(url: str) -> str: |
| 12 | + """Extract the last slug from URL path.""" |
| 13 | + parsed = urlparse(url.strip()) |
| 14 | + path = parsed.path.rstrip("/") |
| 15 | + if not path: |
| 16 | + return "index" |
| 17 | + slug = path.split("/")[-1] |
| 18 | + return slug or "index" |
| 19 | + |
| 20 | + |
| 21 | +def fetch_url(api_url: str, target_file: Path) -> bool: |
| 22 | + """Fetch markdown content via curl to properly respect system proxy settings.""" |
| 23 | + try: |
| 24 | + result = subprocess.run( |
| 25 | + ["curl", "-sSL", "-f", api_url, "-o", str(target_file)], |
| 26 | + check=True, |
| 27 | + capture_output=True, |
| 28 | + text=True, |
| 29 | + ) |
| 30 | + return True |
| 31 | + except subprocess.CalledProcessError as e: |
| 32 | + print(f" -> curl error: {e.stderr.strip()}", file=sys.stderr) |
| 33 | + return False |
| 34 | + |
| 35 | + |
| 36 | +def fetch_and_save(source_file: Path, output_dir: Path) -> None: |
| 37 | + """Read URLs from source_file, call API, and save markdown output.""" |
| 38 | + if not source_file.exists(): |
| 39 | + print(f"Error: source file not found: {source_file}", file=sys.stderr) |
| 40 | + sys.exit(1) |
| 41 | + |
| 42 | + output_dir.mkdir(parents=True, exist_ok=True) |
| 43 | + |
| 44 | + with open(source_file, "r", encoding="utf-8") as f: |
| 45 | + urls = [line.strip() for line in f if line.strip() and not line.startswith("#")] |
| 46 | + |
| 47 | + total = len(urls) |
| 48 | + print(f"Found {total} URLs in {source_file}") |
| 49 | + |
| 50 | + success_count = 0 |
| 51 | + for idx, url in enumerate(urls, 1): |
| 52 | + slug = extract_slug(url) |
| 53 | + target_file = output_dir / f"{slug}.md" |
| 54 | + api_url = f"{API_BASE_URL}{url}" |
| 55 | + |
| 56 | + print(f"[{idx}/{total}] Fetching {slug} ({url}) ...") |
| 57 | + if fetch_url(api_url, target_file): |
| 58 | + print(f" -> Saved to {target_file}") |
| 59 | + success_count += 1 |
| 60 | + |
| 61 | + print(f"Done. Successfully saved {success_count}/{total} files.") |
| 62 | + |
| 63 | + |
| 64 | +def main(): |
| 65 | + script_dir = Path(__file__).resolve().parent |
| 66 | + parser = argparse.ArgumentParser( |
| 67 | + description="Fetch markdown content from URLs listed in source.txt" |
| 68 | + ) |
| 69 | + parser.add_argument( |
| 70 | + "--source", |
| 71 | + "-s", |
| 72 | + type=Path, |
| 73 | + default=script_dir / "source.txt", |
| 74 | + help="Path to source.txt file (default: utils/source.txt)", |
| 75 | + ) |
| 76 | + parser.add_argument( |
| 77 | + "--output-dir", |
| 78 | + "-o", |
| 79 | + type=Path, |
| 80 | + default=script_dir / "learning-zig-src", |
| 81 | + help="Directory to save markdown files (default: utils/learning-zig-src)", |
| 82 | + ) |
| 83 | + |
| 84 | + args = parser.parse_args() |
| 85 | + fetch_and_save(args.source, args.output_dir) |
| 86 | + |
| 87 | + |
| 88 | +if __name__ == "__main__": |
| 89 | + main() |
0 commit comments