From e087a6eb833d2ab4c22f4b8492200bff47c52cdd Mon Sep 17 00:00:00 2001 From: Marcus Date: Tue, 13 Jan 2026 21:40:25 -0600 Subject: [PATCH] Add gopher integration core infrastructure (Phase 1) Create scripts for converting Hugo blog posts to gopher-friendly plain text format. Includes ASCII art assets for headers/footers and markdown-to-gopher conversion with proper line wrapping. --- .gitignore | 3 + scripts/gopher/__init__.py | 1 + scripts/gopher/ascii_art.py | 204 ++++++++++++++++ scripts/gopher/convert_to_gopher.py | 350 ++++++++++++++++++++++++++++ 4 files changed, 558 insertions(+) create mode 100644 scripts/gopher/__init__.py create mode 100644 scripts/gopher/ascii_art.py create mode 100644 scripts/gopher/convert_to_gopher.py diff --git a/.gitignore b/.gitignore index fc20745..ef48340 100644 --- a/.gitignore +++ b/.gitignore @@ -29,3 +29,6 @@ scripts/config.py # Claude Code .claude/ + +# Gopher build output +gopher_build/ diff --git a/scripts/gopher/__init__.py b/scripts/gopher/__init__.py new file mode 100644 index 0000000..fa05f06 --- /dev/null +++ b/scripts/gopher/__init__.py @@ -0,0 +1 @@ +# Gopher conversion scripts for marcus-web diff --git a/scripts/gopher/ascii_art.py b/scripts/gopher/ascii_art.py new file mode 100644 index 0000000..081a5e3 --- /dev/null +++ b/scripts/gopher/ascii_art.py @@ -0,0 +1,204 @@ +#!/usr/bin/env python3 +""" +ASCII art assets for gopher content generation. + +Contains category headers, post templates, footers, and utility functions +for generating ASCII tables and dividers. +""" + +# Line width for gopher content +LINE_WIDTH = 70 + +# Dividers +DOUBLE_LINE = "=" * LINE_WIDTH +SINGLE_LINE = "-" * LINE_WIDTH +BOX_LINE = "\u2500" * LINE_WIDTH # Unicode box drawing horizontal + +# Post header (MNW owl) +POST_HEADER = """\ + ___ ___ ___ + /\\ \\ /\\ \\ /\\ \\ + |::\\ \\ \\:\\ \\ _\\:\\ \\ + |:|:\\ \\ \\:\\ \\ /\\ \\:\\ \\ + __|:|\\:\\ \\ _____\\:\\ \\ _\\:\\ \\:\\ \\ + /::::|_\\:\\__\\ /::::::::\\__\\ /\\ \\:\\ \\:\\__\\ + \\:\\~~\\ \\/__/ \\:\\~~\\~~\\/__/ \\:\\ \\:\\/:/ / + \\:\\ \\ \\:\\ \\ \\:\\ \\::/ / + \\:\\ \\ \\:\\ \\ \\:\\/:/ / + \\:\\__\\ \\:\\__\\ \\::/ / + \\/__/ \\/__/ \\/__/ + +mnw(at)sdf.org | SDF VOIP Ext. 1908 +""" + +# Category headers +HEADER_FUN_CENTER = """\ + ___ ___ _ + | __| _ _ _ / __|___ _ _| |_ ___ _ _ + | _| || | ' \\ | (__/ -_) ' \\ _/ -_) '_| + |_| \\_,_|_||_| \\___\\___|_||_\\__\\___|_| + + Tech posts from the Double Lunch Dispatch +""" + +HEADER_FRANKS_COUCH = """\ + ___ _ _ ___ _ + | __| _ __ _ _ _| |_( )___ / __|___ _ _ __| |_ + | _| '_/ _` | ' \\ / /|_-< | (__/ _ \\ || (_-< ' \\ + |_||_| \\__,_|_||_\\_\\ /__/ \\___\\___/\\_,_/__/_||_| + + Movie reviews from the couch +""" + +HEADER_BEERCALLS = """\ + ___ _ _ + | _ ) ___ ___ _ _ __ __ _| | |___ + | _ \\/ -_) -_) '_/ _/ _` | | (_-< + |___/\\___\\___|_| \\__\\__,_|_|_/__/ + + Thursday night adventures in Austin +""" + +HEADER_BLOG_INDEX = """\ + ___ _ ___ _ _ + | _ )| |___ __ | __|_ _ | |_ _ _ (_)___ ___ + | _ \\| / _ \\/ _|| _|| ' \\| _| '_|| / -_|_-< + |___/|_\\___/\\__||___|_||_|\\__|_| |_\\___/__/ + + Posts from mnw.sdf.org +""" + +# Footer (SDF box) +POST_FOOTER = """\ + __^__ __^__ +( ___ )------------------------------------------------------( ___ ) + | / | | \\ | + | / | This gopher space proudly hosted by SDF | \\ | + | \\ | mnw on pixelfed.social and tilde.zone | / | + | \\ | | / | +(_____)------------------------------------------------------(_____) +""" + +# Map series names to gopher directory names +SERIES_TO_DIR = { + "Fun Center": "fun-center", + "Frank's Couch": "franks-couch", + "Beercalls": "beercalls", +} + +# Map directory names to headers +DIR_TO_HEADER = { + "fun-center": HEADER_FUN_CENTER, + "franks-couch": HEADER_FRANKS_COUCH, + "beercalls": HEADER_BEERCALLS, +} + +# Map directory names to descriptions +DIR_TO_DESCRIPTION = { + "fun-center": "Tech posts from the Double Lunch Dispatch", + "franks-couch": "Movie reviews from the couch", + "beercalls": "Thursday night adventures in Austin", +} + + +def get_section_header(title: str) -> str: + """Generate a section header with double lines.""" + return f"\n{DOUBLE_LINE}\n {title.upper()}\n{DOUBLE_LINE}\n" + + +def get_subheading(title: str) -> str: + """Generate a subheading with dashes.""" + return f"\n--- {title} ---\n" + + +def get_post_meta_header(date: str, series: str) -> str: + """Generate the metadata header for a post.""" + return f"""\ +{SINGLE_LINE} + {date} | {series} +{SINGLE_LINE} +""" + + +def get_title_block(title: str, summary: str = "") -> str: + """Generate a centered title block.""" + # Center the title + centered_title = title.upper().center(LINE_WIDTH) + lines = [centered_title] + if summary: + centered_summary = summary.center(LINE_WIDTH) + lines.append(centered_summary) + return "\n".join(lines) + + +def generate_movie_table( + title: str, + year: int, + director: str = "", + runtime: int = 0, + genres: list = None, + web_url: str = "", +) -> str: + """Generate an ASCII table for movie metadata.""" + genres = genres or [] + width = 65 + border_h = "\u2500" # horizontal line + corner_tl = "\u250c" # top left + corner_tr = "\u2510" # top right + corner_bl = "\u2514" # bottom left + corner_br = "\u2518" # bottom right + border_v = "\u2502" # vertical + tee_l = "\u251c" # left tee + tee_r = "\u2524" # right tee + + def row(content: str) -> str: + return f"{border_v} {content.ljust(width - 4)} {border_v}" + + lines = [ + f"{corner_tl}{border_h * width}{corner_tr}", + row(f"{title.upper()} ({year})"), + f"{tee_l}{border_h * width}{tee_r}", + ] + + if director: + lines.append(row(f"Director: {director}")) + if runtime: + lines.append(row(f"Runtime: {runtime} minutes")) + if genres: + lines.append(row(f"Genres: {', '.join(genres)}")) + + lines.append(row("")) + + if web_url: + lines.append(row(f"View on web: {web_url}")) + + lines.append(f"{corner_bl}{border_h * width}{corner_br}") + + return "\n".join(lines) + + +def wrap_text(text: str, width: int = LINE_WIDTH) -> str: + """Wrap text to specified width, preserving paragraphs.""" + import textwrap + + paragraphs = text.split("\n\n") + wrapped = [] + for para in paragraphs: + # Preserve single newlines within paragraphs as spaces + para = " ".join(para.split()) + if para: + wrapped.append(textwrap.fill(para, width=width)) + else: + wrapped.append("") + return "\n\n".join(wrapped) + + +def format_links_section(links: list) -> str: + """Format a list of links for the footer section.""" + if not links: + return "" + + lines = [get_section_header("LINKS")] + for i, url in enumerate(links, 1): + lines.append(f"[{i}] {url}") + return "\n".join(lines) diff --git a/scripts/gopher/convert_to_gopher.py b/scripts/gopher/convert_to_gopher.py new file mode 100644 index 0000000..d11d2d9 --- /dev/null +++ b/scripts/gopher/convert_to_gopher.py @@ -0,0 +1,350 @@ +#!/usr/bin/env python3 +""" +Convert Hugo markdown posts to gopher-formatted text files. + +Usage: + python scripts/gopher/convert_to_gopher.py content/posts/blog-posting.md + python scripts/gopher/convert_to_gopher.py --all + python scripts/gopher/convert_to_gopher.py --all --output gopher_build/blog/ +""" + +import argparse +import re +import textwrap +import yaml +from pathlib import Path + +from ascii_art import ( + LINE_WIDTH, + POST_HEADER, + POST_FOOTER, + SERIES_TO_DIR, + get_post_meta_header, + get_title_block, + get_section_header, + get_subheading, + generate_movie_table, + format_links_section, +) + +# Paths +SCRIPT_DIR = Path(__file__).parent +PROJECT_ROOT = SCRIPT_DIR.parent.parent +CONTENT_DIR = PROJECT_ROOT / "content" / "posts" +DEFAULT_OUTPUT = PROJECT_ROOT / "gopher_build" / "blog" + + +def parse_frontmatter(content: str) -> tuple[dict, str]: + """Parse YAML frontmatter and return (metadata, body).""" + if not content.startswith("---"): + return {}, content + + # Find the closing --- + end_match = re.search(r"\n---\n", content[3:]) + if not end_match: + return {}, content + + yaml_end = end_match.start() + 3 + yaml_content = content[3:yaml_end] + body = content[yaml_end + 4 :] # Skip the closing ---\n + + try: + metadata = yaml.safe_load(yaml_content) + except yaml.YAMLError: + metadata = {} + + return metadata or {}, body + + +def extract_links(text: str) -> tuple[str, list]: + """Extract markdown links and replace with numbered references.""" + links = [] + link_pattern = re.compile(r"\[([^\]]+)\]\(([^)]+)\)") + + def replace_link(match): + text = match.group(1) + url = match.group(2) + links.append(url) + return f"{text} [{len(links)}]" + + converted = link_pattern.sub(replace_link, text) + return converted, links + + +def convert_headings(text: str) -> str: + """Convert markdown headings to gopher-style text.""" + # H1: Double line with centered text + def h1_replace(match): + title = match.group(1).strip() + return get_section_header(title) + + # H2: Dashed subheading + def h2_replace(match): + title = match.group(1).strip() + return get_subheading(title) + + # H3-H6: Just bold-style text + def h3_replace(match): + title = match.group(1).strip() + return f"\n*{title}*\n" + + text = re.sub(r"^# (.+)$", h1_replace, text, flags=re.MULTILINE) + text = re.sub(r"^## (.+)$", h2_replace, text, flags=re.MULTILINE) + text = re.sub(r"^###+ (.+)$", h3_replace, text, flags=re.MULTILINE) + + return text + + +def convert_formatting(text: str) -> str: + """Convert markdown formatting to gopher-style text.""" + # Bold: **text** -> *text* + text = re.sub(r"\*\*([^*]+)\*\*", r"*\1*", text) + + # Italic: *text* or _text_ -> _text_ + # Be careful not to match our converted bold + text = re.sub(r"(? str: + """Convert code blocks to indented text.""" + # Fenced code blocks + def indent_code(match): + code = match.group(2) + # Indent each line by 4 spaces + indented = "\n".join(" " + line for line in code.split("\n")) + return f"\n{indented}\n" + + text = re.sub(r"```(\w*)\n(.*?)```", indent_code, text, flags=re.DOTALL) + + # Inline code: `code` -> code (just remove backticks) + text = re.sub(r"`([^`]+)`", r"\1", text) + + return text + + +def handle_imdbposter(text: str, metadata: dict, slug: str) -> str: + """Replace imdbposter shortcode with ASCII movie table.""" + # Check if there's an imdbposter shortcode + pattern = r"\{\{<\s*imdbposter\s*>\}\}(.*?)\{\{<\s*/imdbposter\s*>\}\}" + match = re.search(pattern, text, flags=re.DOTALL) + + if not match: + return text + + # Extract movie info from frontmatter + title = metadata.get("title", "Unknown") + year = metadata.get("year", "") + director = metadata.get("director", "") + runtime = metadata.get("runtime", 0) + genres = metadata.get("genres", []) + web_url = f"https://mnw.sdf.org/posts/{slug}/" + + # If year not in frontmatter, try to parse from date + if not year and metadata.get("date"): + date_str = str(metadata.get("date")) + year_match = re.match(r"(\d{4})", date_str) + if year_match: + year = int(year_match.group(1)) + + # Generate the ASCII table + table = generate_movie_table( + title=title, + year=year, + director=director, + runtime=runtime, + genres=genres, + web_url=web_url, + ) + + # Also extract and preserve the viewing info table if present + inner_content = match.group(1).strip() + if inner_content: + # Convert markdown table to plain text + table_lines = [] + for line in inner_content.split("\n"): + line = line.strip() + if line and not line.startswith("|--"): + # Remove leading/trailing pipes and clean up + line = re.sub(r"^\||\|$", "", line) + cells = [c.strip() for c in line.split("|")] + if len(cells) >= 2: + table_lines.append(f" {cells[0]}: {cells[1]}") + if table_lines: + table += "\n\n" + "\n".join(table_lines) + + return text[: match.start()] + table + text[match.end() :] + + +def wrap_paragraphs(text: str) -> str: + """Wrap text to LINE_WIDTH, preserving code blocks and structure.""" + lines = text.split("\n") + result = [] + in_code_block = False + paragraph = [] + + def flush_paragraph(): + if paragraph: + para_text = " ".join(paragraph) + wrapped = textwrap.fill(para_text, width=LINE_WIDTH) + result.append(wrapped) + paragraph.clear() + + for line in lines: + # Detect code blocks (indented by 4 spaces) + if line.startswith(" "): + flush_paragraph() + result.append(line) + continue + + # Empty line = paragraph break + if not line.strip(): + flush_paragraph() + result.append("") + continue + + # Lines that look like headers or dividers - don't wrap + if ( + line.startswith("=") + or line.startswith("-" * 10) + or line.startswith("*") + or line.startswith(" " * 10) + ): + flush_paragraph() + result.append(line) + continue + + # Accumulate paragraph text + paragraph.append(line.strip()) + + flush_paragraph() + return "\n".join(result) + + +def convert_post(filepath: Path, output_dir: Path = None) -> Path | None: + """Convert a single markdown post to gopher format.""" + content = filepath.read_text() + metadata, body = parse_frontmatter(content) + + # Check if phlog is enabled + if not metadata.get("phlog", False): + return None + + # Check if draft + if metadata.get("draft", False): + return None + + # Determine output directory from series + series = metadata.get("series", "Fun Center") + gopher_dir = SERIES_TO_DIR.get(series, "fun-center") + + # Get slug from filename + slug = filepath.stem + + # Output path + output_dir = output_dir or DEFAULT_OUTPUT + category_dir = output_dir / gopher_dir + category_dir.mkdir(parents=True, exist_ok=True) + output_path = category_dir / f"{slug}.txt" + + # Handle imdbposter shortcode first + body = handle_imdbposter(body, metadata, slug) + + # Extract links before other conversions + body, links = extract_links(body) + + # Convert markdown to gopher text + body = convert_headings(body) + body = convert_formatting(body) + body = convert_code_blocks(body) + + # Wrap paragraphs + body = wrap_paragraphs(body) + + # Build the final document + date_str = "" + if metadata.get("date"): + date_obj = metadata["date"] + if hasattr(date_obj, "strftime"): + date_str = date_obj.strftime("%Y-%m-%d") + else: + date_str = str(date_obj)[:10] + + title = metadata.get("title", slug) + summary = metadata.get("summary", "") + + parts = [ + POST_HEADER, + get_post_meta_header(date_str, series), + get_title_block(title, summary), + "", + "-" * LINE_WIDTH, + "", + body.strip(), + "", + ] + + if links: + parts.append(format_links_section(links)) + parts.append("") + + parts.append(POST_FOOTER) + parts.append("") + parts.append(f"Web version: https://mnw.sdf.org/posts/{slug}/") + + output_content = "\n".join(parts) + output_path.write_text(output_content) + + return output_path + + +def convert_all(output_dir: Path = None) -> list[Path]: + """Convert all posts with phlog: true.""" + output_dir = output_dir or DEFAULT_OUTPUT + converted = [] + + for post_path in CONTENT_DIR.glob("*.md"): + result = convert_post(post_path, output_dir) + if result: + converted.append(result) + print(f"Converted: {post_path.name} -> {result.relative_to(output_dir)}") + + return converted + + +def main(): + parser = argparse.ArgumentParser(description="Convert Hugo posts to gopher format") + parser.add_argument("file", nargs="?", help="Single file to convert") + parser.add_argument("--all", action="store_true", help="Convert all phlog posts") + parser.add_argument("--output", "-o", type=Path, help="Output directory") + args = parser.parse_args() + + output = args.output or DEFAULT_OUTPUT + + if args.all: + converted = convert_all(output) + print(f"\nConverted {len(converted)} posts to {output}") + elif args.file: + filepath = Path(args.file) + if not filepath.exists(): + print(f"File not found: {filepath}") + return 1 + result = convert_post(filepath, output) + if result: + print(f"Converted: {result}") + else: + print("Post skipped (no phlog: true or is draft)") + else: + parser.print_help() + return 1 + + return 0 + + +if __name__ == "__main__": + exit(main())