#!/usr/bin/env python3
"""
Convert-MarkdownToKdpHtml.py - build one clean, Kindle-friendly HTML file from Markdown chapters.

Give it a folder of chapter files (one .md per chapter, sorted naturally, so "Chapter 2" comes before
"Chapter 10") or a single .md file with a heading at the start of each chapter. It writes one HTML file
with a title page, an optional copyright page, a linked table of contents, and each chapter starting on a
new page. That file uploads straight to Kindle Direct Publishing, opens in Kindle Previewer, and converts
cleanly with Calibre or Pandoc if you need an EPUB.

Typesetting it does for you:
  * straight quotes become curly quotes, -- becomes an em dash, ... becomes an ellipsis
  * *italic*, _italic_ and **bold** become <em> and <strong>
  * a line that's just ***, * * *, ---, # or ~~~ becomes a centered scene break
  * the first paragraph of each chapter, and the first after a scene break, isn't indented
  * YAML front matter and <!-- comments --> are dropped, so notes to yourself stay out of the book

Chapter titles come from the first heading in each file. The "## Chapter 3" followed by "### The Storm"
pattern is understood too: the label is dropped (the script numbers chapters itself) and the second heading
becomes the title. Files with no heading get a title from the file name.

Examples:
  python3 Convert-MarkdownToKdpHtml.py ./chapters --title "The Lighthouse Keeper" --author "A. Writer"
  python3 Convert-MarkdownToKdpHtml.py book.md --title "The Lighthouse Keeper" --author "A. Writer" --series "The Coastline Novels" --series-number 1 --copyright-year 2026 --dry-run

Only the Python 3.8+ standard library is needed.
"""

import argparse
import html
import re
import sys
from pathlib import Path

SCENE_BREAK = re.compile(r'^\s*(\*\s*\*\s*\*[\s*]*|-{3,}|_{3,}|~{3,}|#)\s*$')
HEADING = re.compile(r'^(#{1,6})\s+(.*?)\s*#*\s*$')
CHAPTER_LABEL = re.compile(r'^(chapter|part|book)\s+([0-9]+|[ivxlcdm]+|[a-z-]+)\.?$', re.IGNORECASE)


def natural_key(path):
    """Sort key that orders embedded numbers numerically."""
    return [int(t) if t.isdigit() else t.lower() for t in re.split(r'(\d+)', path.name)]


def title_from_filename(path):
    """'Chapter-03-the-storm.md' -> 'The Storm'."""
    stem = re.sub(r'^(chapter|ch)?[\s_-]*\d+[\s_.-]*', '', path.stem, flags=re.IGNORECASE)
    stem = re.sub(r'[_-]+', ' ', stem).strip()
    return stem.title() if stem else ''


def strip_noise(text):
    """Remove a BOM, YAML front matter and HTML comments."""
    text = text.lstrip('\ufeff').replace('\r\n', '\n')
    text = re.sub(r'\A---\n.*?\n(---|\.\.\.)\n', '', text, flags=re.DOTALL)
    return re.sub(r'<!--.*?-->', '', text, flags=re.DOTALL)


def smarten(text):
    """Curly quotes, em dashes and ellipses. Runs on escaped text, before any tags exist."""
    text = text.replace('---', '\u2014').replace('--', '\u2014')
    text = re.sub(r'\.\s?\.\s?\.', '\u2026', text)
    # Apostrophes in elisions like '90s and 'til, then opening and closing quotes.
    text = re.sub(r"'(?=\d{2}s\b|til\b|em\b|cause\b|n\b)", '\u2019', text, flags=re.IGNORECASE)
    text = re.sub(r'(^|[\s(\[{\u2014\u2013-])"', '\\1\u201c', text)
    text = text.replace('"', '\u201d')
    text = re.sub(r"(^|[\s(\[{\u2014\u2013-]|\u201c)'", '\\1\u2018', text)
    return text.replace("'", '\u2019')


def inline(text):
    """Escape, smarten, then turn Markdown emphasis into tags."""
    text = html.escape(text, quote=False)
    text = smarten(text)
    text = re.sub(r'\*\*(.+?)\*\*', r'<strong>\1</strong>', text)
    text = re.sub(r'__(.+?)__', r'<strong>\1</strong>', text)
    text = re.sub(r'(?<![\w*])\*(?!\s)(.+?)(?<!\s)\*(?![\w*])', r'<em>\1</em>', text)
    text = re.sub(r'(?<![\w_])_(?!\s)(.+?)(?<!\s)_(?![\w_])', r'<em>\1</em>', text)
    return text


def split_blocks(text):
    """Split text into paragraphs on blank lines. Scene-break lines always stand alone."""
    blocks, current = [], []
    for line in text.split('\n'):
        if not line.strip() or SCENE_BREAK.match(line) or HEADING.match(line):
            if current:
                blocks.append(' '.join(s.strip() for s in current))
                current = []
            if line.strip():
                blocks.append(line.strip())
        else:
            current.append(line)
    if current:
        blocks.append(' '.join(s.strip() for s in current))
    return blocks


def parse_chapter(text, fallback_title):
    """Return (title, blocks) for one chapter's Markdown."""
    blocks = split_blocks(strip_noise(text))
    title = None
    # Consume the leading heading(s): "Chapter 3" labels are dropped, the first real heading is the title.
    while blocks and HEADING.match(blocks[0]):
        heading = HEADING.match(blocks[0]).group(2).strip()
        blocks.pop(0)
        label = re.match(r'^(chapter|part)\s+\S+\s*[:.\u2014-]\s*(.+)$', heading, re.IGNORECASE)
        if label:
            title = label.group(2)
            break
        if CHAPTER_LABEL.match(heading) and title is None:
            continue
        title = heading
        break
    return (title or fallback_title), blocks


def split_single_file(text, level):
    """Split one Markdown file into chapters at headings of the given level (1 = '#')."""
    marker = re.compile(r'^#{%d}\s+\S' % level)
    chapters, current = [], []
    for line in strip_noise(text).split('\n'):
        if marker.match(line) and current and any(l.strip() for l in current):
            chapters.append('\n'.join(current))
            current = []
        current.append(line)
    if any(l.strip() for l in current):
        chapters.append('\n'.join(current))
    return chapters


def render_chapter(index, title, blocks, numbered, scene_glyph):
    """Render one chapter as HTML. Returns (html, words, scene_breaks)."""
    anchor = f'ch{index:02d}'
    out = [f'<section class="chapter" id="{anchor}">']
    if numbered:
        out.append(f'<h1 class="chapter-title"><span class="chapter-number">Chapter {index}</span>'
                   + (f'<br/>{inline(title)}' if title else '') + '</h1>')
    else:
        out.append(f'<h1 class="chapter-title">{inline(title)}</h1>')
    words, breaks, first = 0, 0, True
    for block in blocks:
        if SCENE_BREAK.match(block):
            breaks += 1
            out.append(f'<p class="scene-break">{html.escape(scene_glyph)}</p>')
            first = True
            continue
        heading = HEADING.match(block)
        if heading:
            out.append(f'<h2 class="section-title">{inline(heading.group(2))}</h2>')
            first = True
            continue
        words += len([w for w in block.split() if re.search(r'\w', w)])
        out.append(f'<p class="first">{inline(block)}</p>' if first else f'<p>{inline(block)}</p>')
        first = False
    out.append('</section>')
    return '\n'.join(out), words, breaks


CSS = """
body { font-family: Georgia, "Times New Roman", serif; line-height: 1.5; margin: 0 5%; }
p { margin: 0; text-indent: 1.5em; text-align: justify; }
p.first, p.scene-break, .title-page p, .copyright p, .toc p { text-indent: 0; }
p.scene-break { text-align: center; margin: 1em 0; }
h1, h2 { text-align: center; font-weight: normal; page-break-after: avoid; }
h1.chapter-title { page-break-before: always; margin: 3em 0 1.5em; font-size: 1.5em; }
.chapter-number { font-size: 0.75em; letter-spacing: 0.1em; text-transform: uppercase; }
h2.section-title { font-size: 1.1em; margin: 1.5em 0 1em; }
.title-page { text-align: center; page-break-after: always; }
.title-page .book-title { font-size: 2em; margin-top: 30%; }
.title-page .subtitle { font-style: italic; margin-top: 0.5em; }
.title-page .series { margin-top: 2em; }
.title-page .author { font-size: 1.3em; margin-top: 3em; }
.copyright { page-break-before: always; font-size: 0.85em; margin-top: 30%; }
.copyright p { margin-bottom: 0.8em; text-align: left; }
.toc { page-break-before: always; }
.toc p { margin: 0.4em 0; text-align: left; }
""".strip()


def build_html(args, chapters, back_matter):
    """Assemble the full document. Returns (html, stats)."""
    esc = lambda s: html.escape(s or '', quote=True)
    parts = [
        '<!DOCTYPE html>',
        f'<html lang="{esc(args.lang)}">',
        '<head>',
        '<meta charset="utf-8"/>',
        f'<title>{esc(args.title)}</title>',
        f'<meta name="author" content="{esc(args.author)}"/>',
        f'<style>\n{CSS}\n</style>',
        '</head>',
        '<body>',
        '<section class="title-page">',
        f'<p class="book-title">{inline(args.title)}</p>',
    ]
    if args.subtitle:
        parts.append(f'<p class="subtitle">{inline(args.subtitle)}</p>')
    if args.series:
        series_line = inline(args.series) + (f', Book {esc(args.series_number)}' if args.series_number else '')
        parts.append(f'<p class="series">{series_line}</p>')
    parts.append(f'<p class="author">{inline(args.author)}</p>')
    parts.append('</section>')

    if args.copyright_year:
        holder = args.copyright_holder or args.author
        parts += [
            '<section class="copyright">',
            f'<p>Copyright \u00a9 {esc(args.copyright_year)} {inline(holder)}. All rights reserved.</p>',
            '<p>This is a work of fiction. Names, characters, places and incidents are products of the '
            'author\u2019s imagination or are used fictitiously.</p>',
            '</section>',
        ]

    body, toc, total_words, total_breaks = [], [], 0, 0
    for i, (title, blocks) in enumerate(chapters, start=1):
        chapter_html, words, breaks = render_chapter(i, title, blocks, not args.no_chapter_numbers, args.scene_break)
        body.append(chapter_html)
        label = f'Chapter {i}' + (f': {inline(title)}' if title else '') if not args.no_chapter_numbers else inline(title)
        toc.append(f'<p><a href="#ch{i:02d}">{label}</a></p>')
        total_words += words
        total_breaks += breaks

    if back_matter:
        title, blocks = back_matter
        chapter_html, words, breaks = render_chapter(0, title, blocks, False, args.scene_break)
        body.append(chapter_html.replace('id="ch00"', 'id="back-matter"', 1))
        toc.append(f'<p><a href="#back-matter">{inline(title)}</a></p>')
        total_words += words

    if not args.no_toc:
        parts += ['<section class="toc" id="toc">', '<h1>Contents</h1>'] + toc + ['</section>']
    parts += body + ['</body>', '</html>', '']
    return '\n'.join(parts), {'chapters': len(chapters), 'words': total_words, 'scene_breaks': total_breaks}


def parse_args(argv=None):
    p = argparse.ArgumentParser(description='Build a single Kindle-friendly HTML file from Markdown chapters.')
    p.add_argument('source', type=Path, help='A folder of chapter .md files, or one .md file with a heading per chapter')
    p.add_argument('-o', '--output', type=Path, help='Output .html file (default: <title>.html next to the source)')
    p.add_argument('--title', required=True, help='Book title')
    p.add_argument('--author', required=True, help='Author name as it should appear on the title page')
    p.add_argument('--subtitle', help='Optional subtitle')
    p.add_argument('--series', help='Series name, shown on the title page')
    p.add_argument('--series-number', help='Position in the series, e.g. 2')
    p.add_argument('--copyright-year', help='Adds a copyright page with this year')
    p.add_argument('--copyright-holder', help='Copyright holder if not the author')
    p.add_argument('--back-matter', type=Path, help='Optional .md file (acknowledgments, about the author) added at the end')
    p.add_argument('--pattern', default='*.md', help='File pattern inside a source folder (default: *.md)')
    p.add_argument('--split-level', type=int, default=1, choices=range(1, 4), help='Heading level that starts a chapter in a single-file source (default: 1)')
    p.add_argument('--scene-break', default='* * *', help='What a scene break shows as (default: "* * *")')
    p.add_argument('--no-chapter-numbers', action='store_true', help='Use chapter titles only, no "Chapter N"')
    p.add_argument('--no-toc', action='store_true', help='Leave out the table of contents')
    p.add_argument('--lang', default='en', help='Language code for the <html> tag (default: en)')
    p.add_argument('--dry-run', action='store_true', help='List the chapters and totals without writing anything')
    p.add_argument('--force', action='store_true', help='Overwrite the output file if it exists')
    return p.parse_args(argv)


def main(argv=None):
    args = parse_args(argv)
    source = args.source.expanduser()

    raw = []  # (fallback title, markdown text)
    if source.is_dir():
        files = sorted((f for f in source.glob(args.pattern) if f.is_file()), key=natural_key)
        if args.back_matter:
            files = [f for f in files if f.resolve() != args.back_matter.resolve()]
        for f in files:
            raw.append((title_from_filename(f), f.read_text(encoding='utf-8-sig')))
    elif source.is_file():
        for chunk in split_single_file(source.read_text(encoding='utf-8-sig'), args.split_level):
            raw.append(('', chunk))
    else:
        sys.exit(f'Not found: {source}')
    if not raw:
        sys.exit(f'No chapter files matching {args.pattern} in {source}')

    chapters = [parse_chapter(text, fallback) for fallback, text in raw]
    chapters = [(t, b) for t, b in chapters if b or t]
    back = None
    if args.back_matter:
        back = parse_chapter(args.back_matter.read_text(encoding='utf-8-sig'), title_from_filename(args.back_matter))

    document, stats = build_html(args, chapters, back)

    for i, (title, blocks) in enumerate(chapters, start=1):
        print(f'  Chapter {i:>2}: {title or "(untitled)"} ({len([b for b in blocks if not SCENE_BREAK.match(b)])} paragraphs)')
    print(f'{stats["chapters"]} chapters, {stats["words"]:,} words, {stats["scene_breaks"]} scene breaks')

    out = args.output or (source if source.is_dir() else source.parent) / (re.sub(r'[\\/:*?"<>|]+', '', args.title).strip() + '.html')
    if args.dry_run:
        print(f'Dry run: would write {out}')
        return 0
    if out.exists() and not args.force:
        sys.exit(f'{out} already exists. Use --force to overwrite it.')
    out.write_text(document, encoding='utf-8')
    print(f'Wrote {out}')
    return 0


if __name__ == '__main__':
    sys.exit(main())

