After setting up the full suite of poem formatting, the biggest problem turned out to be finding the poems to markup.

Gwern.net was 15 years old and has quite a lot of poetry here and there, and no single consistent way of writing them. Once I marked up the easy cases found by grep, then what?

The problem with poetry is that it is a semantic category, not syntactic; what makes the quote inside As every kindergartner knows, "Roses are red" something to wrap in a span.poem rather than a simple prose statement of fact? Well, “you know it when you see it”; you know that it’s a quote from a famous poem.

So, this makes it a good use-case for LLMs—LLMs definitely know that’s from a famous poem!

#!/usr/bin/env python3
# -*- coding: utf-8 -*-

"""
poem-finder.py (report-only)

Print un-marked-up poems/verse found in a Pandoc Markdown file.
No modifications are made.

Detects candidates (conservatively):
- strict blockquotes (every line starts with '>')
- fenced code blocks (``` or ~~~)
- indented code blocks (4 spaces or tab)
- HTML <pre> ... </pre> blocks
- inline "slash linebreak" verse (very conservative; optional)

Skips anything already in poem markup:
- <div class="poem"> ... </div>
- :​:​: {.poem} fenced div blocks
- <span class="poem">...</span> (inline)
- Pandoc inline attr spans: {... .poem} / [{...}]{.poem}

Usage:
  $ OPENAI_API_KEY="sk-..." ./poem-finder.py path/to/file.md > poems.txt
  $ ./poem-finder.py --jsonl path/to/file.md | jq .
"""

from __future__ import annotations

import argparse
import hashlib
import json
import os
import re
import sys
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple

from openai import OpenAI


# ----------------------------
# Helpers
# ----------------------------

def sha256_text(s: str) -> str:
    return hashlib.sha256(s.encode("utf-8")).hexdigest()

def eprint(*args: Any, **kwargs: Any) -> None:
    print(*args, file=sys.stderr, **kwargs)

def parse_json_object(s: str) -> Optional[Dict[str, Any]]:
    s = (s or "").strip()
    if s.startswith("```"):
        s = re.sub(r"^```[a-zA-Z0-9_-]*\s*", "", s)
        s = re.sub(r"\s*```$", "", s)
    m = re.search(r"\{.*\}", s, flags=re.DOTALL)
    if not m:
        return None
    try:
        return json.loads(m.group(0))
    except Exception:
        return None


# ----------------------------
# Poem markup region tracking
# ----------------------------

DIV_OPEN_RE = re.compile(r"<div\b", re.IGNORECASE)
DIV_CLOSE_RE = re.compile(r"</div\b", re.IGNORECASE)
DIV_CLASS_RE = re.compile(r"<div\b[^>]*\bclass\s*=\s*(?P<q>[\"'])(?P<cls>.*?)(?P=q)", re.IGNORECASE)

PANDOC_FENCED_DIV_RE = re.compile(r"^([ \t]*)(:{3,})([ \t]*)(?P<attrs>\{.*\})?\s*$")

def class_contains_poem(cls: str) -> bool:
    parts = re.split(r"\s+", (cls or "").strip())
    return any(p.strip().strip(",;") == "poem" for p in parts)

def line_opens_poem_div(line: str) -> bool:
    if "<div" not in line.lower():
        return False
    m = DIV_CLASS_RE.search(line)
    if not m:
        return False
    return class_contains_poem(m.group("cls"))

def update_div_stack(line: str, div_stack: List[bool]) -> None:
    # close tags pop stack
    closes = len(DIV_CLOSE_RE.findall(line))
    for _ in range(closes):
        if div_stack:
            div_stack.pop()
    # open tags push, flagging poem divs
    opens = len(DIV_OPEN_RE.findall(line))
    is_poem = line_opens_poem_div(line)
    for _ in range(opens):
        div_stack.append(is_poem)

def update_fenced_div_stack(line: str, fenced_stack: List[Tuple[int, bool]]) -> None:
    m = PANDOC_FENCED_DIV_RE.match(line.rstrip("\n"))
    if not m:
        return
    fence_len = len(m.group(2))
    attrs = m.group("attrs") or ""

    # closing marker usually has no attrs
    if attrs == "" and fenced_stack:
        # pop most recent matching fence len, else pop last
        idx = None
        for k in range(len(fenced_stack) - 1, -1, -1):
            if fenced_stack[k][0] == fence_len:
                idx = k
                break
        if idx is None:
            fenced_stack.pop()
        else:
            fenced_stack.pop(idx)
        return

    is_poem = ".poem" in attrs
    fenced_stack.append((fence_len, is_poem))

def in_poem_region(div_stack: List[bool], fenced_stack: List[Tuple[int, bool]]) -> bool:
    return any(div_stack) or any(is_poem for _, is_poem in fenced_stack)


# ----------------------------
# Candidate block detection
# ----------------------------

FENCE_START_RE = re.compile(r"^([ \t]*)(?P<fence>`{3,}|~{3,})(?P<info>.*)$")
BLOCKQUOTE_RE = re.compile(r"^[ \t]*>")
PRE_OPEN_RE = re.compile(r"^[ \t]*<pre\b", re.IGNORECASE)
PRE_CLOSE_RE = re.compile(r"</pre\s*>", re.IGNORECASE)

@dataclass(frozen=True)
class Block:
    kind: str               # "blockquote" | "fenced_code" | "indented_code" | "html_pre" | "inline_slash"
    start: int              # 0-based line index (for inline: line index)
    end: int                # inclusive (for inline: same as start)
    raw_lines: List[str]    # original lines (keepends=True), or [line] for inline
    text_for_llm: str       # normalized content
    extra: Dict[str, Any]   # metadata

def capture_fenced_code(lines: List[str], i: int) -> Optional[Tuple[int, List[str]]]:
    m = FENCE_START_RE.match(lines[i].rstrip("\n"))
    if not m:
        return None
    fence = m.group("fence")
    fence_char = fence[0]
    fence_len = len(fence)

    j = i + 1
    end = i
    while j < len(lines):
        s = lines[j].lstrip(" \t")
        if s.startswith(fence_char * fence_len):
            end = j
            break
        j += 1
    else:
        end = len(lines) - 1
    return end, lines[i:end + 1]

def capture_pre(lines: List[str], i: int) -> Optional[Tuple[int, List[str]]]:
    if not PRE_OPEN_RE.match(lines[i]):
        return None
    j = i
    end = i
    while j < len(lines):
        if PRE_CLOSE_RE.search(lines[j]):
            end = j
            break
        j += 1
    else:
        end = len(lines) - 1
    return end, lines[i:end + 1]

def is_indented_code_start(line: str) -> bool:
    if line.strip() == "":
        return False
    if FENCE_START_RE.match(line.rstrip("\n")):
        return False
    if BLOCKQUOTE_RE.match(line):
        return False
    return line.startswith("    ") or line.startswith("\t")

def capture_indented_code(lines: List[str], i: int) -> Tuple[int, List[str]]:
    j = i
    end = i
    while j < len(lines):
        ln = lines[j]
        if ln.strip() == "":
            end = j
            j += 1
            continue
        if ln.startswith("    ") or ln.startswith("\t"):
            end = j
            j += 1
            continue
        break
    return end, lines[i:end + 1]

def capture_blockquote(lines: List[str], i: int) -> Tuple[int, List[str]]:
    j = i
    end = i
    while j < len(lines) and BLOCKQUOTE_RE.match(lines[j]):
        end = j
        j += 1
    return end, lines[i:end + 1]

def normalize_blockquote_text(raw: List[str]) -> str:
    out: List[str] = []
    for ln in raw:
        m = re.match(r"^[ \t]*>[ \t]?(.*)$", ln.rstrip("\n"))
        out.append(m.group(1) if m else ln.rstrip("\n"))
    return "\n".join(out).strip("\n")

def normalize_fenced_code_text(raw: List[str]) -> str:
    if len(raw) < 2:
        return ""
    inner = raw[1:-1]
    return "".join(inner).rstrip("\n")

def normalize_indented_code_text(raw: List[str]) -> str:
    out: List[str] = []
    for ln in raw:
        s = ln.rstrip("\n")
        if s.startswith("\t"):
            out.append(s[1:])
        elif s.startswith("    "):
            out.append(s[4:])
        else:
            out.append(s)
    return "\n".join(out).strip("\n")

def normalize_pre_text(raw: List[str]) -> str:
    joined = "".join(raw)
    # crude stripping of <pre ...> and </pre>
    joined = re.sub(r"^[ \t]*<pre\b[^>]*>\s*", "", joined, flags=re.IGNORECASE)
    joined = re.sub(r"</pre\s*>\s*$", "", joined, flags=re.IGNORECASE)
    return joined.strip("\n")


# ----------------------------
# Inline slash-verse detection (conservative)
# ----------------------------

# Only consider punctuation + " / " patterns to avoid filepaths, URLs, ratios, etc.
INLINE_SLASH_RE = re.compile(r"(?P<body>[A-Za-z][^/\n]{0,120}[,;:]\s*/\s*[^/\n]{0,160}[A-Za-z])")

def line_has_poem_markup(line: str) -> bool:
    if "{.poem" in line: # }
        return True
    if 'class="poem"' in line or "class='poem'" in line:
        return True
    return False

def likely_urlish(line: str) -> bool:
    return ("http://" in line) or ("https://" in line) or ("://" in line) or ("www." in line)

def inline_slash_candidates(line: str) -> List[str]:
    if " / " not in line:
        return []
    if likely_urlish(line):
        return []
    if line_has_poem_markup(line):
        return []
    # skip inline code spans (very rough): if backticks exist, skip completely
    if "`" in line:
        return []
    cands = []
    for m in INLINE_SLASH_RE.finditer(line):
        body = m.group("body").strip()
        # require at least 2 words on each side (conservative)
        parts = [p.strip() for p in body.split("/") if p.strip()]
        if len(parts) != 2:
            continue
        if len(parts[0].split()) < 2 or len(parts[1].split()) < 2:
            continue
        cands.append(body)
    return cands


# ----------------------------
# LLM classification
# ----------------------------

SYSTEM = "You are a careful editor of literary markup for a Pandoc Markdown corpus."

PROMPT = """Decide whether TEXT is a poem/verse (line-breaks or line divisions are part of the content).

Be conservative:
- If uncertain, answer NOT_POEM.

POEM includes: verse, lyrics, hymns, limericks, haiku, epigraph verse, rhymed couplets, any text where line divisions matter.
NOT_POEM includes: prose quotations, dialogue, normal paragraphs, lists, code, logs, configuration, tables, transcripts, citations, math.

KIND: {kind}

TEXT:
<<<
{text}
>>>

Return JSON ONLY with:
- decision: "poem" or "not_poem"
- confidence: integer 0-100
- notes: <= 20 words

No extra keys, no markdown.
"""

def classify(client: OpenAI, model: str, kind: str, text: str, cache: Dict[str, Any], max_chars: int) -> Dict[str, Any]:
    text2 = (text or "").strip("\n")
    if text2 == "":
        return {"decision": "not_poem", "confidence": 0, "notes": "Empty."}
    if len(text2) > max_chars:
        return {"decision": "not_poem", "confidence": 0, "notes": f"Too long ({len(text2)} chars)."}

    key = f"{kind}:{sha256_text(text2)}"
    if key in cache:
        return cache[key]

    completion = client.chat.completions.create(
        model=model,
        temperature=0,
        messages=[
            {"role": "system", "content": SYSTEM},
            {"role": "user", "content": PROMPT.format(kind=kind, text=text2)},
        ],
    )
    raw = completion.choices[0].message.content or ""
    obj = parse_json_object(raw) or {"decision": "not_poem", "confidence": 0, "notes": "Parse failure."}

    # normalize
    decision = str(obj.get("decision", "not_poem")).strip().lower()
    conf = int(obj.get("confidence", 0) or 0)
    notes = str(obj.get("notes", "")).strip()
    out = {"decision": "poem" if decision == "poem" else "not_poem", "confidence": conf, "notes": notes}
    cache[key] = out
    return out


# ----------------------------
# Main scanning
# ----------------------------

def scan_blocks(lines: List[str], do_inline: bool) -> List[Block]:
    blocks: List[Block] = []

    div_stack: List[bool] = []
    fenced_stack: List[Tuple[int, bool]] = []

    i = 0
    while i < len(lines):
        line = lines[i]

        # Code/pre blocks are opaque: do not update poem-region state based on their contents.
        fc = capture_fenced_code(lines, i)
        if fc is not None:
            end, raw = fc
            if not in_poem_region(div_stack, fenced_stack):
                text = normalize_fenced_code_text(raw)
                blocks.append(Block(kind="fenced_code", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
            i = end + 1
            continue

        pb = capture_pre(lines, i)
        if pb is not None:
            end, raw = pb
            if not in_poem_region(div_stack, fenced_stack):
                text = normalize_pre_text(raw)
                blocks.append(Block(kind="html_pre", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
            i = end + 1
            continue

        if is_indented_code_start(line):
            end, raw = capture_indented_code(lines, i)
            if not in_poem_region(div_stack, fenced_stack):
                text = normalize_indented_code_text(raw)
                blocks.append(Block(kind="indented_code", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
            i = end + 1
            continue

        # Blockquote block: treat as a unit, but still update region stacks line-by-line as we consume it.
        if BLOCKQUOTE_RE.match(line):
            poem_region_now = in_poem_region(div_stack, fenced_stack)
            end, raw = capture_blockquote(lines, i)
            if not poem_region_now:
                # skip if it already contains poem markup (rare but cheap check)
                if not any(line_has_poem_markup(ln) for ln in raw):
                    text = normalize_blockquote_text(raw)
                    blocks.append(Block(kind="blockquote", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
            # update stacks over the consumed lines
            for j in range(i, end + 1):
                update_fenced_div_stack(lines[j], fenced_stack)
                update_div_stack(lines[j], div_stack)
            i = end + 1
            continue

        # Normal line: update region tracking.
        update_fenced_div_stack(line, fenced_stack)
        update_div_stack(line, div_stack)

        # Inline scan (only outside poem regions, and only if enabled).
        if do_inline and not in_poem_region(div_stack, fenced_stack):
            if not line_has_poem_markup(line):
                cands = inline_slash_candidates(line.rstrip("\n"))
                for body in cands:
                    blocks.append(Block(
                        kind="inline_slash",
                        start=i,
                        end=i,
                        raw_lines=[line],
                        text_for_llm=body,
                        extra={"line": line.rstrip("\n"), "match": body},
                    ))

        i += 1

    return blocks


def load_cache(path: Optional[Path]) -> Dict[str, Any]:
    if path is None:
        return {}
    try:
        return json.loads(path.read_text(encoding="utf-8"))
    except FileNotFoundError:
        return {}
    except Exception:
        return {}

def save_cache(path: Optional[Path], cache: Dict[str, Any]) -> None:
    if path is None:
        return
    path.parent.mkdir(parents=True, exist_ok=True)
    tmp = path.with_suffix(path.suffix + ".tmp")
    tmp.write_text(json.dumps(cache, indent=2, ensure_ascii=False, sort_keys=True), encoding="utf-8")
    tmp.replace(path)


def main(argv: Optional[List[str]] = None) -> int:
    ap = argparse.ArgumentParser(description="Print unmarked poems/verse in a Markdown file (report-only).")
    ap.add_argument("path", help="Input Markdown file.")
    ap.add_argument("--model", default="gpt-4.1-mini", help="OpenAI model name.")
    ap.add_argument("--threshold", type=int, default=75, help="Min confidence required to report as poem.")
    ap.add_argument("--max-chars", type=int, default=5000, help="Max chars per candidate sent to LLM.")
    ap.add_argument("--inline", action="store_true", help="Also scan inline slash-verse candidates.")
    ap.add_argument("--jsonl", action="store_true", help="Output JSONL objects instead of text report.")
    ap.add_argument("--cache", default=str(Path.home() / ".cache" / "poem-finder-report-cache.json"), help="Cache path.")
    ap.add_argument("--no-cache", action="store_true", help="Disable cache read/write.")
    ap.add_argument("--show-notes", action="store_true", help="Include LLM notes in text output.")
    ap.add_argument("--limit", type=int, default=0, help="Stop after reporting N poems (0 = no limit).")
    args = ap.parse_args(argv)

    path = Path(args.path)
    text = path.read_text(encoding="utf-8")
    lines = text.splitlines(keepends=True)

    cache_path = None if args.no_cache else Path(args.cache)
    cache = load_cache(cache_path)

    client = OpenAI()

    candidates = scan_blocks(lines, do_inline=args.inline)

    reported = 0
    for blk in candidates:
        # extra “already marked” guard
        if blk.kind != "inline_slash":
            if any(line_has_poem_markup(ln) for ln in blk.raw_lines):
                continue

        result = classify(client, args.model, blk.kind, blk.text_for_llm, cache, args.max_chars)
        if result["decision"] != "poem" or int(result["confidence"]) < args.threshold:
            continue

        reported += 1
        if args.jsonl:
            obj = {
                "file": str(path),
                "kind": blk.kind,
                "start_line": blk.start + 1,
                "end_line": blk.end + 1,
                "confidence": int(result["confidence"]),
                "notes": result["notes"],
                "text": blk.text_for_llm,
                "raw": "".join(blk.raw_lines),
            }
            print(json.dumps(obj, ensure_ascii=False))
        else:
            print("-" * 80)
            print(f"{path}:{blk.start+1}-{blk.end+1}  kind={blk.kind}  confidence={int(result['confidence'])}")
            if args.show_notes and result.get("notes"):
                print(f"notes: {result['notes']}")
            print()
            print(blk.text_for_llm.strip("\n"))
            print()
            print("RAW:")
            sys.stdout.write("".join(blk.raw_lines))
            if not blk.raw_lines[-1].endswith("\n"):
                print()

        if args.limit and reported >= args.limit:
            break

    save_cache(cache_path, cache)

    if not args.jsonl:
        eprint(f"[poem-finder] reported {reported} poems (threshold={args.threshold}, inline={args.inline})")

    return 0


if __name__ == "__main__":
    raise SystemExit(main())

After some tuning and pilot runs, and one or two tweaks I forget, the script found a good 20–50 remaining instances, which I then fixed. I have not noticed many, if any, unmarked poems since. I don’t know how much it cost to run the script a few times on the site corpus, but I didn’t notice it on my OA API bill, so I assume it cost <$5; well worth the money!