After setting up the full suite of poem formatting, the biggest problem turned out to be finding the poems to markup.
Gwern.net was 15 years old and has quite a lot of poetry here and there, and no single consistent way of writing them. Once I marked up the easy cases found by grep, then what?
The problem with poetry is that it is a semantic category, not syntactic; what makes the quote inside As every kindergartner knows, "Roses are red" something to wrap in a span.poem rather than a simple prose statement of fact? Well, “you know it when you see it”; you know that it’s a quote from a famous poem.
So, this makes it a good use-case for LLMs—LLMs definitely know that’s from a famous poem!
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""
poem-finder.py (report-only)
Print un-marked-up poems/verse found in a Pandoc Markdown file.
No modifications are made.
Detects candidates (conservatively):
- strict blockquotes (every line starts with '>')
- fenced code blocks (``` or ~~~)
- indented code blocks (4 spaces or tab)
- HTML <pre> ... </pre> blocks
- inline "slash linebreak" verse (very conservative; optional)
Skips anything already in poem markup:
- <div class="poem"> ... </div>
- ::: {.poem} fenced div blocks
- <span class="poem">...</span> (inline)
- Pandoc inline attr spans: {... .poem} / [{...}]{.poem}
Usage:
$ OPENAI_API_KEY="sk-..." ./poem-finder.py path/to/file.md > poems.txt
$ ./poem-finder.py --jsonl path/to/file.md | jq .
"""
from __future__ import annotations
import argparse
import hashlib
import json
import os
import re
import sys
from dataclasses import dataclass
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from openai import OpenAI
# ----------------------------
# Helpers
# ----------------------------
def sha256_text(s: str) -> str:
return hashlib.sha256(s.encode("utf-8")).hexdigest()
def eprint(*args: Any, **kwargs: Any) -> None:
print(*args, file=sys.stderr, **kwargs)
def parse_json_object(s: str) -> Optional[Dict[str, Any]]:
s = (s or "").strip()
if s.startswith("```"):
s = re.sub(r"^```[a-zA-Z0-9_-]*\s*", "", s)
s = re.sub(r"\s*```$", "", s)
m = re.search(r"\{.*\}", s, flags=re.DOTALL)
if not m:
return None
try:
return json.loads(m.group(0))
except Exception:
return None
# ----------------------------
# Poem markup region tracking
# ----------------------------
DIV_OPEN_RE = re.compile(r"<div\b", re.IGNORECASE)
DIV_CLOSE_RE = re.compile(r"</div\b", re.IGNORECASE)
DIV_CLASS_RE = re.compile(r"<div\b[^>]*\bclass\s*=\s*(?P<q>[\"'])(?P<cls>.*?)(?P=q)", re.IGNORECASE)
PANDOC_FENCED_DIV_RE = re.compile(r"^([ \t]*)(:{3,})([ \t]*)(?P<attrs>\{.*\})?\s*$")
def class_contains_poem(cls: str) -> bool:
parts = re.split(r"\s+", (cls or "").strip())
return any(p.strip().strip(",;") == "poem" for p in parts)
def line_opens_poem_div(line: str) -> bool:
if "<div" not in line.lower():
return False
m = DIV_CLASS_RE.search(line)
if not m:
return False
return class_contains_poem(m.group("cls"))
def update_div_stack(line: str, div_stack: List[bool]) -> None:
# close tags pop stack
closes = len(DIV_CLOSE_RE.findall(line))
for _ in range(closes):
if div_stack:
div_stack.pop()
# open tags push, flagging poem divs
opens = len(DIV_OPEN_RE.findall(line))
is_poem = line_opens_poem_div(line)
for _ in range(opens):
div_stack.append(is_poem)
def update_fenced_div_stack(line: str, fenced_stack: List[Tuple[int, bool]]) -> None:
m = PANDOC_FENCED_DIV_RE.match(line.rstrip("\n"))
if not m:
return
fence_len = len(m.group(2))
attrs = m.group("attrs") or ""
# closing marker usually has no attrs
if attrs == "" and fenced_stack:
# pop most recent matching fence len, else pop last
idx = None
for k in range(len(fenced_stack) - 1, -1, -1):
if fenced_stack[k][0] == fence_len:
idx = k
break
if idx is None:
fenced_stack.pop()
else:
fenced_stack.pop(idx)
return
is_poem = ".poem" in attrs
fenced_stack.append((fence_len, is_poem))
def in_poem_region(div_stack: List[bool], fenced_stack: List[Tuple[int, bool]]) -> bool:
return any(div_stack) or any(is_poem for _, is_poem in fenced_stack)
# ----------------------------
# Candidate block detection
# ----------------------------
FENCE_START_RE = re.compile(r"^([ \t]*)(?P<fence>`{3,}|~{3,})(?P<info>.*)$")
BLOCKQUOTE_RE = re.compile(r"^[ \t]*>")
PRE_OPEN_RE = re.compile(r"^[ \t]*<pre\b", re.IGNORECASE)
PRE_CLOSE_RE = re.compile(r"</pre\s*>", re.IGNORECASE)
@dataclass(frozen=True)
class Block:
kind: str # "blockquote" | "fenced_code" | "indented_code" | "html_pre" | "inline_slash"
start: int # 0-based line index (for inline: line index)
end: int # inclusive (for inline: same as start)
raw_lines: List[str] # original lines (keepends=True), or [line] for inline
text_for_llm: str # normalized content
extra: Dict[str, Any] # metadata
def capture_fenced_code(lines: List[str], i: int) -> Optional[Tuple[int, List[str]]]:
m = FENCE_START_RE.match(lines[i].rstrip("\n"))
if not m:
return None
fence = m.group("fence")
fence_char = fence[0]
fence_len = len(fence)
j = i + 1
end = i
while j < len(lines):
s = lines[j].lstrip(" \t")
if s.startswith(fence_char * fence_len):
end = j
break
j += 1
else:
end = len(lines) - 1
return end, lines[i:end + 1]
def capture_pre(lines: List[str], i: int) -> Optional[Tuple[int, List[str]]]:
if not PRE_OPEN_RE.match(lines[i]):
return None
j = i
end = i
while j < len(lines):
if PRE_CLOSE_RE.search(lines[j]):
end = j
break
j += 1
else:
end = len(lines) - 1
return end, lines[i:end + 1]
def is_indented_code_start(line: str) -> bool:
if line.strip() == "":
return False
if FENCE_START_RE.match(line.rstrip("\n")):
return False
if BLOCKQUOTE_RE.match(line):
return False
return line.startswith(" ") or line.startswith("\t")
def capture_indented_code(lines: List[str], i: int) -> Tuple[int, List[str]]:
j = i
end = i
while j < len(lines):
ln = lines[j]
if ln.strip() == "":
end = j
j += 1
continue
if ln.startswith(" ") or ln.startswith("\t"):
end = j
j += 1
continue
break
return end, lines[i:end + 1]
def capture_blockquote(lines: List[str], i: int) -> Tuple[int, List[str]]:
j = i
end = i
while j < len(lines) and BLOCKQUOTE_RE.match(lines[j]):
end = j
j += 1
return end, lines[i:end + 1]
def normalize_blockquote_text(raw: List[str]) -> str:
out: List[str] = []
for ln in raw:
m = re.match(r"^[ \t]*>[ \t]?(.*)$", ln.rstrip("\n"))
out.append(m.group(1) if m else ln.rstrip("\n"))
return "\n".join(out).strip("\n")
def normalize_fenced_code_text(raw: List[str]) -> str:
if len(raw) < 2:
return ""
inner = raw[1:-1]
return "".join(inner).rstrip("\n")
def normalize_indented_code_text(raw: List[str]) -> str:
out: List[str] = []
for ln in raw:
s = ln.rstrip("\n")
if s.startswith("\t"):
out.append(s[1:])
elif s.startswith(" "):
out.append(s[4:])
else:
out.append(s)
return "\n".join(out).strip("\n")
def normalize_pre_text(raw: List[str]) -> str:
joined = "".join(raw)
# crude stripping of <pre ...> and </pre>
joined = re.sub(r"^[ \t]*<pre\b[^>]*>\s*", "", joined, flags=re.IGNORECASE)
joined = re.sub(r"</pre\s*>\s*$", "", joined, flags=re.IGNORECASE)
return joined.strip("\n")
# ----------------------------
# Inline slash-verse detection (conservative)
# ----------------------------
# Only consider punctuation + " / " patterns to avoid filepaths, URLs, ratios, etc.
INLINE_SLASH_RE = re.compile(r"(?P<body>[A-Za-z][^/\n]{0,120}[,;:]\s*/\s*[^/\n]{0,160}[A-Za-z])")
def line_has_poem_markup(line: str) -> bool:
if "{.poem" in line: # }
return True
if 'class="poem"' in line or "class='poem'" in line:
return True
return False
def likely_urlish(line: str) -> bool:
return ("http://" in line) or ("https://" in line) or ("://" in line) or ("www." in line)
def inline_slash_candidates(line: str) -> List[str]:
if " / " not in line:
return []
if likely_urlish(line):
return []
if line_has_poem_markup(line):
return []
# skip inline code spans (very rough): if backticks exist, skip completely
if "`" in line:
return []
cands = []
for m in INLINE_SLASH_RE.finditer(line):
body = m.group("body").strip()
# require at least 2 words on each side (conservative)
parts = [p.strip() for p in body.split("/") if p.strip()]
if len(parts) != 2:
continue
if len(parts[0].split()) < 2 or len(parts[1].split()) < 2:
continue
cands.append(body)
return cands
# ----------------------------
# LLM classification
# ----------------------------
SYSTEM = "You are a careful editor of literary markup for a Pandoc Markdown corpus."
PROMPT = """Decide whether TEXT is a poem/verse (line-breaks or line divisions are part of the content).
Be conservative:
- If uncertain, answer NOT_POEM.
POEM includes: verse, lyrics, hymns, limericks, haiku, epigraph verse, rhymed couplets, any text where line divisions matter.
NOT_POEM includes: prose quotations, dialogue, normal paragraphs, lists, code, logs, configuration, tables, transcripts, citations, math.
KIND: {kind}
TEXT:
<<<
{text}
>>>
Return JSON ONLY with:
- decision: "poem" or "not_poem"
- confidence: integer 0-100
- notes: <= 20 words
No extra keys, no markdown.
"""
def classify(client: OpenAI, model: str, kind: str, text: str, cache: Dict[str, Any], max_chars: int) -> Dict[str, Any]:
text2 = (text or "").strip("\n")
if text2 == "":
return {"decision": "not_poem", "confidence": 0, "notes": "Empty."}
if len(text2) > max_chars:
return {"decision": "not_poem", "confidence": 0, "notes": f"Too long ({len(text2)} chars)."}
key = f"{kind}:{sha256_text(text2)}"
if key in cache:
return cache[key]
completion = client.chat.completions.create(
model=model,
temperature=0,
messages=[
{"role": "system", "content": SYSTEM},
{"role": "user", "content": PROMPT.format(kind=kind, text=text2)},
],
)
raw = completion.choices[0].message.content or ""
obj = parse_json_object(raw) or {"decision": "not_poem", "confidence": 0, "notes": "Parse failure."}
# normalize
decision = str(obj.get("decision", "not_poem")).strip().lower()
conf = int(obj.get("confidence", 0) or 0)
notes = str(obj.get("notes", "")).strip()
out = {"decision": "poem" if decision == "poem" else "not_poem", "confidence": conf, "notes": notes}
cache[key] = out
return out
# ----------------------------
# Main scanning
# ----------------------------
def scan_blocks(lines: List[str], do_inline: bool) -> List[Block]:
blocks: List[Block] = []
div_stack: List[bool] = []
fenced_stack: List[Tuple[int, bool]] = []
i = 0
while i < len(lines):
line = lines[i]
# Code/pre blocks are opaque: do not update poem-region state based on their contents.
fc = capture_fenced_code(lines, i)
if fc is not None:
end, raw = fc
if not in_poem_region(div_stack, fenced_stack):
text = normalize_fenced_code_text(raw)
blocks.append(Block(kind="fenced_code", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
i = end + 1
continue
pb = capture_pre(lines, i)
if pb is not None:
end, raw = pb
if not in_poem_region(div_stack, fenced_stack):
text = normalize_pre_text(raw)
blocks.append(Block(kind="html_pre", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
i = end + 1
continue
if is_indented_code_start(line):
end, raw = capture_indented_code(lines, i)
if not in_poem_region(div_stack, fenced_stack):
text = normalize_indented_code_text(raw)
blocks.append(Block(kind="indented_code", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
i = end + 1
continue
# Blockquote block: treat as a unit, but still update region stacks line-by-line as we consume it.
if BLOCKQUOTE_RE.match(line):
poem_region_now = in_poem_region(div_stack, fenced_stack)
end, raw = capture_blockquote(lines, i)
if not poem_region_now:
# skip if it already contains poem markup (rare but cheap check)
if not any(line_has_poem_markup(ln) for ln in raw):
text = normalize_blockquote_text(raw)
blocks.append(Block(kind="blockquote", start=i, end=end, raw_lines=raw, text_for_llm=text, extra={}))
# update stacks over the consumed lines
for j in range(i, end + 1):
update_fenced_div_stack(lines[j], fenced_stack)
update_div_stack(lines[j], div_stack)
i = end + 1
continue
# Normal line: update region tracking.
update_fenced_div_stack(line, fenced_stack)
update_div_stack(line, div_stack)
# Inline scan (only outside poem regions, and only if enabled).
if do_inline and not in_poem_region(div_stack, fenced_stack):
if not line_has_poem_markup(line):
cands = inline_slash_candidates(line.rstrip("\n"))
for body in cands:
blocks.append(Block(
kind="inline_slash",
start=i,
end=i,
raw_lines=[line],
text_for_llm=body,
extra={"line": line.rstrip("\n"), "match": body},
))
i += 1
return blocks
def load_cache(path: Optional[Path]) -> Dict[str, Any]:
if path is None:
return {}
try:
return json.loads(path.read_text(encoding="utf-8"))
except FileNotFoundError:
return {}
except Exception:
return {}
def save_cache(path: Optional[Path], cache: Dict[str, Any]) -> None:
if path is None:
return
path.parent.mkdir(parents=True, exist_ok=True)
tmp = path.with_suffix(path.suffix + ".tmp")
tmp.write_text(json.dumps(cache, indent=2, ensure_ascii=False, sort_keys=True), encoding="utf-8")
tmp.replace(path)
def main(argv: Optional[List[str]] = None) -> int:
ap = argparse.ArgumentParser(description="Print unmarked poems/verse in a Markdown file (report-only).")
ap.add_argument("path", help="Input Markdown file.")
ap.add_argument("--model", default="gpt-4.1-mini", help="OpenAI model name.")
ap.add_argument("--threshold", type=int, default=75, help="Min confidence required to report as poem.")
ap.add_argument("--max-chars", type=int, default=5000, help="Max chars per candidate sent to LLM.")
ap.add_argument("--inline", action="store_true", help="Also scan inline slash-verse candidates.")
ap.add_argument("--jsonl", action="store_true", help="Output JSONL objects instead of text report.")
ap.add_argument("--cache", default=str(Path.home() / ".cache" / "poem-finder-report-cache.json"), help="Cache path.")
ap.add_argument("--no-cache", action="store_true", help="Disable cache read/write.")
ap.add_argument("--show-notes", action="store_true", help="Include LLM notes in text output.")
ap.add_argument("--limit", type=int, default=0, help="Stop after reporting N poems (0 = no limit).")
args = ap.parse_args(argv)
path = Path(args.path)
text = path.read_text(encoding="utf-8")
lines = text.splitlines(keepends=True)
cache_path = None if args.no_cache else Path(args.cache)
cache = load_cache(cache_path)
client = OpenAI()
candidates = scan_blocks(lines, do_inline=args.inline)
reported = 0
for blk in candidates:
# extra “already marked” guard
if blk.kind != "inline_slash":
if any(line_has_poem_markup(ln) for ln in blk.raw_lines):
continue
result = classify(client, args.model, blk.kind, blk.text_for_llm, cache, args.max_chars)
if result["decision"] != "poem" or int(result["confidence"]) < args.threshold:
continue
reported += 1
if args.jsonl:
obj = {
"file": str(path),
"kind": blk.kind,
"start_line": blk.start + 1,
"end_line": blk.end + 1,
"confidence": int(result["confidence"]),
"notes": result["notes"],
"text": blk.text_for_llm,
"raw": "".join(blk.raw_lines),
}
print(json.dumps(obj, ensure_ascii=False))
else:
print("-" * 80)
print(f"{path}:{blk.start+1}-{blk.end+1} kind={blk.kind} confidence={int(result['confidence'])}")
if args.show_notes and result.get("notes"):
print(f"notes: {result['notes']}")
print()
print(blk.text_for_llm.strip("\n"))
print()
print("RAW:")
sys.stdout.write("".join(blk.raw_lines))
if not blk.raw_lines[-1].endswith("\n"):
print()
if args.limit and reported >= args.limit:
break
save_cache(cache_path, cache)
if not args.jsonl:
eprint(f"[poem-finder] reported {reported} poems (threshold={args.threshold}, inline={args.inline})")
return 0
if __name__ == "__main__":
raise SystemExit(main())After some tuning and pilot runs, and one or two tweaks I forget, the script found a good 20–50 remaining instances, which I then fixed. I have not noticed many, if any, unmarked poems since. I don’t know how much it cost to run the script a few times on the site corpus, but I didn’t notice it on my OA API bill, so I assume it cost <$5; well worth the money!