Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
144 changes: 140 additions & 4 deletions src/claude_code_transcripts/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -698,6 +698,142 @@ def format_json(obj):
return f"<pre>{html.escape(str(obj))}</pre>"


# Any ANSI escape sequence. Group 1 is the parameter list and group 2 the final byte,
# so SGR ("m") can be rendered while cursor moves, erases and the rest are dropped.
ANSI_ESCAPE_RE = re.compile(r"\x1b(?:\[([0-9;:]*)([A-Za-z])|[@-Z\\-_])")

# The 16 named colours, as the palette a terminal would use.
ANSI_COLORS = [
"#000000",
"#cd0000",
"#00cd00",
"#cdcd00",
"#0000ee",
"#cd00cd",
"#00cdcd",
"#e5e5e5",
"#7f7f7f",
"#ff0000",
"#00ff00",
"#ffff00",
"#5c5cff",
"#ff00ff",
"#00ffff",
"#ffffff",
]


def _xterm256_color(n):
"""Hex for one of the 256 xterm palette entries."""
if n < 16:
return ANSI_COLORS[n]
if n < 232:
n -= 16
levels = (0, 95, 135, 175, 215, 255)
return "#{:02x}{:02x}{:02x}".format(
levels[n // 36], levels[(n // 6) % 6], levels[n % 6]
)
grey = 8 + (n - 232) * 10
return "#{:02x}{:02x}{:02x}".format(grey, grey, grey)


def _sgr_styles(params, styles):
"""Apply one SGR parameter list to the running *styles* dict."""
codes = [int(p) if p.isdigit() else 0 for p in (params or "0").split(";")]
i = 0
while i < len(codes):
code = codes[i]
if code == 0:
styles.clear()
elif code == 1:
styles["font-weight"] = "bold"
elif code == 2:
styles["opacity"] = "0.7"
elif code == 3:
styles["font-style"] = "italic"
elif code == 4:
styles["text-decoration"] = "underline"
elif code == 9:
styles["text-decoration"] = "line-through"
elif code in (22, 23, 24, 29):
# The matching "off" codes. 22 turns off both bold and dim.
for prop in {
22: ("font-weight", "opacity"),
23: ("font-style",),
24: ("text-decoration",),
29: ("text-decoration",),
}[code]:
styles.pop(prop, None)
elif code in (38, 48) and i + 1 < len(codes):
# Extended colour: 5;N (256-palette) or 2;R;G;B (truecolor).
prop = "color" if code == 38 else "background-color"
if codes[i + 1] == 5 and i + 2 < len(codes):
styles[prop] = _xterm256_color(codes[i + 2])
i += 2
elif codes[i + 1] == 2 and i + 4 < len(codes):
styles[prop] = "#{:02x}{:02x}{:02x}".format(*codes[i + 2 : i + 5])
i += 4
elif 30 <= code <= 37:
styles["color"] = ANSI_COLORS[code - 30]
elif 90 <= code <= 97:
styles["color"] = ANSI_COLORS[code - 90 + 8]
elif 40 <= code <= 47:
styles["background-color"] = ANSI_COLORS[code - 40]
elif 100 <= code <= 107:
styles["background-color"] = ANSI_COLORS[code - 100 + 8]
elif code == 39:
styles.pop("color", None)
elif code == 49:
styles.pop("background-color", None)
i += 1


def render_ansi_text(text):
"""HTML-escape *text*, turning ANSI colour codes into spans.

Terminal output reaches the transcript with its escape sequences intact -- /context
is the usual source -- and escaping it as plain text put the raw codes on the page.

The text is always escaped first; only the markup this function generates itself is
trusted, and that is built from parsed integers. Non-SGR sequences (cursor moves,
erase-line) carry nothing to show and are dropped.
"""
if not text or "\x1b" not in text:
return html.escape(text or "")

parts = []
styles = {}
open_span = False
position = 0

def close():
nonlocal open_span
if open_span:
parts.append("</span>")
open_span = False

for match in ANSI_ESCAPE_RE.finditer(text):
chunk = text[position : match.start()]
if chunk:
parts.append(html.escape(chunk))
position = match.end()

if match.group(2) != "m":
continue # not SGR: nothing to render
close()
_sgr_styles(match.group(1), styles)
if styles:
declarations = ";".join(f"{k}:{v}" for k, v in sorted(styles.items()))
parts.append(f'<span style="{declarations}">')
open_span = True

remainder = text[position:]
if remainder:
parts.append(html.escape(remainder))
close()
return "".join(parts)


def render_markdown_text(text):
if not text:
return ""
Expand Down Expand Up @@ -790,7 +926,7 @@ def render_content_block(block):
# Add any content before this commit
before = content[last_end : match.start()].strip()
if before:
parts.append(f"<pre>{html.escape(before)}</pre>")
parts.append(f"<pre>{render_ansi_text(before)}</pre>")

commit_hash = match.group(1)
commit_msg = match.group(2)
Expand All @@ -802,11 +938,11 @@ def render_content_block(block):
# Add any remaining content after last commit
after = content[last_end:].strip()
if after:
parts.append(f"<pre>{html.escape(after)}</pre>")
parts.append(f"<pre>{render_ansi_text(after)}</pre>")

content_html = "".join(parts)
else:
content_html = f"<pre>{html.escape(content)}</pre>"
content_html = f"<pre>{render_ansi_text(content)}</pre>"
elif isinstance(content, list):
# Handle tool result content that contains multiple blocks (text, images, etc.)
parts = []
Expand All @@ -816,7 +952,7 @@ def render_content_block(block):
if item_type == "text":
text = item.get("text", "")
if text:
parts.append(f"<pre>{html.escape(text)}</pre>")
parts.append(f"<pre>{render_ansi_text(text)}</pre>")
elif item_type == "image":
source = item.get("source", {})
media_type = source.get("media_type", "image/png")
Expand Down
76 changes: 76 additions & 0 deletions tests/test_generate_html.py
Original file line number Diff line number Diff line change
Expand Up @@ -1638,3 +1638,79 @@ def test_search_total_pages_available(self, output_dir):

# Total pages should be embedded for JS to know how many pages to fetch
assert "totalPages" in index_html or "total_pages" in index_html


class TestAnsiEscapeCodes:
"""Terminal output reaching the transcript with its ANSI escape codes intact.

A command like /context writes colour with SGR sequences. Those arrived in the
JSONL verbatim and were escaped as literal text, so the rendered transcript read
"\x1b[1mContext Usage\x1b[22m ..." -- the output most worth reading was the least
readable thing on the page (issue #95).
"""

def test_sgr_codes_are_not_shown_as_literal_text(self):
block = {
"type": "tool_result",
"content": "\x1b[1mContext Usage\x1b[22m done",
}
result = render_content_block(block)

assert "\x1b" not in result
assert "[1m" not in result
assert "[22m" not in result
assert "Context Usage" in result
assert "done" in result

def test_bold_becomes_markup_rather_than_being_discarded(self):
block = {"type": "tool_result", "content": "\x1b[1mloud\x1b[22m quiet"}
result = render_content_block(block)

assert "font-weight:bold" in result.replace(" ", "")
assert "loud" in result

def test_truecolor_is_carried_through(self):
# The exact form /context emits: 38;2;R;G;B
block = {
"type": "tool_result",
"content": "\x1b[38;2;136;136;136m dim \x1b[0mx",
}
result = render_content_block(block)

assert "#888888" in result

def test_256_colour_is_carried_through(self):
block = {"type": "tool_result", "content": "\x1b[38;5;196mred\x1b[0m"}
result = render_content_block(block)

assert "red" in result
assert "#" in result

def test_non_sgr_sequences_are_removed_entirely(self):
# Cursor movement and erase-line carry nothing worth rendering.
block = {"type": "tool_result", "content": "\x1b[2K\x1b[1Gprogress"}
result = render_content_block(block)

assert "\x1b" not in result
assert "2K" not in result
assert "progress" in result

def test_html_in_ansi_output_is_still_escaped(self):
"""The reason this cannot just be marked safe."""
block = {
"type": "tool_result",
"content": "\x1b[1m<script>alert(1)</script>\x1b[22m",
}
result = render_content_block(block)

assert "<script>" not in result
assert "&lt;script&gt;" in result

def test_plain_output_is_unchanged(self):
"""The control: text with no escape sequences must render exactly as before."""
block = {"type": "tool_result", "content": "just plain output"}

assert render_content_block(block) == render_content_block(
{"type": "tool_result", "content": "just plain output"}
)
assert "<span" not in render_content_block(block)