diff --git a/strix/interface/tui/app.py b/strix/interface/tui/app.py index 72a53b80..21a45b9f 100644 --- a/strix/interface/tui/app.py +++ b/strix/interface/tui/app.py @@ -41,13 +41,13 @@ from strix.interface.tui.renderers import render_tool_widget from strix.interface.tui.renderers.agent_message_renderer import AgentMessageRenderer from strix.interface.tui.renderers.user_message_renderer import UserMessageRenderer from strix.interface.utils import build_tui_stats_text -from strix.report.fenced import ( +from strix.report.state import ReportState, set_global_report_state +from strix.report.writer import ( guess_language_name, parse_fenced_code, resolve_lexer, safe_fence, ) -from strix.report.state import ReportState, set_global_report_state from strix.runtime import session_manager diff --git a/strix/interface/tui/renderers/reporting_renderer.py b/strix/interface/tui/renderers/reporting_renderer.py index d9ac4f11..8f609a93 100644 --- a/strix/interface/tui/renderers/reporting_renderer.py +++ b/strix/interface/tui/renderers/reporting_renderer.py @@ -5,7 +5,7 @@ from pygments.styles import get_style_by_name from rich.text import Text from textual.widgets import Static -from strix.report.fenced import parse_fenced_code, resolve_lexer +from strix.report.writer import parse_fenced_code, resolve_lexer from .base_renderer import BaseToolRenderer from .registry import register_tool_renderer diff --git a/strix/report/fenced.py b/strix/report/fenced.py deleted file mode 100644 index 2d997e2b..00000000 --- a/strix/report/fenced.py +++ /dev/null @@ -1,73 +0,0 @@ -import re - -from pygments.lexer import Lexer -from pygments.lexers import PythonLexer, get_lexer_by_name, guess_lexer -from pygments.lexers.special import TextLexer -from pygments.util import ClassNotFound - - -_FENCE_RE = re.compile(r"^```([^\n`]*)\r?\n(.*?)\r?\n?```$", re.DOTALL) -_BACKTICK_RUN = re.compile(r"`+") - - -def safe_fence(content: str) -> str: - """Return a backtick fence that ``content`` cannot break out of. - - Per CommonMark a fenced code block is closed only by a run of backticks at - least as long as the opening fence. LLM-authored, attacker-influenced values - (PoC scripts, code snippets) may contain their own ``` runs, so we open with - a fence one backtick longer than the longest run inside ``content`` (never - fewer than three). Everything in ``content`` then renders verbatim. - """ - longest = max((len(m.group()) for m in _BACKTICK_RUN.finditer(content)), default=0) - return "`" * max(3, longest + 1) - - -def parse_fenced_code(raw: str) -> tuple[str | None, str]: - """Split an optionally fenced code string into ``(language, code)``. - - Agent-generated code fields (e.g. ``poc_script_code``) are stored wrapped in - a markdown fence carrying the language, like ``` ```python\n...\n``` ```. - Return the fence's language tag and the inner code, or ``(None, raw)`` when - the value isn't fenced. - """ - match = _FENCE_RE.match(raw.strip()) - if not match: - return None, raw - info = match.group(1).strip() - language = info.split()[0] if info else None - return (language or None), match.group(2) - - -def resolve_lexer(language: str | None, code: str) -> Lexer: - """Pick a pygments lexer for ``code``. - - Prefer the explicit fence ``language`` when it names a known lexer, otherwise - auto-detect from the source. Fall back to Python when detection is - inconclusive, since legacy (unfenced) PoC scripts are Python. - """ - if language: - try: - return get_lexer_by_name(language) - except ClassNotFound: - pass - try: - lexer = guess_lexer(code) - except ClassNotFound: - return PythonLexer() - # ``guess_lexer`` returns the plain-text lexer when it can't detect anything. - if isinstance(lexer, TextLexer): - return PythonLexer() - return lexer - - -def guess_language_name(code: str) -> str: - """Return a markdown fence tag for ``code``, defaulting to ``python`` when - auto-detection is inconclusive.""" - try: - lexer = guess_lexer(code) - except ClassNotFound: - return "python" - if isinstance(lexer, TextLexer) or not lexer.aliases: - return "python" - return str(lexer.aliases[0]) diff --git a/strix/report/writer.py b/strix/report/writer.py index 8cfb3895..32fc961f 100644 --- a/strix/report/writer.py +++ b/strix/report/writer.py @@ -6,19 +6,92 @@ import csv import io import json import logging +import re import tempfile from datetime import UTC, datetime from pathlib import Path -from typing import Any +from typing import TYPE_CHECKING, Any + +from pygments.lexers import PythonLexer, get_lexer_by_name, guess_lexer +from pygments.lexers.special import TextLexer +from pygments.util import ClassNotFound from strix.core.paths import run_record_path -from strix.report.fenced import guess_language_name, parse_fenced_code, safe_fence +if TYPE_CHECKING: + from pygments.lexer import Lexer + logger = logging.getLogger(__name__) _SEVERITY_ORDER = {"critical": 0, "high": 1, "medium": 2, "low": 3, "info": 4} +_FENCE_RE = re.compile(r"^```([^\n`]*)\r?\n(.*?)\r?\n?```$", re.DOTALL) +_BACKTICK_RUN = re.compile(r"`+") + + +def safe_fence(content: str) -> str: + """Return a backtick fence that ``content`` cannot break out of. + + Per CommonMark a fenced code block is closed only by a run of backticks at + least as long as the opening fence. LLM-authored, attacker-influenced values + (PoC scripts, code snippets) may contain their own ``` runs, so we open with + a fence one backtick longer than the longest run inside ``content`` (never + fewer than three). Everything in ``content`` then renders verbatim. + """ + longest = max((len(m.group()) for m in _BACKTICK_RUN.finditer(content)), default=0) + return "`" * max(3, longest + 1) + + +def parse_fenced_code(raw: str) -> tuple[str | None, str]: + """Split an optionally fenced code string into ``(language, code)``. + + Agent-generated code fields (e.g. ``poc_script_code``) are stored wrapped in + a markdown fence carrying the language, like ``` ```python\n...\n``` ```. + Return the fence's language tag and the inner code, or ``(None, raw)`` when + the value isn't fenced. + """ + match = _FENCE_RE.match(raw.strip()) + if not match: + return None, raw + info = match.group(1).strip() + language = info.split()[0] if info else None + return (language or None), match.group(2) + + +def resolve_lexer(language: str | None, code: str) -> Lexer: + """Pick a pygments lexer for ``code``. + + Prefer the explicit fence ``language`` when it names a known lexer, otherwise + auto-detect from the source. Fall back to Python when detection is + inconclusive, since legacy (unfenced) PoC scripts are Python. + """ + if language: + try: + return get_lexer_by_name(language) + except ClassNotFound: + pass + try: + lexer = guess_lexer(code) + except ClassNotFound: + return PythonLexer() + # ``guess_lexer`` returns the plain-text lexer when it can't detect anything. + if isinstance(lexer, TextLexer): + return PythonLexer() + return lexer + + +def guess_language_name(code: str) -> str: + """Return a markdown fence tag for ``code``, defaulting to ``python`` when + auto-detection is inconclusive.""" + try: + lexer = guess_lexer(code) + except ClassNotFound: + return "python" + if isinstance(lexer, TextLexer) or not lexer.aliases: + return "python" + return str(lexer.aliases[0]) + def read_run_record(run_dir: Path) -> dict[str, Any]: path = run_record_path(run_dir) diff --git a/tests/test_fenced_code.py b/tests/test_fenced_code.py index b711781c..bef2b8ae 100644 --- a/tests/test_fenced_code.py +++ b/tests/test_fenced_code.py @@ -4,7 +4,7 @@ from __future__ import annotations from pygments.lexers import BashLexer, PythonLexer -from strix.report.fenced import ( +from strix.report.writer import ( guess_language_name, parse_fenced_code, resolve_lexer,