"""
ralph — the generic Ralph loop and its two instantiations.
The Ralph loop is the "produce, inspect, fix, repeat until a verdict" pattern
used throughout the sprezzature suite. This module implements the generic
driver and two concrete variants:
* ``eyeball_loop``: inspects a visual artifact (PNG) with a vision-language
model; fixes the source code that generated it.
* ``prose_loop``: inspects prose with a text model enforcing a writing charter;
fixes the text at paragraph-pair seams.
Both variants share the same generic driver so the convergence logic, iteration
budget, and no-op guard are implemented once.
Author
------
Warith Harchaoui <warith.harchaoui@deraison.ai>
"""
from __future__ import annotations
import json
from collections.abc import Callable
from typing import Any
import os_helper as osh
# ---------------------------------------------------------------------------
# Prompts used by the eyeball and prose variants
# ---------------------------------------------------------------------------
# System prompt for the visual critique step: asks the VLM to evaluate the
# rendered artifact under fixed bold headings so the verdict step can parse
# structured output from the free-form critique.
_EYEBALL_CRITIQUE_SYSTEM = """\
You are a strict visual-quality reviewer. Examine the image and write a
structured critique under these exact headings:
Layout, Contrast, Hierarchy, Spacing, Accessibility, Colors, Text readability,
Overall verdict.
For each heading, give one or two sentences. Be specific: name the element,
its position, and the concrete problem. If a heading has no issues, write
"No issues." Do not add headings beyond the list above.\
"""
_EYEBALL_CRITIQUE_PROMPT = "Critique this image. Use only the headings listed in the system prompt."
# Verdict prompt: asks the VLM to return a machine-readable pass/fail object
# so the generic driver can decide whether to continue iterating.
_EYEBALL_VERDICT_PROMPT = """\
Given this critique, decide whether the artifact is ready to ship.
Respond ONLY with a JSON object following this schema exactly:
{{"ship": <true|false>, "blocking": [<list of short blocking issues>], "score": <0.0-1.0>}}
"ship" is true only when there are no blocking issues and the score exceeds 0.7.
"blocking" lists the one-line labels of unresolved problems; empty list if none.
"score" is your overall quality estimate from 0 (broken) to 1 (perfect).
Critique:
{critique}
"""
# Prompt used by apply_fix in the eyeball loop: feeds the critique back to a
# text model to produce an edited source.
_EYEBALL_FIX_SYSTEM = """\
You are a code editor. You receive a visual artifact source and a critique.
Edit the source to address every blocking complaint in the critique.
Change nothing the critique did not raise.
Return ONLY the full revised source. No explanation, no markdown fences.\
"""
_EYEBALL_FIX_PROMPT = """\
CURRENT SOURCE:
{source}
REVIEWER CRITIQUE:
{critique}
"""
# Prose loop: seam inspection prompt.
_PROSE_SEAM_SYSTEM = """\
You are a writing-quality reviewer enforcing a strict prose charter.
Examine the junction between paragraph A and paragraph B.
Return ONLY a JSON object with this schema:
{{"needs_fix": <true|false>,
"reasons": [<list of short reason labels>]}}
Possible reasons: "echoed-words", "bolted-on-transition", "logic-gap",
"A-does-not-call-for-B", "charter-violation".
Set needs_fix to true only when at least one reason applies.\
"""
_PROSE_SEAM_PROMPT = """\
CHARTER EXCERPT:
{charter}
PARAGRAPH A:
{a}
PARAGRAPH B:
{b}
"""
# Prose fix prompt: asks the text model to return two revised paragraphs as JSON.
_PROSE_FIX_SYSTEM = """\
You are a prose editor. Revise the junction between paragraph A and paragraph B
to fix the reasons listed in the critique. Touch only the last sentence of A
and the first sentence of B unless the problem is deeper.
Return ONLY a JSON object: {{"a": "<revised paragraph A>", "b": "<revised paragraph B>"}}
No explanation, no markdown.\
"""
_PROSE_FIX_PROMPT = """\
CHARTER EXCERPT:
{charter}
PARAGRAPH A:
{a}
PARAGRAPH B:
{b}
CRITIQUE:
{critique}
"""
# ---------------------------------------------------------------------------
# Generic driver
# ---------------------------------------------------------------------------
[docs]
def ralph_loop(
source: Any,
*,
render: Callable[[Any], Any],
inspect: Callable[[Any], str],
apply_fix: Callable[[Any, str], Any],
verdict: Callable[[str], dict[str, Any]],
max_iters: int = 6,
on_iteration: Callable[[int, Any, Any, str, dict[str, Any]], None] | None = None,
) -> tuple[Any, list[tuple[int, str, dict[str, Any]]]]:
"""
Run the produce-inspect-fix-repeat loop until convergence or budget is spent.
The caller supplies four callbacks that define the loop's behaviour; the
driver handles iteration, convergence detection, and history recording.
Parameters
----------
source : Any
Initial artifact source: a file path, a code string, a prose block.
The loop edits this value in place across iterations.
render : callable
``render(source) -> artifact`` — turn the source into an inspectable
artifact. For the eyeball loop this renders a PNG; for the prose loop
the artifact is the text itself (identity).
inspect : callable
``inspect(artifact) -> critique: str`` — examine the artifact and return
a free-form critique string.
apply_fix : callable
``apply_fix(source, critique) -> new_source`` — edit the source to
address the critique. Must return the same type as ``source``.
verdict : callable
``verdict(critique) -> dict`` — decide whether to ship. The dict must
have at minimum a boolean ``"ship"`` key.
max_iters : int
Maximum number of produce-inspect-fix cycles. Default 6.
on_iteration : callable or None
Optional callback called at the end of each iteration with arguments
``(iter_index, source, artifact, critique, verdict_dict)``. Use for
logging or writing assessment files.
Returns
-------
tuple[Any, list[tuple[int, str, dict]]]
``(final_source, history)`` where ``history`` is a list of
``(iteration_index, critique, verdict_dict)`` triples.
Examples
--------
>>> def mock_render(s): return s + "_rendered"
>>> def mock_inspect(a): return "no issues"
>>> def mock_fix(s, c): return s
>>> def mock_verdict(c): return {"ship": True, "blocking": [], "score": 1.0}
>>> src, hist = ralph_loop(
... "source",
... render=mock_render, inspect=mock_inspect,
... apply_fix=mock_fix, verdict=mock_verdict,
... )
>>> src
'source'
>>> hist[0][2]["ship"]
True
"""
history: list[tuple[int, str, dict[str, Any]]] = []
for i in range(max_iters):
# Step 1: render the source into an inspectable artifact
artifact = render(source)
# Step 2: inspect the artifact and produce a critique
critique = inspect(artifact)
# Step 3: decide whether the artifact is ready to ship
vdict = verdict(critique)
history.append((i, critique, vdict))
if on_iteration is not None:
on_iteration(i, source, artifact, critique, vdict)
osh.info(f"Ralph iteration {i + 1}/{max_iters}: ship={vdict.get('ship')}")
# Stop when the verdict says ship — the artifact is good enough
if vdict.get("ship"):
osh.info(f"Ralph converged (ship) after {i + 1} iteration(s)")
return source, history
# Step 4: apply the fix to the source
new_source = apply_fix(source, critique)
# Guard: if the fix is a no-op the loop would spin; stop early
if new_source == source:
osh.info(f"Ralph stopped early: fix was a no-op at iteration {i + 1}")
return source, history
source = new_source
# Budget exhausted: return whatever we have
osh.warning(f"Ralph budget exhausted after {max_iters} iteration(s) without shipping")
return source, history
# ---------------------------------------------------------------------------
# Eyeball loop (visual quality)
# ---------------------------------------------------------------------------
[docs]
def eyeball_loop(
source: str,
*,
kind: str,
llm_chat: Callable[..., Any],
renderers: dict[str, Callable[[str], bytes]] | None = None,
max_iters: int = 6,
on_iteration: Callable[[int, str, bytes, str, dict[str, Any]], None] | None = None,
) -> tuple[str, list[tuple[int, str, dict[str, Any]]]]:
"""
Run the eyeball loop on a visual artifact source.
Renders the source to a PNG, critiques it with a VLM, applies a text-model
fix to the source, and repeats. Uses the generic ``ralph_loop`` driver
internally.
Parameters
----------
source : str
Source text to render: a Vega-Lite JSON string, HTML, Mermaid, or TikZ.
kind : str
Surface kind. Controls which renderer is selected: ``"vega"``,
``"html"``, ``"mermaid"``, ``"tikz"``, ``"svg"``.
llm_chat : callable
The ``chat`` function from ``llm.py`` (or a compatible mock). Injected
so tests can patch it without touching the module-level default.
renderers : dict or None
Optional dict mapping kind strings to render callables
``(source_str) -> bytes``. When None the function raises
``NotImplementedError`` pointing to the sprezzature-figures renderers.
max_iters : int
Maximum iteration budget. Default 6.
on_iteration : callable or None
Optional per-iteration callback.
Returns
-------
tuple[str, list[tuple[int, str, dict]]]
``(final_source, history)``.
Raises
------
NotImplementedError
If ``renderers`` is None and no built-in renderer is available for
``kind``. Wire in the renderers from sprezzature-figures.
"""
# The renderers are defined in sprezzature-figures/scripts and are not a
# dependency of this package; the caller must inject them.
if renderers is None or kind not in renderers:
raise NotImplementedError(
f"No renderer for kind={kind!r}. "
"Pass a 'renderers' dict with a callable for this surface. "
"See sprezzature-figures/scripts/ralph_eyeball_loop.py for the "
"existing renderer implementations."
)
render_fn = renderers[kind]
def render(src: str) -> bytes:
"""Render source to a PNG byte string."""
return render_fn(src)
def inspect(png: bytes) -> str:
"""Critique the rendered PNG using the configured VLM."""
# The critique prompt embeds the PNG and asks for structured feedback
return str(llm_chat(
_EYEBALL_CRITIQUE_PROMPT,
system=_EYEBALL_CRITIQUE_SYSTEM,
images=[png],
))
def apply_fix(src: str, critique: str) -> str:
"""Generate an edited source that addresses the critique."""
prompt = _EYEBALL_FIX_PROMPT.format(source=src, critique=critique)
return str(llm_chat(prompt, system=_EYEBALL_FIX_SYSTEM))
def verdict(critique: str) -> dict[str, Any]:
"""Ask the VLM for a structured ship/no-ship decision."""
prompt = _EYEBALL_VERDICT_PROMPT.format(critique=critique)
raw = llm_chat(prompt, json_schema={"type": "object"})
# llm.chat parses JSON when json_schema is provided; fall back gracefully
if isinstance(raw, dict):
return raw
try:
parsed: dict[str, Any] = json.loads(str(raw))
return parsed
except json.JSONDecodeError:
# Malformed verdict: treat as "do not ship" so the loop continues
osh.warning("Eyeball verdict was not valid JSON; treating as do-not-ship")
return {"ship": False, "blocking": ["verdict parse failure"], "score": 0.0}
return ralph_loop(
source,
render=render,
inspect=inspect,
apply_fix=apply_fix,
verdict=verdict,
max_iters=max_iters,
on_iteration=on_iteration,
)
# ---------------------------------------------------------------------------
# Prose loop (writing charter)
# ---------------------------------------------------------------------------
def _split_paragraphs(text: str) -> list[str]:
"""
Split text into paragraphs on blank lines, stripping leading/trailing space.
Parameters
----------
text : str
Input prose block.
Returns
-------
list[str]
Non-empty paragraphs; blank-line separators discarded.
"""
# Two or more newlines delimit paragraph boundaries
import re
parts = re.split(r"\n{2,}", text.strip())
return [p.strip() for p in parts if p.strip()]
[docs]
def prose_loop(
text: str,
*,
charter: str,
llm_chat: Callable[..., Any],
max_passes: int = 3,
) -> str:
"""
Enforce the writing charter on a prose block via paragraph-pair seam checks.
Implements WRITING.md §10 "Flow by Paragraph Pairs" locally. For each
overlapping window (para n, para n+1), a text model checks the seam for
bolted-on transitions, logic gaps, echoed words, and charter violations.
When a seam needs fixing, a second call edits the last sentence of n and the
opening of n+1. The loop repeats until a full pass makes no edit or the
budget is spent.
Parameters
----------
text : str
Prose block to refine. Paragraphs are separated by blank lines.
charter : str
Writing charter excerpt to embed in every seam and fix prompt. Keeps the
model focused on the specific rules that matter for this language.
llm_chat : callable
The ``chat`` function from ``llm.py`` or a compatible mock.
max_passes : int
Maximum number of full passes over all paragraph pairs. Default 3.
Returns
-------
str
The refined prose block with the same paragraph structure.
Examples
--------
>>> # With a mock llm that fixes nothing, the output equals the input
>>> def noop_chat(p, **kw): return '{"needs_fix": false, "reasons": []}'
>>> text = "Paragraph one.\\n\\nParagraph two."
>>> result = prose_loop(text, charter="No dashes.", llm_chat=noop_chat)
>>> result == text
True
"""
paras = _split_paragraphs(text)
if len(paras) < 2:
# Nothing to check at a paragraph seam when there is only one paragraph
return text
osh.info(f"Prose loop over {len(paras)} paragraph(s), up to {max_passes} pass(es)")
for _pass in range(max_passes):
changed = False
# Slide an overlapping window over every adjacent pair
for n in range(len(paras) - 1):
a, b = paras[n], paras[n + 1]
seam_prompt = _PROSE_SEAM_PROMPT.format(charter=charter, a=a, b=b)
raw = llm_chat(
seam_prompt,
system=_PROSE_SEAM_SYSTEM,
json_schema={"type": "object"},
)
# Parse the seam verdict; treat malformed responses as "no fix needed"
if isinstance(raw, dict):
seam_verdict = raw
else:
try:
seam_verdict = json.loads(str(raw))
except json.JSONDecodeError:
seam_verdict = {"needs_fix": False, "reasons": []}
if not seam_verdict.get("needs_fix", False):
continue
# Fix: ask the model to revise only the seam sentences
fix_prompt = _PROSE_FIX_PROMPT.format(
charter=charter, a=a, b=b,
critique=json.dumps(seam_verdict.get("reasons", [])),
)
fix_raw = llm_chat(
fix_prompt,
system=_PROSE_FIX_SYSTEM,
json_schema={"type": "object"},
)
if isinstance(fix_raw, dict):
fix = fix_raw
else:
try:
fix = json.loads(str(fix_raw))
except json.JSONDecodeError:
# Skip this pair if the fix response is unparseable
continue
new_a = fix.get("a", a)
new_b = fix.get("b", b)
if new_a != a or new_b != b:
paras[n] = new_a
paras[n + 1] = new_b
changed = True
# A full pass with no changes means the text has converged
if not changed:
osh.info(f"Prose loop converged after pass {_pass + 1}")
break
return "\n\n".join(paras)