<?xml version="1.0" encoding="utf-8"?>
<feed xmlns="http://www.w3.org/2005/Atom">
  <title>adelzaalouk</title>
  <subtitle>Personal blog of Adel Zaalouk</subtitle>
  <link href="https://adelzaalouk.me//atom.xml" rel="self" />
  <link href="https://adelzaalouk.me/" />
  <id>https://adelzaalouk.me/</id>
  <updated>2026-06-27T00:00:00Z</updated>
  <author>
    <name>Adel Zaalouk</name>
  </author>
  
    <entry>
      <title>Agentic Code Review</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/agentic-code-review/" />
      <id>https://adelzaalouk.me/2026/Jun/27/agentic-code-review/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>AI Engineering Report 2026: The Acceleration Whiplash</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/ai-engineering-report-2026-the-acceleration-whipla/" />
      <id>https://adelzaalouk.me/2026/Jun/27/ai-engineering-report-2026-the-acceleration-whipla/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>AI Tool Impact on Developer Productive Output</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/ai-tool-impact-on-developer-productive-output/" />
      <id>https://adelzaalouk.me/2026/Jun/27/ai-tool-impact-on-developer-productive-output/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>BitsAI-CR: Two-Stage Code Review at ByteDance</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/bitsai-cr-two-stage-code-review-at-bytedance/" />
      <id>https://adelzaalouk.me/2026/Jun/27/bitsai-cr-two-stage-code-review-at-bytedance/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>CodeRabbit&apos;s analysis of 470 open-source PRs</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/coderabbit-s-analysis-of-470-open-source-prs/" />
      <id>https://adelzaalouk.me/2026/Jun/27/coderabbit-s-analysis-of-470-open-source-prs/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>HighStakes on GitHub</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/highstakes-on-github/" />
      <id>https://adelzaalouk.me/2026/Jun/27/highstakes-on-github/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>Semantically-Seeded Impact Analysis</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/semantically-seeded-impact-analysis/" />
      <id>https://adelzaalouk.me/2026/Jun/27/semantically-seeded-impact-analysis/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>HighStakes: where humans review, where AI handles the rest</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/your-code-review-process-is-already-broken/" />
      <id>https://adelzaalouk.me/2026/Jun/27/your-code-review-process-is-already-broken/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary>Not all code changes carry the same risk. HighStakes scores every file by blast radius so your senior engineers review the code that matters and AI handles the rest.</summary>
      <content type="html">Something quietly changed in how engineering teams work over the past year, and most leaders I talk to haven&apos;t fully reckoned with it yet.

Your team adopted AI coding tools. Output went up. That part was visible. What wasn&apos;t visible was the review burden that came with it.

Faros AI published their [AI Engineering Report 2026](https://www.faros.ai/research/ai-acceleration-whiplash), tracking telemetry from 22,000 developers across 4,000 teams. The numbers tell the story clearly. Code churn up 861%. Per-developer defect rate from 9% to 54%. Review duration up 441%. PRs merged with zero human review up 31%.

Separately, [CodeRabbit&apos;s analysis of 470 open-source PRs](https://www.coderabbit.ai/blog/state-of-ai-vs-human-code-generation-report) found AI-written code produces 1.7x more issues, with logic errors up 75% and security vulnerabilities 1.5 to 2x more common.

Meanwhile, [GitClear&apos;s productivity analysis](https://www.gitclear.com/research/ai_tool_impact_on_developer_productive_output_from_2022_to_2025) showed that while AI users produce 4x raw output, actual durable productivity gains are far more modest. Code churn doubled from a 3.3% baseline to over 7%, meaning much of that output gets rewritten or deleted.

The gap between &quot;4x output&quot; and real productivity is review debt. Someone has to read all that generated code and decide whether to trust it. I wrote about this earlier in [The Verification Bottleneck](/2026/Feb/25/human-verification-bandwidth/): AI scales execution to near-zero cost, but verifying that output stays biologically bounded. The bottleneck was never intelligence, it&apos;s human verification bandwidth.

## The bottleneck shifted, and most teams haven&apos;t adjusted

Writing code used to be the expensive part. Now a junior developer with Copilot produces as much raw output as a senior one. But the ability to look at a diff and say &quot;I understand what this does, and I&apos;m confident it&apos;s correct&quot; is still a senior skill. That hasn&apos;t been automated.

Your most experienced engineers are now spending the majority of their time reviewing code, not writing it. Their queue is 4x what it was a year ago. The common responses I see are hiring more reviewers or quietly skipping review on &quot;safe-looking&quot; PRs. The first is expensive and slow. The second is dangerous, and that 31% zero-review merge rate suggests it&apos;s already widespread.

## Not all code is equally dangerous

Addy Osmani put it well in his piece on [Agentic Code Review](https://addyo.substack.com/p/agentic-code-review): &quot;The hard part of engineering moved from writing code to deciding whether to trust it.&quot;

But here&apos;s the thing most review processes miss. A bug in your auth middleware causes a security breach. A bug in your log formatter causes ugly timestamps. Both sit in the same review queue. Both get the same scrutiny.

Your senior engineers spend 45 minutes reviewing a PR that changes CSS colors because it&apos;s queued behind the one that modifies token validation.

The fix isn&apos;t more reviewers, it&apos;s knowing which code is high-stakes and which isn&apos;t.

I keep coming back to four questions when thinking about whether a file matters. Does it enforce a security boundary? Does it handle sensitive data? Could a failure here take down the service? Is this on a path your customers see?

A 10-line auth check can be more dangerous than a 500-line log parser. Complexity alone doesn&apos;t tell you the risk; blast radius does.

## Static analysis is necessary but not sufficient

Every code quality tool counts branches, measures cyclomatic complexity, and traces imports. These signals matter. A file with complexity 50 deserves more scrutiny than one with complexity 3. A file imported by 20 others has higher breakage potential than a leaf module.

But when I ran static analysis alone on a 382-file Rust and Python codebase, every file scored LOW. Zero differentiation. The tool could tell me that `auth.rs` has complexity 23. It could not tell me that `auth.rs` validates OIDC tokens, and a bug there means unauthorized access to every sandbox in the system.

Static analysis sees structure, not purpose. And purpose is what determines blast radius.

## Combining static analysis with LLMs closes the gap

A [recent paper on semantically-seeded impact analysis](https://arxiv.org/abs/2606.18855) confirmed what I suspected. Pure structural analysis caps at 78.6% recall. Fusing semantic understanding with structural signals brings that to 100%. Neither approach works as well alone as they work together.

So instead of replacing static analysis, I added a second layer. Each source file gets sent to an LLM with one question: &quot;if this code has a bug, what breaks?&quot;

The LLM reads `sandbox.py` and returns: &quot;Sandbox management failures could compromise isolation, leading to security breaches or service outage.&quot; It scores 73, landing in the HIGH tier.

It reads `theme.rs` and says: &quot;Breakage causes cosmetic terminal color issues.&quot; Score of 6. LOW tier.

The LLM provides the semantic understanding. Static analysis provides complexity, dependency centrality, and change frequency. Git history adds churn data and incident correlation. Combined into a weighted score, the same 382-file repo that showed zero differentiation now breaks down into 30 HIGH, 141 MEDIUM, and 134 LOW files. Those 30 files are where your senior engineers should focus. The other 134 are safe for AI-assisted review or auto-merge.

## Try it in 2 minutes

I built [HighStakes](https://github.com/zanetworker/highstakes) to do this. It&apos;s open source, a single Go binary, no dependencies beyond git.

```sh
go install github.com/zanetworker/highstakes/cmd/highstakes@latest
export OPENROUTER_API_KEY=&quot;sk-or-...&quot;

cd /path/to/your/repo
highstakes init &amp;&amp; highstakes analyze
```

That&apos;s it. It scans every source file, sends it to an LLM for blast radius assessment, combines with static analysis and git history, and gives you a score per file.

You need Go installed and an API key from [OpenRouter](https://openrouter.ai), which is free to sign up and pay-per-use. You can also point it at any OpenAI-compatible API directly, whether that&apos;s DeepSeek, OpenAI, or even a local Ollama instance. If you want to skip the LLM entirely, `--no-llm` gives you static-only analysis for free, though with less accuracy.

First analysis of a 500-file repo costs about $0.15. After that, only changed files get re-assessed, so ongoing cost is near zero.

## What you get

Every file in your repo gets a blast radius summary that explains *why* it matters, not just a number. You see scores across four impact dimensions: security, data, availability, and user impact. And you get concrete review requirements that tell you how many reviewers are needed, whether a security scan should run, and whether auto-merge is safe.

Three ways to explore the results:

An interactive treemap where the big red blocks are the files that need your attention first:

![HighStakes treemap dashboard showing files grouped by module, sized by code volume, colored by heat score](./treemap.png)

A terminal TUI with colored tier indicators, blast radius detail panel, and keyboard navigation:

![HighStakes terminal TUI showing file tree with scores and impact dimension bars](./tui.png)

A file explorer with collapsible directory tree, heat bars, and reasoning inline:

![HighStakes explorer view showing collapsible file tree with blast radius reasoning](./explorer.png)

Or machine-readable JSON for CI:

```sh
highstakes list --tier high --json
```

## The real value is in CI

When someone opens a PR, highstakes checks which files changed and posts a comment.

**HIGH: 2 files need senior review**

| File | Score | Tier |
|------|-------|------|
| src/auth/oidc.rs | 63 | HIGH |
| src/sandbox/proxy.rs | 63 | HIGH |
| src/tui/theme.rs | 6 | LOW |

**Review: 2 reviewers (senior), auto-merge blocked**

You can gate merges. If a PR touches a HIGH or CRITICAL file, the pipeline fails until a senior reviewer approves. LOW-tier PRs sail through with AI review only.

This isn&apos;t replacing human review, it&apos;s routing it. The senior engineer who reviewed 20 PRs a day now reviews the 5 that actually need them.

## The math

![ROI napkin math: $0.15 to analyze a repo, $29,000/year saved per senior engineer](./roi-math.png)

$0.15 to analyze a repo. $0.01 per PR after that.

Compare: a senior engineer spending 45 minutes reviewing a CSS change because it was in the same queue as a token validation fix. At $150/hour loaded cost, that&apos;s $112 wasted per misrouted review.

Route five reviews correctly per week. That&apos;s $29,000 per year per senior engineer in recovered capacity.

Fifteen cents versus twenty-nine thousand dollars. That&apos;s not a 10x gain, it&apos;s a 200,000x gain.

## Works with any model, anywhere

HighStakes uses the OpenAI-compatible chat API. No vendor lock-in. I tested it three ways, same binary, zero code changes:

**Cloud API (OpenRouter):** $0.15 per repo, access to every model. The default.

```sh
export OPENROUTER_API_KEY=&quot;sk-or-...&quot;
highstakes analyze
```

**Model as a Service (vLLM on OpenShift AI):** I pointed HighStakes at Nemotron Nano 3 running on an 8xH200 GPU cluster. No external API, code stays on your infrastructure, zero cost per query.

```sh
export HIGHSTAKES_API_KEY=&quot;$(oc whoami -t)&quot;
export HIGHSTAKES_API_URL=&quot;https://nemotron-nano-3.apps.your-cluster.dev/v1/chat/completions&quot;
highstakes analyze --model nvidia/nemotron-3-nano
```

It scored `scorer.go` (core calculation engine) at 53 MEDIUM with &quot;Scores code changes for tiering,&quot; while test files scored 22 LOW. Same differentiation as the cloud API, from a model running on your own cluster.

**Local Ollama:** I also tested with Qwen 1.7B on my laptop. No internet, no GPU cluster, no cost at all.

```sh
export HIGHSTAKES_API_KEY=&quot;ollama&quot;
export HIGHSTAKES_API_URL=&quot;http://localhost:11434/v1/chat/completions&quot;
highstakes analyze --model qwen3:1.7b
```

Both models agreed on the ranking: scorer.go and gitanalyzer.go are the most critical files. Test files and config are safe. The 1.7B model had more JSON parse failures (46% vs 38%), but when it succeeded, its scores were in the same range as Nemotron.

The point: you choose where your code goes. Cloud API for convenience, your own infrastructure for privacy, local laptop for zero dependency. Same tool, same output format.

## Who this is for

You&apos;re an engineering lead or senior IC. Your team adopted AI coding tools. Your review queue tripled. You&apos;re either spending all your time reviewing code that doesn&apos;t need you, or you&apos;re letting things merge that shouldn&apos;t.

HighStakes tells you which is which.

[github.com/zanetworker/highstakes](https://github.com/zanetworker/highstakes)</content>
      <category term="entry" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>The Verification Bottleneck: Why AI&apos;s Real Cost Is Human Attention</title>
      <link href="https://adelzaalouk.me/2026/Jun/27/the-verification-bottleneck-why-ai-s-real-cost-is-/" />
      <id>https://adelzaalouk.me/2026/Jun/27/the-verification-bottleneck-why-ai-s-real-cost-is-/</id>
      <updated>2026-06-27T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: your-code-review-process-is-already-broken/index.md_</content>
      <category term="link" />
      <category term="code-review" />
      <category term="ai-tools" />
      <category term="developer-productivity" />
      <category term="blast-radius" />
      <category term="highstakes" />
      <category term="verification-economy" />
    </entry>

    <entry>
      <title>Quote</title>
      <link href="https://adelzaalouk.me/2026/May/12/note-customer-names-in-the-experiment/" />
      <id>https://adelzaalouk.me/2026/May/12/note-customer-names-in-the-experiment/</id>
      <updated>2026-05-12T00:00:00Z</updated>
      <summary></summary>
      <content type="html">&gt; Note: Customer names in the experiment (GlobalBank, NebulaML) are fictional. The model invents them to satisfy the rubric&apos;s &quot;customer evidence&quot; criterion. In production, you would feed real data from interviews or support tickets.

_Extracted from: building-outcome-loops/index.md_</content>
      <category term="quote" />
      <category term="ai-agents" />
      <category term="ogx" />
      <category term="mlflow" />
      <category term="evaluation" />
      <category term="harness-engineering" />
    </entry>

    <entry>
      <title>Quote</title>
      <link href="https://adelzaalouk.me/2026/May/12/note-if-your-agent-already-uses-the-res/" />
      <id>https://adelzaalouk.me/2026/May/12/note-if-your-agent-already-uses-the-res/</id>
      <updated>2026-05-12T00:00:00Z</updated>
      <summary></summary>
      <content type="html">&gt; Note: If your agent already uses the Responses API or Interactions API with built-in tool loops, the outcome loop wraps around that agentic loop as a quality gate. The agent finishes its work (tool calls, multi-turn reasoning, whatever it does), then the outcome loop scores the final output and decides whether to send it back for revision. It is a superset of the agentic loop, not a replacement for it.

_Extracted from: building-outcome-loops/index.md_</content>
      <category term="quote" />
      <category term="ai-agents" />
      <category term="ogx" />
      <category term="mlflow" />
      <category term="evaluation" />
      <category term="harness-engineering" />
    </entry>

    <entry>
      <title>actual experiment code</title>
      <link href="https://adelzaalouk.me/2026/May/12/actual-experiment-code/" />
      <id>https://adelzaalouk.me/2026/May/12/actual-experiment-code/</id>
      <updated>2026-05-12T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: building-outcome-loops/index.md_</content>
      <category term="link" />
      <category term="ai-agents" />
      <category term="ogx" />
      <category term="mlflow" />
      <category term="evaluation" />
      <category term="harness-engineering" />
    </entry>

    <entry>
      <title>define a rubric</title>
      <link href="https://adelzaalouk.me/2026/May/12/define-a-rubric/" />
      <id>https://adelzaalouk.me/2026/May/12/define-a-rubric/</id>
      <updated>2026-05-12T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: building-outcome-loops/index.md_</content>
      <category term="link" />
      <category term="ai-agents" />
      <category term="ogx" />
      <category term="mlflow" />
      <category term="evaluation" />
      <category term="harness-engineering" />
    </entry>

    <entry>
      <title>Building Agentic Outcome Loops with an Open-Source Stack</title>
      <link href="https://adelzaalouk.me/2026/May/12/building-outcome-loops/" />
      <id>https://adelzaalouk.me/2026/May/12/building-outcome-loops/</id>
      <updated>2026-05-12T00:00:00Z</updated>
      <summary>Anthropic charges $0.08/hr for rubric-based agent evaluation. Here is how to build the same pattern the open-source way.</summary>
      <content type="html">I use LLMs to draft product specs (we call them RFEs). The output is usually *fine*, but rarely first-draft ready. Customer evidence is missing. Architecture decisions leak into what should be a business need. Three features get bundled into one. I wanted a way to automatically score drafts against our quality rubric and have the model revise until the spec is actually good.

If you have read Karpathy&apos;s [autoresearch](https://x.com/karpathy/status/1936185849498837253) work, or my earlier post on [generalizing the agentic experiment loop](/2026/Mar/15/autoimprove-autonomous-optimization), the idea is familiar: point an agent at a measurable target and let it iterate. I am calling this pattern **outcome loops**: the same iterate-until-good approach, simplified to be embeddable in any agentic workflow. No autonomous overnight runs, no complex experiment infrastructure. Just a judge call between the agent&apos;s output and the user&apos;s inbox.

&gt; **Note:** If your agent already uses the Responses API or Interactions API with built-in tool loops, the outcome loop wraps *around* that agentic loop as a quality gate. The agent finishes its work (tool calls, multi-turn reasoning, whatever it does), then the outcome loop scores the final output and decides whether to send it back for revision. It is a superset of the agentic loop, not a replacement for it.

![The outcome loop: agent generates, judge scores against rubric, feedback loops back for revision](./outcome-loop-concept.png)

## What Anthropic built

Anthropic launched **Outcomes** in their [Managed Agents](https://www.anthropic.com/engineering/managed-agents) platform in April 2026. You [define a rubric](https://platform.claude.com/docs/en/managed-agents/define-outcomes), a separate grader model scores the output, and the agent iterates until the rubric is satisfied. Results: **8-10% quality improvement** on document generation tasks.

```python
client.beta.sessions.events.create(
    session_id=session.id,
    events=[
        {&quot;type&quot;: &quot;user.message&quot;, &quot;content&quot;: [{&quot;type&quot;: &quot;text&quot;, &quot;text&quot;: &quot;Build a DCF model for Costco as .xlsx&quot;}]},
        {&quot;type&quot;: &quot;user.define_outcome&quot;, &quot;description&quot;: &quot;DCF model in xlsx&quot;,
         &quot;rubric&quot;: {&quot;type&quot;: &quot;text&quot;, &quot;content&quot;: &quot;# Success Criteria\n- 5-year projections\n- WACC with sources\n- Sensitivity table\n- Gordon Growth terminal value\n- Valid .xlsx output&quot;},
         &quot;max_iterations&quot;: 5},
    ],
)
```

The catch: Outcomes only work through Managed Agents. [**$0.08 per session-hour**](https://platform.claude.com/docs/en/about-claude/pricing) on top of tokens. Locked to Claude. Switch models, lose your quality gates.

## The open source version

You need a model, a judge, a rubric, and a loop. I chose [OGX](https://github.com/ogx-ai/ogx) for inference and [MLFlow](https://mlflow.org/docs/latest/genai/eval-monitor/) for evaluation.

**Why OGX.** An outcome loop makes *two* model calls per iteration: agent and judge. You want a self-hosted model for the agent (cost, data residency) and a strong hosted model for the judge (calibrated grading). OGX routes both through the same endpoint. `kimi/kimi-k2-6` goes to vLLM on your cluster; `openai/gpt-5-mini` goes to OpenAI&apos;s API. Same `base_url`, different model names. No separate SDKs, no separate auth.

**Why MLFlow.** The loop generates data you need to inspect: which criteria failed, did the score improve or regress, what did the judge actually say. MLFlow&apos;s [`make_genai_metric`](https://mlflow.org/docs/latest/python_api/mlflow.metrics.html) turns rubrics into versioned, reusable metrics. Scores, justifications, and output artifacts are logged automatically.

![Outcome loop architecture: task flows to OGX agent, judge scores output, feedback loops back on fail, MLFlow logs everything](./outcome-loop-animated.svg)

The implementation is about 30 lines. Define the rubric as an MLFlow metric, write the loop, point it at OGX:

```python
import openai, mlflow, pandas as pd
from mlflow.metrics.genai import make_genai_metric

# Define rubric once, reuse across tasks
rubric = make_genai_metric(
    name=&quot;code_review_quality&quot;,
    definition=&quot;Evaluate if the code review is thorough and actionable&quot;,
    grading_prompt=&quot;&quot;&quot;Score 1-5 based on:
    1. All critical bugs identified
    2. Security issues flagged
    3. Performance concerns noted
    4. Actionable suggestions (not just &quot;fix this&quot;)
    5. No false positives&quot;&quot;&quot;,
    model=&quot;endpoints:/ogx-judge&quot;,
    parameters={&quot;temperature&quot;: 0.0},
    greater_is_better=True,
    grading_context_columns=[&quot;task_description&quot;],
)

# The outcome loop
client = openai.OpenAI(base_url=&quot;https://ogx.apps.example.com/v1&quot;)

def run_with_outcome(task: str, rubric_name: str, max_iterations: int = 3):
    messages = [{&quot;role&quot;: &quot;user&quot;, &quot;content&quot;: task}]
    with mlflow.start_run(run_name=f&quot;agent-task-{rubric_name}&quot;):
        for iteration in range(max_iterations):
            response = client.chat.completions.create(model=&quot;llama-4-maverick&quot;, messages=messages)
            output = response.choices[0].message.content
            mlflow.log_text(output, f&quot;output_iteration_{iteration}.md&quot;)

            eval_result = mlflow.evaluate(
                data=pd.DataFrame({&quot;output&quot;: [output], &quot;task_description&quot;: [task]}),
                metrics=[rubric], model_type=&quot;text&quot;,
            )
            score = eval_result.metrics[f&quot;code_review_quality/v1/mean&quot;]
            feedback = eval_result.tables[&quot;eval_results_table&quot;][&quot;justification&quot;][0]
            mlflow.log_metric(f&quot;score_iteration_{iteration}&quot;, score)

            if score &gt;= 4:
                mlflow.log_metric(&quot;final_score&quot;, score)
                return output

            messages.append({&quot;role&quot;: &quot;assistant&quot;, &quot;content&quot;: output})
            messages.append({&quot;role&quot;: &quot;user&quot;, &quot;content&quot;: f&quot;Score: {score}/5. Feedback: {feedback}\n\nRevise.&quot;})
        return output
```

No agent framework. A standard OpenAI client, an MLFlow evaluate call, and a for loop. The judge does not see the agent&apos;s reasoning. It only sees the final output scored against the rubric, the same isolation Anthropic enforces with their separate grader context. This is a simplified illustration of the pattern; the [actual experiment code](https://github.com/zanetworker/ogx-experiments/tree/main/outcome-loops) uses a direct judge call with JSON parsing for more control over the scoring.

This works across any wire format OGX supports (`/v1/responses`, `/v1/interactions`, `/v1/chat/completions`). Write the rubric once, switch models and APIs without rebuilding your quality gates.

![One rubric works across all agentic API surfaces through OGX](./api-portability.png)

## What happened when I ran it

I tested this against a real task: improving low-quality RFEs using our production quality rubric. Five self-hosted models on our cluster (Kimi K2, Gemma 4, Llama 4 Scout, Nemotron 30B, Qwen 3.5 9B), two OpenAI judges (gpt-4.1-mini, gpt-5-mini), four deliberately bad RFEs. All routed through one OGX endpoint. Same code, different models.

The four bad RFEs were each broken in a different way: `vague_no_evidence` says &quot;better GPU support&quot; with no specifics or customer names. `prescriptive_architecture` mandates Redis, Envoy, and Go internals instead of describing the need. `task_not_need` reads like a migration ticket (&quot;upgrade PostgreSQL 14 to 16&quot;) rather than a business need. `bundled_scope` packs four independent features into one RFE. Each targets a different rubric criterion.

**The judge matters more than the agent.**

With gpt-4.1-mini as judge, every model scored 10/10 on the first iteration. The loop never fired. Switching to gpt-5-mini changed everything: original scores dropped to 3-5/10, and models needed **1-3 iterations** to reach the threshold. *The judge is the investment, not the loop.* A 20-line loop with a bad judge is worse than no loop at all.

| Judge | Avg original score | Avg iterations to pass | Outcome |
|-------|-------------------|----------------------|---------|
| gpt-4.1-mini | 5.0/10 | 1.0 | Every model passes immediately |
| gpt-5-mini | 3.8/10 | 1.8 | Models need 1-3 iterations, some regress |

The same bad RFE scores differently depending on the judge. A lenient judge makes the outcome loop worthless. A strict judge makes it valuable. The judge is the most important component in the system, not the agent model and not the loop code.

![Iterations needed per model and RFE type with gpt-5-mini as judge](https://mdn.alipayobjects.com/one_clip/afts/img/DO5FS4pX6jkAAAAARlAAAAgAoEACAQFr/original)

**More iterations can make things worse.**

Llama 4 Scout on a task-reframing RFE went 4, 8, 10, 9, 8. It fixed the original problems, then introduced new ones. Outcome loops need a **stop condition**: exit at the first passing score, do not keep revising.

![Score improvement curves showing regression on over-iteration](https://mdn.alipayobjects.com/one_clip/afts/img/hiMHS7zELwoAAAAAStAAAAgAoEACAQFr/original)

Each line tracks one model&apos;s score across iterations on a specific bad RFE. The orange line (Scout on task_not_need) shows the regression pattern: score peaks at iteration 2, then drops as the model over-revises.

**Context window is the real constraint for small models.**

Nemotron 30B fixed every RFE in 1 iteration. Qwen 3.5 9B hit server errors on 3 of 4 tasks because each iteration adds the previous output plus judge feedback to the conversation. By iteration 3, the 9B model&apos;s context was exhausted.

| Model | Avg iterations | Avg improvement | Context errors |
|-------|---------------|----------------|----------------|
| Nemotron 30B | 1.0 | +5.0 | 0 |
| Kimi K2 | 1.0 | +4.8 | 0 |
| Llama 4 Scout | 2.0 | +4.8 | 0 |
| Gemma 4 | 2.3 | +5.8 | 0 |
| Qwen 3.5 9B | 1.0 | +2.2 | 3 of 4 |

## MLFlow is the experiment journal

Every run is inspectable. Sort by `improvement` for the biggest quality gains, `latency_s_0` for model speed, `iterations_needed` for self-correction ability.

![MLFlow experiment view: all runs sorted by iterations needed](./mlflow-experiment-view.png)

Click into any run. The Artifacts tab has the original bad RFE, each iteration&apos;s rewrite, and the judge&apos;s justification. This is where you validate the judge: read the rewrite, read the justification, decide if you agree.

![The original bad RFE: a PostgreSQL migration task with no business justification](./mlflow-original-rfe.png)

![The improved version: reframed as a business need with customer evidence and measurable criteria](./mlflow-improved-rfe.png)

&gt; **Note:** Customer names in the experiment (GlobalBank, NebulaML) are fictional. The model invents them to satisfy the rubric&apos;s &quot;customer evidence&quot; criterion. In production, you would feed real data from interviews or support tickets.

## Takeaways

- **The judge is the system.** Swapping gpt-4.1-mini for gpt-5-mini flipped the outcome from &quot;everything passes&quot; to &quot;models need 1-3 iterations.&quot; Judge alignment is a deep topic (and out of scope here), but it is the single most important investment.
- **Humans still validate the judge.** Outcome loops reduce human review, they do not eliminate it. Read the judge&apos;s justifications in MLFlow. If you disagree with the scores, the rubric needs work. The loop automates the feedback; a human decides whether the feedback is right.
- **The stack is accessible.** OGX and MLFlow are open source. The loop is 20 lines of Python. If you have an inference endpoint and a tracking server, you can add outcome loops to an existing workflow in an afternoon.
- **The rubric is a product.** It evolves. The rubric you start with will not be the rubric you end with. Tighten where the judge is too lenient, loosen where it is too strict.
- **Watch for regression.** More iterations can degrade quality. Stop at the first passing score. Budget for context growth (~500-1000 tokens per iteration).
- **Try it.** Pick a task you delegate to agents today, write a rubric, run the loop. It scales to multi-agent pipelines: each agent gets its own rubric, and handoffs carry a quality guarantee.

Full experiment code: [ogx-experiments/outcome-loops](https://github.com/zanetworker/ogx-experiments/tree/main/outcome-loops).</content>
      <category term="entry" />
      <category term="ai-agents" />
      <category term="ogx" />
      <category term="mlflow" />
      <category term="evaluation" />
      <category term="harness-engineering" />
    </entry>

    <entry>
      <title>Quote</title>
      <link href="https://adelzaalouk.me/2026/May/10/although-the-term-is-widely-used-by-man/" />
      <id>https://adelzaalouk.me/2026/May/10/although-the-term-is-widely-used-by-man/</id>
      <updated>2026-05-10T00:00:00Z</updated>
      <summary></summary>
      <content type="html">&gt; &quot;Although the term is widely used by many people working in closely related areas, it defies attempts to produce a single universally accepted definition.&quot; — Michael Wooldridge &amp; Nicholas Jennings, Intelligent Agents: Theory and Practice (1994)

_Extracted from: beyond-llms-agentic-systems/index.mdx_</content>
      <category term="quote" />
      <category term="ai" />
      <category term="agents" />
      <category term="mcp" />
      <category term="frameworks" />
      <category term="agentic" />
    </entry>

    <entry>
      <title>acquired Manus for roughly $2B</title>
      <link href="https://adelzaalouk.me/2026/May/10/acquired-manus-for-roughly-2b/" />
      <id>https://adelzaalouk.me/2026/May/10/acquired-manus-for-roughly-2b/</id>
      <updated>2026-05-10T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: beyond-llms-moats-and-lifecycle/index.md_</content>
      <category term="link" />
      <category term="ai" />
      <category term="products" />
      <category term="strategy" />
      <category term="moats" />
    </entry>

    <entry>
      <title>Anthropic announced MCP</title>
      <link href="https://adelzaalouk.me/2026/May/10/anthropic-announced-mcp/" />
      <id>https://adelzaalouk.me/2026/May/10/anthropic-announced-mcp/</id>
      <updated>2026-05-10T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: beyond-llms-agentic-systems/index.mdx_</content>
      <category term="link" />
      <category term="ai" />
      <category term="agents" />
      <category term="mcp" />
      <category term="frameworks" />
      <category term="agentic" />
    </entry>

    <entry>
      <title>Berkeley AI Research group published a piece on compound AI systems</title>
      <link href="https://adelzaalouk.me/2026/May/10/berkeley-ai-research-group-published-a-piece-on-co/" />
      <id>https://adelzaalouk.me/2026/May/10/berkeley-ai-research-group-published-a-piece-on-co/</id>
      <updated>2026-05-10T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: beyond-llms-agentic-systems/index.mdx_</content>
      <category term="link" />
      <category term="ai" />
      <category term="agents" />
      <category term="mcp" />
      <category term="frameworks" />
      <category term="agentic" />
    </entry>

    <entry>
      <title>built roughly one million lines of production software</title>
      <link href="https://adelzaalouk.me/2026/May/10/built-roughly-one-million-lines-of-production-soft/" />
      <id>https://adelzaalouk.me/2026/May/10/built-roughly-one-million-lines-of-production-soft/</id>
      <updated>2026-05-10T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: beyond-llms-moats-and-lifecycle/index.md_</content>
      <category term="link" />
      <category term="ai" />
      <category term="products" />
      <category term="strategy" />
      <category term="moats" />
    </entry>

    <entry>
      <title>Claude 3.5 Sonnet</title>
      <link href="https://adelzaalouk.me/2026/May/10/claude-3-5-sonnet/" />
      <id>https://adelzaalouk.me/2026/May/10/claude-3-5-sonnet/</id>
      <updated>2026-05-10T00:00:00Z</updated>
      <summary></summary>
      <content type="html">_Extracted from: beyond-llms-moats-and-lifecycle/index.md_</content>
      <category term="link" />
      <category term="ai" />
      <category term="products" />
      <category term="strategy" />
      <category term="moats" />
    </entry>
</feed>