# Roll out a workflow rewrite

> Canary a rewritten multi-step workflow on a small share of runs, score both paths, and ramp up safely.

A rewrite rarely changes one step. It changes how several steps work together. `group.experiment()` can hold a whole workflow path in each variant, so you can send one run in a hundred through the rewrite, score both paths the same way, and ramp up once the results hold.

## 1. Put each path in a variant

Each callback runs a complete path. Only the selected path runs; the other callback is skipped:

```typescript {{ title: "TypeScript", filename: "inngest/invoice.ts" }}
import { experiment } from "inngest";
import { inngest } from "./client";

export default inngest.createFunction(
  { id: "generate-invoice", triggers: { event: "billing/invoice.requested" } },
  async ({ event, step, group, runId }) => {
    const { result: invoice, experimentRef } = await group.experiment(
      "invoice-engine",
      {
        variants: {
          current: async () => {
            const invoice = await step.run("generate-current", () =>
              generateInvoiceV1(event.data)
            );
            await step.run("send-current", () => sendInvoice(invoice));
            return invoice;
          },
          rewrite: async () => {
            const draft = await step.run("draft-rewrite", () =>
              draftInvoiceV2(event.data)
            );
            const invoice = await step.run("price-rewrite", () =>
              applyPricingV2(draft)
            );
            await step.run("send-rewrite", () => sendInvoice(invoice));
            return invoice;
          },
        },
        select: experiment.weighted({ current: 99, rewrite: 1 }),
      }
    );

    // Save what a later outcome needs to find this run and variant.
    await step.run("save-invoice", () =>
      db.invoices.insert({ id: invoice.id, runId, experimentRef })
    );

    await inngest.score.experiment({
      name: "invoice-valid",
      value: validateInvoice(invoice),
      experiment: experimentRef,
    });

    return invoice;
  }
);
```

```python {{ title: "Python", filename: "invoice.py" }}
# Requires the inngest release after 0.5.19.
import inngest
from inngest.experimental import experiment

from .client import inngest_client
from .stubs import (
    Invoice,
    apply_pricing_v2,
    db,
    draft_invoice_v2,
    generate_invoice_v1,
    send_invoice,
    validate_invoice,
)

@inngest_client.create_function(
    fn_id="generate-invoice",
    trigger=inngest.TriggerEvent(event="billing/invoice.requested"),
)
async def generate_invoice(ctx: inngest.Context) -> Invoice:
    data = ctx.event.data

    async def current() -> Invoice:
        async def generate() -> Invoice:
            return await generate_invoice_v1(data)

        invoice = await ctx.step.run("generate-current", generate)

        async def send() -> None:
            await send_invoice(invoice)

        await ctx.step.run("send-current", send)
        return invoice

    async def rewrite() -> Invoice:
        async def draft_invoice() -> Invoice:
            return await draft_invoice_v2(data)

        draft = await ctx.step.run("draft-rewrite", draft_invoice)

        async def price() -> Invoice:
            return await apply_pricing_v2(draft)

        invoice = await ctx.step.run("price-rewrite", price)

        async def send() -> None:
            await send_invoice(invoice)

        await ctx.step.run("send-rewrite", send)
        return invoice

    res = await ctx.group.experiment(
        "invoice-engine",
        variants={"current": current, "rewrite": rewrite},
        # Python has no weighted selector. Bucketing on the run ID sends
        # about one new run in a hundred through the rewrite.
        select=experiment.bucket(
            ctx.run_id, weights={"current": 99, "rewrite": 1}
        ),
    )
    invoice = res.result

    # Save what a later outcome needs to find this run and variant.
    async def save() -> None:
        await db.invoices.insert(
            id=invoice["id"],
            run_id=ctx.run_id,
            experiment_ref=res.experiment_ref.model_dump(),
        )

    await ctx.step.run("save-invoice", save)

    async def score_valid() -> None:
        await inngest_client.score_experiment(
            name="invoice-valid",
            value=validate_invoice(invoice),
            experiment=res.experiment_ref,
            run_id=ctx.run_id,
        )

    await ctx.step.run("score-invoice-valid", score_valid)

    return invoice
```

Weights are relative: `99:1` sends about one new run in a hundred through the rewrite. Put every side effect, such as sending the invoice, inside a step so a replay doesn't repeat it.

## 2. Score the real outcome later

A valid invoice is useful, but payment or a customer correction shows whether the rewrite worked. When your payment webhook fires, load the saved values and score the original run:

```typescript {{ title: "TypeScript", filename: "inngest/payments.ts" }}
export const scorePayment = inngest.createFunction(
  { id: "score-invoice-paid", triggers: { event: "billing/invoice.paid" } },
  async ({ event, step }) => {
    const saved = await step.run("load-invoice", () =>
      db.invoices.get(event.data.invoiceId)
    );

    await inngest.score.experiment({
      name: "invoice-paid",
      value: true,
      experiment: saved.experimentRef,
      runId: saved.runId,
    });
  }
);
```

```python {{ title: "Python", filename: "payments.py" }}
# Requires the inngest release after 0.5.19.
import typing

import inngest
from inngest.experimental.experiment import ExperimentRef

from .client import inngest_client
from .stubs import db

@inngest_client.create_function(
    fn_id="score-invoice-paid",
    trigger=inngest.TriggerEvent(event="billing/invoice.paid"),
)
async def score_payment(ctx: inngest.Context) -> None:
    invoice_id = str(ctx.event.data["invoiceId"])

    async def load() -> dict[str, typing.Any]:
        return await db.invoices.get(invoice_id)

    saved = await ctx.step.run("load-invoice", load)

    async def score_paid() -> None:
        await inngest_client.score_experiment(
            name="invoice-paid",
            value=True,
            experiment=ExperimentRef.model_validate(saved["experiment_ref"]),
            run_id=saved["run_id"],
        )

    await ctx.step.run("score-invoice-paid", score_paid)
```

Passing the original `runId` places the score under the experiment. Without it, the score attaches to the `score-invoice-paid` run.

## 3. Watch, then ramp

While the rewrite runs on a small share, compare both variants on `invoice-valid` and `invoice-paid`, and check errors and latency in the traces. A 1% canary proves the new path runs; it may take a while to collect enough outcomes to show which path is better.

When the rewrite holds up, deploy new weights in stages: `{ current: 90, rewrite: 10 }`, then `50:50`, then `0:100`. Runs that already selected a variant keep it on retries and replays; new runs use the new weights. If results worsen, set the rewrite's weight to `0`, or read the split from a flag with `experiment.custom()` so you can roll back without a deploy.

When the old path is no longer needed, pin the winner with `experiment.fixed("rewrite")` or remove the experiment and keep only the rewrite's steps.

## Next steps

- [Experiments](/docs-markdown/agent-evals/experiments) covers assignment strategies and replay behavior.
- [Read results and roll out](/docs-markdown/agent-evals/guides/interpreting-results) explains how much evidence to collect before you ramp.
- [Manage eval costs](/docs-markdown/agent-evals/guides/cost-management) explains the extra selection step and scores.