Skip to content

End to end

This is the loop the package exists for: name the version, push it, run it, read the number, change the graph, do it again. Everything below is one script, split up so each move can be explained.

  1. The graph is a dict, in Python, where a script can change it.

  2. chatty-lab names it — and refuses if it is not a graph the engine could run, which costs nothing and happens before anything else does.

  3. It is committed into a history in ./work/.chatty and pushed to the platform, where the uuids are different and the digests are not.

  4. An experiment is pinned to that version and run over the dataset.

  5. The number belongs to that version, and cannot later come to belong to a different one.

import os
import chatty_lab as cl
lab = cl.Platform.from_env() # CHATTY_API, CHATTY_TOKEN, CHATTY_WORKSPACE
# The rows, named by a version of their own.
ds = lab.new_dataset("medqa", columns=[("input", "text"), ("reference", "text")])
ds.upload(rows, message="MedQA test, first 30")
# The lineage on the platform. It has to exist over there before anything
# can be pushed at it, so it is created once with a placeholder blob.
graph_id = lab.call("POST", "/graphs", {
"name": "medqa",
"graph_data": {"nodes": [], "edges": []},
})["id"]
# The history on this disk, pointed at the same lineage.
here = cl.Workdir(cl.Store.at("./work"), graph=graph_id)
there = cl.Store.on(lab.api, token=os.environ["CHATTY_TOKEN"], graph=graph_id,
workspace=lab.workspace)
from chatty_lab import Input, Llm, StructuredOutput, Eval, Output, build
LETTERS = {"type": "object", "required": ["answer"], "additionalProperties": False,
"properties": {"answer": {"type": "string", "enum": ["A", "B", "C", "D"]}}}
v1 = build(
Input()
>> Llm("reader", model="NV:openai/gpt-oss-20b",
prompt="A question from the US medical licensing exam.\n\n"
"{{dataset_input}}\n\nWork through it, then pick an option.",
temperature=0.2, max_tokens=700)
>> StructuredOutput("verdict", model="NV:openai/gpt-oss-20b",
prompt="Which option did this pick?\n\n{{reader}}",
output_schema=LETTERS)
>> Eval("score", metrics=["exact_match"], prediction_var="verdict")
>> Output()
)
made = here.commit(v1, message="one reader, one pass")
made.already # False the first time; True if you run this again unchanged
made.version.n # 1
made.version.digest.short
carried = here.push(there)
over_there = carried.head.head # the version's id ON THE PLATFORM
experiment = lab.call("POST", "/experiments", {
"graph_id": graph_id,
"name": f"medqa v1 · {made.version.digest.short}",
"dataset_id": ds.id,
"param_space": {
"type": "grid",
"params": [],
"score": {"metric": "exact_match", "direction": "maximize"},
},
})["id"]
lab.call("POST", f"/experiments/{experiment}/pin", {"graph_version_id": over_there})

This one line is the whole of versioned research: the runner reads the pinned version’s blob and not the live graph, so an experiment cannot be measuring one thing while the canvas shows another.

Before spending the whole dataset, spend one row:

lab.call("POST", f"/experiments/{experiment}/preview", {"sample_size": 1})
lab.call("POST", f"/experiments/{experiment}/run", {})
# … poll until nothing is still running …
trials = lab.call("GET", f"/experiments/{experiment}/trials")
(trial,) = [t for t in trials if not t["is_preview"]]
trial["eval"]["exact_match"] # the metric
trial["items_answered"], trial["items_total"] # read these together
trial["input_tokens"] + trial["output_tokens"]
trial["cost_usd"]

Change the dict, commit, push, pin a new experiment. Same five moves.

# Three model families, so the disagreement is worth something: three seats on
# one model with three system prompts is one model talking to itself.
NVIDIA = "NV:nvidia/nemotron-3-super-120b-a12b"
META = "NV:meta/llama-3.2-90b-vision-instruct"
def seat(id, persona, model):
return Llm(id, model=model, system=persona, temperature=0.2, max_tokens=700,
prompt="{{dataset_input}}\n\nGive your reading in under 120 words.",
on_error=cl.Retry(1, continue_on_fail=True, timeout_secs=240))
v2 = build(
Input()
>> (
seat("internist", "You are a general internist.", "NV:openai/gpt-oss-20b")
| seat("pharmacologist", "You are a clinical pharmacologist.", NVIDIA)
| seat("surgeon", "You are a surgeon.", META)
)
>> Merge("table", strategy="concatenate")
>> Llm("chair", model="NV:openai/gpt-oss-20b",
prompt="Three readings:\n\n{{table}}\n\nDecide, and say why.")
>> StructuredOutput("verdict", model="NV:openai/gpt-oss-20b",
prompt="Which option did this pick?\n\n{{chair}}",
output_schema=LETTERS)
>> Eval("score", metrics=["exact_match"], prediction_var="verdict")
>> Output()
)
here.commit(v2, message="three specialists and a chair")
here.push(there)
for v in here.log():
print(f"v{v.n} {v.digest.short} {v.message}")
d = here.diff("HEAD~1", "HEAD")
d.not_comparable # read this first
for c in d.changes:
print(c.id, c.findings, c.because_of)

And to try a variant of an older shape without disturbing the line you are on:

alt = here.start("cheaper-chair", at="HEAD~1")
alt.commit(v3, message="the same panel, a smaller chair")
alt.push(there)
here.diff("main", "cheaper-chair")

Real model calls against whatever provider the workspace has keys for. Nothing in this loop needs a GPU, a vector database, or an embedding model.