Link Test Cases to Traces
Link evaluation test cases and turns to their traces for full observability.
Overview
When you run evaluations through an AI Connection, Confident AI can link each result back to the trace your AI app produced. This gives you full observability—jump straight from an evaluation result to the exact trace that generated it.
There are two flavors of trace linking:
- Linking test cases to traces for single-turn evaluations
- Linking turns to traces for multi-turn evaluations and red-team attacks
Both work by passing an identifier (testCaseId or turnId) from your payload into your tracing setup. With confident-trace, use trace_context / traceContext to supply the identifier before an instrumented call starts, or update_trace / updateTrace if a custom span has already started the trace.
Linking Test Cases to Traces
For single-turn evaluations, you can link each test case to its corresponding trace for full observability. This is done by including testCaseId in your payload (enabled by default) and passing it to your tracing setup.
sequenceDiagram
participant C as Confident AI
participant E as Your Endpoint
participant T as Tracing
C->>E: Ping AI Connection with testCaseId
E->>T: Create trace with testCaseId
E-->>C: Return actual_output
T-->>C: Send trace to Confident AI
Note over C: Trace linked to test case
Include testCaseId in your payload configuration and ensure your AI connection is configured to accept it.
{
"input": golden.input,
"testCaseId": testCaseId
}Because testCaseId is available before your app starts its traced work, pass
it through a trace_context / traceContext. The context does not create a
trace or an extra span. It supplies the ID to the trace created by the
auto-instrumented integration call inside it.
Each example uses a FastAPI request handler and calls init() once when the
server starts.
from fastapi import FastAPI
from pydantic import BaseModel
from langchain_openai import ChatOpenAI
from confident_trace import init, trace_context
init()
app = FastAPI()
model = ChatOpenAI(model="gpt-4o")
class GenerateRequest(BaseModel):
input: str
testCaseId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(test_case_id=request.testCaseId):
output = model.invoke(request.input).content
return {"output": output}from fastapi import FastAPI
from pydantic import BaseModel
from langchain_openai import ChatOpenAI
from langgraph.prebuilt import create_react_agent
from confident_trace import init, trace_context
init()
app = FastAPI()
agent = create_react_agent(model=ChatOpenAI(model="gpt-4o"), tools=[])
class GenerateRequest(BaseModel):
input: str
testCaseId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(test_case_id=request.testCaseId):
result = agent.invoke({
"messages": [{"role": "user", "content": request.input}]
})
return {"output": result["messages"][-1].content}from fastapi import FastAPI
from pydantic import BaseModel
from openai import OpenAI
from confident_trace import init, trace_context
init()
app = FastAPI()
client = OpenAI()
class GenerateRequest(BaseModel):
input: str
testCaseId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(test_case_id=request.testCaseId):
response = client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": request.input}],
)
return {"output": response.choices[0].message.content}from fastapi import FastAPI
from pydantic import BaseModel
from langchain_openai import ChatOpenAI
from openinference.instrumentation.langchain import LangChainInstrumentor
from confident_trace import init, trace_context
init(instrumentations=())
LangChainInstrumentor().instrument()
app = FastAPI()
model = ChatOpenAI(model="gpt-4o")
class GenerateRequest(BaseModel):
input: str
testCaseId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(test_case_id=request.testCaseId):
output = model.invoke(request.input).content
return {"output": output}Each example uses an Express request handler and calls init() once when the
server starts.
import express from "express";
import OpenAI from "openai";
import { init, traceContext } from "confident-trace";
init();
const app = express();
const client = new OpenAI();
app.use(express.json());
app.post("/generate", async (req, res) => {
const output = await traceContext(
{
testCaseId: req.body.testCaseId,
},
async () => {
const response = await client.chat.completions.create({
model: "gpt-4o",
messages: [{ role: "user", content: req.body.input }],
});
return response.choices[0].message.content;
},
);
res.json({ output });
});
app.listen(3000);import express from "express";
import { generateText } from "ai";
import { openai } from "@ai-sdk/openai";
import { init, traceContext } from "confident-trace";
init();
const app = express();
app.use(express.json());
app.post("/generate", async (req, res) => {
const output = await traceContext(
{
testCaseId: req.body.testCaseId,
},
async () => {
const { text } = await generateText({
model: openai("gpt-4o"),
prompt: req.body.input,
});
return text;
},
);
res.json({ output });
});
app.listen(3000);import express from "express";
import { registerInstrumentations } from "@opentelemetry/instrumentation";
import { OpenAIInstrumentation } from "@arizeai/openinference-instrumentation-openai";
import { init, traceContext } from "confident-trace";
init({ instrumentations: [] });
registerInstrumentations({
instrumentations: [new OpenAIInstrumentation()],
});
const app = express();
const { default: OpenAI } = await import("openai");
const client = new OpenAI();
app.use(express.json());
app.post("/generate", async (req, res) => {
const output = await traceContext(
{
testCaseId: req.body.testCaseId,
},
async () => {
const response = await client.chat.completions.create({
model: "gpt-4o",
messages: [{ role: "user", content: req.body.input }],
});
return response.choices[0].message.content;
},
);
res.json({ output });
});
app.listen(3000);Run any of the TypeScript examples with the preload:
node --import tsx --import confident-trace/register src/index.tsLinking Turns to Traces
For multi-turn evaluations and multi-turn red-team attacks, Confident AI calls your endpoint once per turn. Each turn has its own turnId that you can pass to your tracing setup. This links each turn's trace to the specific turn in the conversation, letting you view traces per-turn from the evaluation or assessment results.
sequenceDiagram
participant C as Confident AI
participant E as Your Endpoint
participant T as Tracing
Note over C,E: Turn 1
C->>E: { turnId, ...payload }
E->>T: Create trace with turnId
E-->>C: Return actual_output
Note over C,E: Turn 2
C->>E: { turnId, ...payload }
E->>T: Create trace with turnId
E-->>C: Return actual_output
T-->>C: Send traces to Confident AI
Note over C: Each turn linked to its trace
Include turnId (alongside testCaseId) in your payload configuration:
{
"input": golden.input,
"testCaseId": testCaseId,
"turnId": turnId,
"state": state
}Then, supply both IDs to the trace created for each request:
Each example uses a FastAPI request handler and calls init() once when the
server starts.
from fastapi import FastAPI
from pydantic import BaseModel
from langchain_openai import ChatOpenAI
from confident_trace import init, trace_context
init()
app = FastAPI()
model = ChatOpenAI(model="gpt-4o")
class GenerateRequest(BaseModel):
input: str
testCaseId: str
turnId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(
test_case_id=request.testCaseId,
turn_id=request.turnId,
):
output = model.invoke(request.input).content
return {"output": output}from fastapi import FastAPI
from pydantic import BaseModel
from langchain_openai import ChatOpenAI
from langgraph.prebuilt import create_react_agent
from confident_trace import init, trace_context
init()
app = FastAPI()
agent = create_react_agent(model=ChatOpenAI(model="gpt-4o"), tools=[])
class GenerateRequest(BaseModel):
input: str
testCaseId: str
turnId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(
test_case_id=request.testCaseId,
turn_id=request.turnId,
):
result = agent.invoke({
"messages": [{"role": "user", "content": request.input}]
})
return {"output": result["messages"][-1].content}from fastapi import FastAPI
from pydantic import BaseModel
from openai import OpenAI
from confident_trace import init, trace_context
init()
app = FastAPI()
client = OpenAI()
class GenerateRequest(BaseModel):
input: str
testCaseId: str
turnId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(
test_case_id=request.testCaseId,
turn_id=request.turnId,
):
response = client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": request.input}],
)
return {"output": response.choices[0].message.content}from fastapi import FastAPI
from pydantic import BaseModel
from langchain_openai import ChatOpenAI
from openinference.instrumentation.langchain import LangChainInstrumentor
from confident_trace import init, trace_context
init(instrumentations=())
LangChainInstrumentor().instrument()
app = FastAPI()
model = ChatOpenAI(model="gpt-4o")
class GenerateRequest(BaseModel):
input: str
testCaseId: str
turnId: str
@app.post("/generate")
def generate(request: GenerateRequest):
with trace_context(
test_case_id=request.testCaseId,
turn_id=request.turnId,
):
output = model.invoke(request.input).content
return {"output": output}Each example uses an Express request handler and calls init() once when the
server starts.
import express from "express";
import OpenAI from "openai";
import { init, traceContext } from "confident-trace";
init();
const app = express();
const client = new OpenAI();
app.use(express.json());
app.post("/generate", async (req, res) => {
const output = await traceContext(
{
testCaseId: req.body.testCaseId,
turnId: req.body.turnId,
},
async () => {
const response = await client.chat.completions.create({
model: "gpt-4o",
messages: [{ role: "user", content: req.body.input }],
});
return response.choices[0].message.content;
},
);
res.json({ output });
});
app.listen(3000);import express from "express";
import { generateText } from "ai";
import { openai } from "@ai-sdk/openai";
import { init, traceContext } from "confident-trace";
init();
const app = express();
app.use(express.json());
app.post("/generate", async (req, res) => {
const output = await traceContext(
{
testCaseId: req.body.testCaseId,
turnId: req.body.turnId,
},
async () => {
const { text } = await generateText({
model: openai("gpt-4o"),
prompt: req.body.input,
});
return text;
},
);
res.json({ output });
});
app.listen(3000);import express from "express";
import { registerInstrumentations } from "@opentelemetry/instrumentation";
import { OpenAIInstrumentation } from "@arizeai/openinference-instrumentation-openai";
import { init, traceContext } from "confident-trace";
init({ instrumentations: [] });
registerInstrumentations({
instrumentations: [new OpenAIInstrumentation()],
});
const app = express();
const { default: OpenAI } = await import("openai");
const client = new OpenAI();
app.use(express.json());
app.post("/generate", async (req, res) => {
const output = await traceContext(
{
testCaseId: req.body.testCaseId,
turnId: req.body.turnId,
},
async () => {
const response = await client.chat.completions.create({
model: "gpt-4o",
messages: [{ role: "user", content: req.body.input }],
});
return response.choices[0].message.content;
},
);
res.json({ output });
});
app.listen(3000);Run any of the TypeScript examples with the preload:
node --import tsx --import confident-trace/register src/index.tsNext Steps
With traces linked to your evaluation results, you can debug failures end-to-end. Explore related observability and connection features next.
Multi-Turn State
Persist information across turns during multi-turn simulations.
LLM Tracing
Learn how tracing works across your AI app.
Trace-Level Detections
See which span introduced a vulnerability once red-team attacks are linked to traces.
AI Connections
Configure the payload template that carries testCaseId and turnId.
Last updated on