Thread Traces
Group your traces as threads to evaluate an entire conversation workflow
Overview
A "thread" on Confident AI is a group of one or more traces linked by a shared thread ID. This is useful for building conversational AI apps — chatbots, multi-turn agents, etc. — where you want to view and evaluate an entire conversation as a single unit.
Each call to your app creates a trace, and traces with the same thread ID are grouped together chronologically as turns in a conversation.
Create a Thread
The simplest way to create a thread is to wrap each turn of your app in turn(). It starts a fresh trace for that turn and stamps it with the thread ID you give it, so any traces that share the same thread ID are grouped into a single thread.
from openai import OpenAI
from confident_trace import init, turn, update_trace, shutdown
init()
client = OpenAI()
def llm_app(query: str, thread_id: str):
with turn("llm_app", thread_id=thread_id):
res = client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": query}]
).choices[0].message.content
update_trace(input=query, output=res)
return res
try:
llm_app("What's the weather in SF?", thread_id="your-thread-id")
llm_app("What about tomorrow?", thread_id="your-thread-id")
finally:
shutdown()import OpenAI from "openai";
import { init, turn, updateTrace } from "confident-trace";
const runtime = init();
const openai = new OpenAI();
const llmApp = async (query: string, threadId: string) => {
return turn({ threadId }, async () => {
const res = await openai.chat.completions.create({
model: "gpt-4o",
messages: [{ role: "user", content: query }],
});
const data = res.choices[0].message.content;
updateTrace({ input: query, output: data });
return data;
});
};
try {
await llmApp("What's the weather in SF?", "your-thread-id");
await llmApp("What about tomorrow?", "your-thread-id");
} finally {
await runtime.shutdown();
}Remember to run your entry point with the Node preload so auto-instrumented spans (like the OpenAI call above) nest inside the turn.
The thread_id / threadId can be any string — typically a session ID or conversation ID from your app. turn() also accepts optional user_id / userId and customer_id / customerId fields if you want to identify the user and customer up front, and it works with both sync and async code.
Add a Thread ID to an Existing Trace
Setting the thread ID on an existing trace is not the same as starting a turn.
It only labels the current trace; it does not create a new one. Use this pattern
only when your application already guarantees a separate root trace for every
request. If each request is genuinely a new turn in a conversation, use
turn() to establish that boundary instead.
from openai import OpenAI
from confident_trace import span, update_trace
client = OpenAI()
def llm_app(query: str):
with span("llm_app", type="agent"):
res = client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": query}]
).choices[0].message.content
update_trace(thread_id="your-thread-id", input=query, output=res)
return resimport OpenAI from "openai";
import { withSpan, updateTrace } from "confident-trace";
const openai = new OpenAI();
const llmApp = async (query: string) => {
return withSpan({ name: "llm_app", type: "agent" }, async () => {
const res = await openai.chat.completions.create({
model: "gpt-4o",
messages: [{ role: "user", content: query }],
});
const data = res.choices[0].message.content;
updateTrace({ threadId: "your-thread-id", input: query, output: data });
return data;
});
};If your app is fully auto-instrumented and already starts a separate trace for every request, a trace context can stamp the thread ID (and optionally the user and customer) on that trace without adding a wrapper span:
from confident_trace import trace_context
def llm_app(query: str, thread_id: str):
with trace_context(thread_id=thread_id):
return client.chat.completions.create(
model="gpt-4o",
messages=[{"role": "user", "content": query}]
).choices[0].message.contentimport { traceContext } from "confident-trace";
const llmApp = async (query: string, threadId: string) => {
const res = await traceContext({ threadId }, () =>
openai.chat.completions.create({
model: "gpt-4o",
messages: [{ role: "user", content: query }],
}),
);
return res.choices[0].message.content;
};A trace context creates neither a trace nor a span; it only supplies defaults to
traces started inside it. Use turn() when you need to guarantee a new trace
for the turn or set thread I/O explicitly. See Update Trace
Properties
for the full trace-context semantics.
Set Thread I/O
Although not strictly enforced, you should set the input to the raw user text and the output to the generated LLM text for each trace. These are used as the conversation turns for display on Confident AI and for thread evaluations.
from openai import OpenAI
from confident_trace import turn, update_trace
client = OpenAI()
def llm_app(query: str):
with turn("llm_app", thread_id="your-thread-id"):
messages = [{"role": "user", "content": query}]
res = client.chat.completions.create(
model="gpt-4o",
messages=messages
).choices[0].message.content
# ✅ Do this — query is the raw user input
update_trace(input=query, output=res)
# ❌ Don't do this — messages is not the raw user input
# update_trace(input=messages, output=res)
return resimport OpenAI from "openai";
import { turn, updateTrace } from "confident-trace";
const openai = new OpenAI();
const llmApp = async (query: string) => {
return turn({ threadId: "your-thread-id" }, async () => {
const messages = [{ role: "user" as const, content: query }];
const res = await openai.chat.completions.create({
model: "gpt-4o",
messages,
});
const data = res.choices[0].message.content;
// ✅ Do this — query is the raw user input
updateTrace({ input: query, output: data });
// ❌ Don't do this — messages is not the raw user input
// updateTrace({ input: messages, output: data });
return data;
});
};You don't have to set both input and output on every trace. If a turn only has a user input or only an LLM output, you can set just one. Confident AI will format the turns accordingly on the UI and for evals.
# ✅ Set only input (e.g. user message with no immediate LLM response)
update_trace(thread_id="your-thread-id", input=query)
# ✅ Set only output (e.g. proactive LLM message with no user input)
update_trace(thread_id="your-thread-id", output=res)
# ✅ Omit both (e.g. background processing step in the conversation)
update_trace(thread_id="your-thread-id")// ✅ Set only input
updateTrace({ threadId: "your-thread-id", input: query });
// ✅ Set only output
updateTrace({ threadId: "your-thread-id", output: data });
// ✅ Omit both
updateTrace({ threadId: "your-thread-id" });Set Thread Fields
You can attach custom metadata and tags to a thread to label production conversations with attributes like DVA version, client, agent ID, or status flags. Both are filterable and groupable across the observatory, which makes it easy to slice production traffic.
Thread fields are separate from the tags and metadata on an individual trace — they describe the conversation as a whole. When starting a turn, pass a thread object to set the ID, tags, and metadata together. Metadata values can be any JSON-serializable type, and tags are an array of strings.
from confident_trace import turn
with turn(thread={
"id": "chat-42",
"tags": ["support"],
"metadata": {"channel": "web"},
}):
agent.invoke(...)import { turn } from "confident-trace";
await turn({
thread: {
id: "chat-42",
tags: ["support"],
metadata: { channel: "web" },
},
}, async () => {
await agent.invoke(...);
});You can identify the thread in either of two ways — use one form or the other:
- Pass
thread_id/threadIdwhen you only need the ID. - Pass a
threadobject when you also want to set thread tags or metadata. The object can containid,tags, andmetadata.turn()still requires an ID in the object.
The same two options are available with a trace context and the trace update helper.
Set Tools Called
If your LLM app uses tool/function calling, you can log which tools were invoked for a given turn. This is attached to the trace alongside the output it helped generate, and each tool is a plain object with at least a name.
from confident_trace import turn, update_trace
def llm_app(query: str):
with turn("llm_app", thread_id="your-thread-id"):
res, tools = call_agent(query)
update_trace(
input=query,
output=res,
tools_called=[{"name": "WebSearch"}, {"name": "Calculator"}],
)
return resimport { turn, updateTrace } from "confident-trace";
const llmApp = async (query: string) => {
return turn({ threadId: "your-thread-id" }, async () => {
const { res, tools } = await callAgent(query);
updateTrace({
input: query,
output: res,
toolsCalled: [{ name: "WebSearch" }, { name: "Calculator" }],
});
return res;
});
};Set Retrieval Context
For RAG-based conversational apps, you can log the retrieval context used to generate a response. This enables Confident AI to evaluate retrieval quality across conversation turns.
from confident_trace import turn, update_trace
def llm_app(query: str):
with turn("llm_app", thread_id="your-thread-id"):
chunks = retrieve(query)
res = generate(query, chunks)
update_trace(
input=query,
output=res,
retrieval_context=[chunk.text for chunk in chunks],
)
return resimport { turn, updateTrace } from "confident-trace";
const llmApp = async (query: string) => {
return turn({ threadId: "your-thread-id" }, async () => {
const chunks = await retrieve(query);
const res = await generate(query, chunks);
updateTrace({
input: query,
output: res,
retrievalContext: chunks.map((c) => c.text),
});
return res;
});
};Next Steps
With threads set up, evaluate conversation quality or add more context to your traces.
Evaluate Threads
Run online evaluations on entire conversation threads to monitor multi-turn quality.
Customize Traces
Add tags, metadata, user info, and customer info to your traces.
Last updated on