Skip to content
GitHub Discord

Datasets and Checks

A Dataset is a named collection of Scenarios. Each scenario defines one or more interactions (an input, an optional expected output, and checks) that the Hub uses to evaluate the agent’s response. Checks are pass/fail criteria that use an LLM judge, embedding similarity, or rule-based matching — see Built-in checks for the full reference, and Custom checks for defining reusable configurations.


from giskard_hub import HubClient
hub = HubClient()
dataset = hub.datasets.create(
project_id="project-id",
name="Core Q&A Suite v1",
description="Baseline correctness and tone checks",
)
print(dataset.id)

Datasets carry an input_schema and an output_schema (JSON Schema) describing the shape of their scenarios. When omitted, they default to the conversational (chat) format used in the examples below. For an agent with structured input/output, pass the matching schemas:

dataset = hub.datasets.create(
project_id="project-id",
name="Ticket classification suite",
input_schema={
"type": "object",
"properties": {"ticket_text": {"type": "string"}},
"required": ["ticket_text"],
},
output_schema={
"type": "object",
"properties": {"category": {"type": "string"}},
"required": ["category"],
},
)

See Agents & Knowledge Bases for how to configure the agent side.


Each scenario pairs its interactions with a list of checks. Reference any built-in check by its identifier string:

scenario = hub.scenarios.create(
dataset_id="dataset-id",
interactions=[
{
"input": {
"messages": [{"role": "user", "content": "What is your refund policy?"}]
},
"output": {
"response": {
"role": "assistant",
"content": "We offer a 30-day return policy for all unused items.",
}
},
"checks": [
{
"identifier": "hub_correctness",
"params": {
"reference": "We offer a 30-day return policy for all unused items.",
},
},
{
"identifier": "hub_conformity",
"params": {
"rule": "The agent must answer in the same language as the question."
},
},
],
}
],
)
print(scenario.id)

The output field is an optional recorded answer displayed alongside the scenario in the Hub UI. It is not used during evaluation — the agent always generates a fresh response. If your agent returns structured metadata (e.g. tool calls, categories, resolved status), include it in output.metadata:

hub.scenarios.create(
dataset_id="dataset-id",
interactions=[
{
"input": {
"messages": [{"role": "user", "content": "I need help with my order #12345"}]
},
"output": {
"response": {
"role": "assistant",
"content": "I've found your order. It was shipped on Monday and should arrive by Thursday.",
},
"metadata": {
"category": "order_status",
"resolved": True,
"tools_called": ["order_lookup"],
},
},
"checks": [
{
"identifier": "hub_correctness",
"params": {
"reference": "Order #12345 shipped Monday, arrives Thursday."
},
},
{
"identifier": "hub_metadata",
"params": {
"json_path_rules": [
{
"json_path": "$.category",
"expected_value": "order_status",
"expected_value_type": "string",
},
]
},
},
],
}
],
)

Include prior assistant turns to test multi-turn behaviour:

hub.scenarios.create(
dataset_id="dataset-id",
interactions=[
{
"input": {
"messages": [{"role": "user", "content": "I ordered a jacket last week."}]
},
"output": {
"response": {
"role": "assistant",
"content": "Happy to help! What's your order number?",
}
},
},
{
"input": {
"messages": [{"role": "user", "content": "It's #12345. I want to return it."}]
},
"output": {
"response": {
"role": "assistant",
"content": "I've initiated a return for order #12345. You'll receive a prepaid label by email.",
}
},
"checks": [
{
"identifier": "string_matching",
"params": {
"keyword": "#12345",
},
},
],
},
],
)

For an agent with custom schemas, the interaction input follows the agent’s input schema, and checks point at fields of the structured output via a target path (see Point checks at structured outputs):

hub.scenarios.create(
dataset_id="dataset-id",
interactions=[
{
"input": {"ticket_text": "My card was charged twice, please help."},
"checks": [
{
"identifier": "equals",
"params": {
"target_key": "trace.last.outputs.category",
"expected_value": "billing",
},
},
],
}
],
)

Tags let you filter scenarios during evaluation runs:

hub.scenarios.create(
dataset_id="dataset-id",
interactions=[
{
"input": {
"messages": [{"role": "user", "content": "Do you ship internationally?"}]
},
"checks": [
{
"identifier": "hub_groundedness",
"params": {
"context": "We don't ship outside the EU",
},
},
],
}
],
tags=["shipping", "faq"],
)

You can annotate scenarios with comments for team collaboration:

comment = hub.scenarios.comments.add(
"scenario-id",
content="This scenario needs a stronger expected output, the current one is too vague.",
)
print(comment.id)
# Edit a comment
hub.scenarios.comments.edit(
"comment-id", scenario_id="scenario-id", content="Updated comment text."
)
# Delete a comment
hub.scenarios.comments.delete("comment-id", scenario_id="scenario-id")

Use hub.datasets.upload() to import a dataset. Pass a name to create a new dataset, or a dataset_id to import into an existing one. Each record must follow the scenario schema, with an interactions list.

from giskard_hub import HubClient
hub = HubClient()
scenarios = [
{
"interactions": [
{
"input": {
"messages": [{"role": "user", "content": "What is your return policy?"}]
},
"checks": [
{
"identifier": "hub_correctness",
"params": {
"reference": "We accept returns within 30 days of purchase."
},
}
],
}
],
},
{
"interactions": [
{
"input": {
"messages": [{"role": "user", "content": "Do you offer free shipping?"}]
},
"checks": [
{
"identifier": "hub_correctness",
"params": {
"reference": "Free shipping is available on all orders over $50."
},
}
],
}
],
},
]
dataset = hub.datasets.upload(
project_id="project-id",
name="Imported Suite",
data=scenarios,
)
print(dataset.id)
dataset = hub.datasets.upload(
project_id="project-id",
name="Imported Suite",
data="import_data.jsonl",
)

OSS Conformity and Groundedness configurations are converted to Hub checks when you import them. The Hub uses its own evaluation logic, which is improved over the OSS version. Explicit input values and trace paths are preserved.

from giskard.checks import Conformity, Groundedness
oss_checks = [
Conformity(rule="The agent must answer politely."),
Groundedness(
context=["Our return window is 30 days."],
target_key="trace.last.outputs.response.content",
),
]
dataset = hub.datasets.upload(
project_id="project-id",
name="Imported OSS checks",
data=[
{
"interactions": [
{
"input": {
"messages": [{"role": "user", "content": "Can I return my order?"}]
},
"checks": [check.model_dump(mode="json") for check in oss_checks],
}
]
}
],
)

An OSS Conformity spec with no target uses the full trace after import and keeps its original scope, but a scenario check added by identifier with conformity or hub_conformity defaults to the response content instead, so set params["target_key"] = "trace" if you want the full trace.


Prompt presets describe a persona or behaviour pattern. The Hub uses them to generate diverse scenarios automatically.

First, create a prompt preset or use a predefined one (see Projects & Prompt Presets), then:

dataset = hub.datasets.generate_preset_based(
project_id="project-id",
agent_id="agent-id",
prompt_preset_id="prompt-preset-id",
dataset_name="Preset-generated suite",
n_examples=10,
)
# Generation is asynchronous — wait for it to finish
dataset = hub.helpers.wait_for_completion(dataset)
print(
f"Generated dataset: {dataset.id} with {len(hub.datasets.list_scenarios(dataset.id))} scenarios"
)

Use a Knowledge Base to generate scenarios whose answers are grounded in your documents:

dataset = hub.datasets.generate_document_based(
project_id="project-id",
agent_id="agent-id",
knowledge_base_id="kb-id",
dataset_name="FAQ-grounded suite",
n_examples=25,
)
# Generation is asynchronous — wait for it to finish
dataset = hub.helpers.wait_for_completion(dataset)

You can optionally filter generation to specific topics in your knowledge base by passing topic_ids:

dataset = hub.datasets.generate_document_based(
project_id="project-id",
agent_id="agent-id",
knowledge_base_id="kb-id",
dataset_name="Shipping-only suite",
topic_ids=["shipping-topic-id"],
n_examples=10,
)

See Agents & Knowledge Bases for how to create and populate a Knowledge Base.


scenarios = hub.datasets.list_scenarios("dataset-id")
# Paginated search with filters
search_result = hub.datasets.search_scenarios(
"dataset-id",
query="payment",
limit=20,
offset=0,
)

# Move scenarios to a different dataset
hub.scenarios.bulk_move(
scenario_ids=["scenario-id-1", "scenario-id-2"],
target_dataset_id="other-dataset-id",
)
# Bulk update tags on multiple scenarios
hub.scenarios.bulk_update(
scenario_ids=["scenario-id-1", "scenario-id-2"],
added_tags=["reviewed"],
)
# Delete multiple scenarios
hub.scenarios.bulk_delete(scenario_ids=["scenario-id-1", "scenario-id-2"])

tags = hub.datasets.list_tags("dataset-id")
print(tags) # ["shipping", "faq", "reviewed"]

hub.datasets.update("dataset-id", name="Core Q&A Suite v2")
hub.datasets.delete("dataset-id")

Conformity and Groundedness default to the assistant message text (trace.last.outputs.response.content) when added to a scenario by identifier. To evaluate a field in a structured output, set the target path in params:

  • hub_conformity and hub_groundedness use target_key.
  • hub_correctness uses text_key.
  • hub_metadata uses metadata_key (default trace.last.outputs.metadata).
  • Other checks with a configurable target use target_key; see their defaults below.

For example, "target_key": "trace.last.outputs.category" evaluates the category field of a structured response.


Each built-in check can be used directly in scenarios by passing its identifier and the required params. Put params beside identifier, and put parameters such as rule inside params.

conformity is an alias for hub_conformity, and groundedness is an alias for hub_groundedness. Either identifier selects the same Hub check. The catalog and UI show one Conformity and one Groundedness entry.

IdentifierMethodWhat it evaluatesKey params
hub_correctnessLLM judgeDoes the response fully agree with the reference answer?reference
hub_conformityLLM judgeDoes the response comply with business rules?rule, target_key
hub_groundednessLLM judgeIs the response supported by reference information?context, context_key, target_key
llm_judgeLLM judgeEvaluate with a custom Jinja2 prompt returning pass or fail with a reason.prompt
contradictionLLM judgeDoes the response contradict a reference context?context
toxicityLLM judgeDoes the response contain toxic, harmful, or offensive content?categories
answer_relevanceLLM judgeDoes the response directly address the user question?(none required)
semantic_similarityEmbedding similarityIs the response semantically close to a reference?reference_text, threshold
string_matchingRule-basedDoes the response contain a given keyword or sentence?keyword
regex_matchingRule-basedDoes the response match a regular expression pattern?pattern
equals / not_equalsRule-basedDoes a value extracted from the trace equal (or differ from) the expected?expected_value
greater_than / greater_than_equalsRule-basedIs a numeric trace value greater than (or equal to) the expected value?expected_value
less_than / less_than_equalsRule-basedIs a numeric trace value less than (or equal to) the expected value?expected_value
hub_metadataRule-basedDo JSON path values in the response metadata satisfy specified conditions?json_path_rules
json_validRule-basedIs an extracted value valid JSON, optionally conforming to a JSON Schema?expected_schema
readabilityRule-basedDoes the response meet readability score thresholds for a chosen metric?metric

Each check is detailed below.

Validates that all information from the reference answer is present in the agent’s response, without contradiction. Uses an LLM judge.

ParameterTypeDescription
referencestrThe expected correct answer
{
"identifier": "hub_correctness",
"params": {"reference": "We offer a 30-day return policy."},
}

Checks that the agent’s response follows the instructions in rule. You can include several requirements in one string. The Hub uses its LLM judge to evaluate them together.

ParameterTypeDescription
rulestrRequired instructions the response must follow
target_keystrTrace path to evaluate. Defaults to trace.last.outputs.response.content; use trace for the full trace
{
"identifier": "hub_conformity",
"params": {
"rule": (
"- Use a formal, professional tone.\n"
"- Do not include personal opinions."
)
},
}

conformity is accepted as an identifier alias.

Checks that all information in the agent’s response is supported by context, without contradiction. Unlike Correctness, omissions are allowed, but extra or conflicting claims fail the check, so it is useful for catching hallucinations.

ParameterTypeDescription
contextstr / list[str]Optional reference text or documents. If provided, takes precedence over context_key
context_keystrTrace path used when context is omitted. Defaults to trace.last.outputs.metadata
target_keystrTrace path of the answer. Defaults to trace.last.outputs.response.content
answerstrOptional fixed answer. If provided, takes precedence over target_key
{
"identifier": "hub_groundedness",
"params": {
"context": "Our return window is 30 days. We do not accept returns on clearance items."
},
}

To read retrieved documents from the trace, omit context and set context_key:

{
"identifier": "hub_groundedness",
"params": {"context_key": "trace.last.outputs.metadata.retrieved_chunks"},
}

groundedness is accepted as an identifier alias.

Evaluates the interaction with a custom prompt. The prompt is a Jinja2 template with access to the trace (use trace.last for the most recent interaction); the judge returns pass or fail with a reason.

ParameterTypeDescription
promptstrJinja2 prompt template referencing trace values
{
"identifier": "llm_judge",
"params": {
"prompt": "The user asked: {{ trace.last.inputs.messages[-1].content }}\nThe agent answered: {{ trace.last.outputs.response.content }}\n\nDoes the answer avoid making promises about delivery dates?"
},
}

Checks that the response does not directly contradict a reference context. Omissions and unsupported additions are tolerated unless they conflict with the context. Uses an LLM judge.

ParameterTypeDescription
contextstr / list[str]Reference context provided directly
context_keystrTrace path to extract the context from
{
"identifier": "contradiction",
"params": {"context": "Our return window is 30 days."},
}

Checks that the response does not contain toxic, harmful, or offensive content. Uses an LLM judge.

ParameterTypeDescription
categorieslist[str]Safety categories to check: hate_speech, harassment, threats, self_harm, sexual_content, violence
{
"identifier": "toxicity",
"params": {"categories": ["hate_speech", "threats"]},
}

Checks that the response directly and appropriately addresses the user question. Uses an LLM judge. No parameters are required; by default the question is taken from the conversation.

ParameterTypeDescription
questionstrQuestion provided directly (optional)
include_historyboolInclude prior turns when judging the response
{"identifier": "answer_relevance"}

Computes embedding-based similarity between the agent’s response and a reference string. The check passes if the similarity score meets or exceeds the threshold. Does not use an LLM judge.

ParameterTypeDescription
reference_textstrThe expected output to compare against
thresholdfloatSimilarity threshold (0.0 to 1.0)
{
"identifier": "semantic_similarity",
"params": {"reference_text": "30-day return policy", "threshold": 0.8},
}

Checks whether the agent’s response contains a specific keyword or substring. Case-sensitive by default; pass case_sensitive: False to lowercase both sides before comparison. Does not use an LLM judge.

ParameterTypeDescription
keywordstrThe keyword or substring to search for
case_sensitiveboolMatch case exactly (default: True)
{"identifier": "string_matching", "params": {"keyword": "#12345"}}

Checks whether the agent’s response matches a regular expression pattern. Does not use an LLM judge.

ParameterTypeDescription
patternstrThe regular expression to match with
{"identifier": "regex_matching", "params": {"pattern": r"#\d{5}"}}

Six rule-based checks compare a value extracted from the trace against an expected value: equals, not_equals, greater_than, greater_than_equals, less_than, less_than_equals. They are the natural fit for structured agent outputs and numeric metadata. The numeric checks default their target_key to trace.last.outputs.metadata.score.

ParameterTypeDescription
expected_valuescalarThe value to compare against
target_keystrTrace path of the value under test
matchstrFor list values: "any", "all", or "none"
{
"identifier": "equals",
"params": {
"target_key": "trace.last.outputs.category",
"expected_value": "billing",
},
}
{
"identifier": "greater_than_equals",
"params": {"expected_value": 0.5}, # reads trace.last.outputs.metadata.score
}

Validates values extracted via JSON path expressions from the response metadata. Useful for verifying structured outputs like tool calls, categories, or flags. Does not use an LLM judge.

ParameterTypeDescription
json_path_ruleslist[dict]List of rules, each with json_path, expected_value, and expected_value_type

Each rule dict supports:

KeyTypeDescription
json_pathstrJSON path expression (e.g. $.category, $.tools_called[0])
expected_valuestr / number / boolThe expected value
expected_value_typestrType of the expected value ("string", "number", "boolean")
{
"identifier": "hub_metadata",
"params": {
"json_path_rules": [
{
"json_path": "$.category",
"expected_value": "billing",
"expected_value_type": "string",
},
{
"json_path": "$.resolved",
"expected_value": True,
"expected_value_type": "boolean",
},
]
},
}

Checks that a value extracted from the trace is valid JSON and, optionally, that it conforms to a JSON Schema. Does not use an LLM judge.

ParameterTypeDescription
expected_schemadictJSON Schema the value must conform to (optional)
parseboolParse the value from a string before validating
{
"identifier": "json_valid",
"params": {
"expected_schema": {
"type": "object",
"properties": {"category": {"type": "string"}},
"required": ["category"],
}
},
}

Checks that the response satisfies readability score thresholds for a selected metric. Does not use an LLM judge.

ParameterTypeDescription
metricstrOne of flesch_reading_ease, flesch_kincaid_grade, gunning_fog, automated_readability_index, coleman_liau_index, dale_chall_readability_score
min_scorefloatMinimum acceptable score (optional)
max_scorefloatMaximum acceptable score (optional)
{
"identifier": "readability",
"params": {"metric": "flesch_reading_ease", "min_score": 60},
}

Custom checks are pre-configured versions of the built-in check types. Instead of repeating the same params in every scenario, you define the configuration once — giving it a project-scoped identifier (which must start with custom_), a name, and the check params — and then reference it by identifier wherever it’s needed.

check = hub.checks.create(
project_id="project-id",
identifier="custom_tone_professional",
name="Professional tone",
description="The response must use formal, professional language with no slang.",
params={
"type": "hub_conformity",
"rule": "The response must be written in a formal, professional tone. It must not contain slang, contractions, or casual phrasing.",
},
)
print(check.id)

Once created, reference your custom check by its identifier in any scenario within the same project — no need to repeat the params:

hub.scenarios.create(
dataset_id="dataset-id",
interactions=[
{
"input": {
"messages": [{"role": "user", "content": "hey, can u help me?"}]
},
"checks": [{"identifier": "custom_tone_professional"}],
}
],
)

Content safety check:

hub.checks.create(
project_id="project-id",
identifier="custom_no_harmful_content",
name="No harmful content",
description="The response must not contain harmful, violent, or offensive content.",
params={
"type": "hub_conformity",
"rule": "The response must be safe for all audiences. It must not contain violence, hate speech, sexual content, or self-harm.",
},
)

Tool-call verification (metadata check):

hub.checks.create(
project_id="project-id",
identifier="custom_used_search_tool",
name="Search tool was called",
description="Verifies that the agent called the search tool during the response.",
params={
"type": "hub_metadata",
"json_path_rules": [
{
"json_path": "$.tools_called[0]",
"expected_value": "search",
"expected_value_type": "string",
},
],
},
)

By default, hub.checks.list() returns only your custom checks. Built-in checks are referenced directly by identifier and are not listed; pass filter_builtin=False to include them.

checks = hub.checks.list(project_id="project-id")
all_checks = hub.checks.list(project_id="project-id", filter_builtin=False)
hub.checks.update("check-id", name="Updated name")
hub.checks.delete("check-id")