Run one KnowQL query turn
Send ask and, on a new session, a scope of 1–10 contexts; continue an existing one with
session_id or previous_query_id. Scoped search contexts must be curated, and a scope may
not mix work and search contexts. Read the answer from output[].content[].text.
stream and background are mutually exclusive. background: true returns 202 with an
in_progress query to poll. stream: true emits an SSE stream whose event: names are the
event type (see QueryEvent), closed by a final query frame carrying the whole turn:
id: 1
event: response.created
data: {"type":"response.created","query_id":"qry_8b3d5f1a","session_id":"ses_4f9c1e2a"}
id: 2
event: response.output_text.delta
data: {"type":"response.output_text.delta","query_id":"qry_8b3d5f1a","delta":"Revenue grew 12%..."}
id: 3
event: response.completed
data: {"type":"response.completed","query_id":"qry_8b3d5f1a"}
event: query
data: {"id":"qry_8b3d5f1a","object":"query","status":"completed","output":[...]}POST /query
curl --request POST \
--url https://{host}/api/query \
--header 'Authorization: Bearer <token>' \
--header 'X-Pinecone-Api-Version: <x-pinecone-api-version>' \
--header 'Content-Type: application/json' \
--data '{
"ask": "<string>",
"scope": [
"<string>"
],
"session_id": "<string>",
"previous_query_id": "<string>",
"workflow": "<string>",
"system_prompt": "<string>",
"guardrails": "<string>",
"shape": {},
"model": "<string>",
"models": [
"<string>"
],
"tools": [
"<string>"
],
"stream": true,
"background": true,
"timeout_seconds": 123,
"max_steps": 123,
"thinking_level": "<string>",
"compose": true,
"retrieval_only": true,
"pointers_only": true,
"chunks_only": true,
"artifacts_only": true,
"max_retrieved": 123,
"max_retrieved_chars": 123,
"comparison_group": "<string>"
}'import requests
url = "https://{host}/api/query"
payload = {
"ask": "<string>",
"scope": [
"<string>"
],
"session_id": "<string>",
"previous_query_id": "<string>",
"workflow": "<string>",
"system_prompt": "<string>",
"guardrails": "<string>",
"shape": {},
"model": "<string>",
"models": [
"<string>"
],
"tools": [
"<string>"
],
"stream": True,
"background": True,
"timeout_seconds": 123,
"max_steps": 123,
"thinking_level": "<string>",
"compose": True,
"retrieval_only": True,
"pointers_only": True,
"chunks_only": True,
"artifacts_only": True,
"max_retrieved": 123,
"max_retrieved_chars": 123,
"comparison_group": "<string>"
}
headers = {
"Authorization": "Bearer <token>",
"X-Pinecone-Api-Version": "<x-pinecone-api-version>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {method: "POST", headers: {"Authorization": "Bearer <token>", "X-Pinecone-Api-Version": "<x-pinecone-api-version>", "Content-Type": "application/json"}, body: JSON.stringify({
"ask": "<string>",
"scope": [
"<string>"
],
"session_id": "<string>",
"previous_query_id": "<string>",
"workflow": "<string>",
"system_prompt": "<string>",
"guardrails": "<string>",
"shape": {},
"model": "<string>",
"models": [
"<string>"
],
"tools": [
"<string>"
],
"stream": true,
"background": true,
"timeout_seconds": 123,
"max_steps": 123,
"thinking_level": "<string>",
"compose": true,
"retrieval_only": true,
"pointers_only": true,
"chunks_only": true,
"artifacts_only": true,
"max_retrieved": 123,
"max_retrieved_chars": 123,
"comparison_group": "<string>"
})};
fetch("https://{host}/api/query", options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://{host}/api/query",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => "{\"ask\":\"<string>\",\"scope\":[\"<string>\"],\"session_id\":\"<string>\",\"previous_query_id\":\"<string>\",\"workflow\":\"<string>\",\"system_prompt\":\"<string>\",\"guardrails\":\"<string>\",\"shape\":{},\"model\":\"<string>\",\"models\":[\"<string>\"],\"tools\":[\"<string>\"],\"stream\":true,\"background\":true,\"timeout_seconds\":123,\"max_steps\":123,\"thinking_level\":\"<string>\",\"compose\":true,\"retrieval_only\":true,\"pointers_only\":true,\"chunks_only\":true,\"artifacts_only\":true,\"max_retrieved\":123,\"max_retrieved_chars\":123,\"comparison_group\":\"<string>\"}",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"X-Pinecone-Api-Version: <x-pinecone-api-version>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://{host}/api/query"
payload := strings.NewReader("{\"ask\":\"<string>\",\"scope\":[\"<string>\"],\"session_id\":\"<string>\",\"previous_query_id\":\"<string>\",\"workflow\":\"<string>\",\"system_prompt\":\"<string>\",\"guardrails\":\"<string>\",\"shape\":{},\"model\":\"<string>\",\"models\":[\"<string>\"],\"tools\":[\"<string>\"],\"stream\":true,\"background\":true,\"timeout_seconds\":123,\"max_steps\":123,\"thinking_level\":\"<string>\",\"compose\":true,\"retrieval_only\":true,\"pointers_only\":true,\"chunks_only\":true,\"artifacts_only\":true,\"max_retrieved\":123,\"max_retrieved_chars\":123,\"comparison_group\":\"<string>\"}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("X-Pinecone-Api-Version", "<x-pinecone-api-version>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://{host}/api/query")
.header("Authorization", "Bearer <token>")
.header("X-Pinecone-Api-Version", "<x-pinecone-api-version>")
.header("Content-Type", "application/json")
.body("{\"ask\":\"<string>\",\"scope\":[\"<string>\"],\"session_id\":\"<string>\",\"previous_query_id\":\"<string>\",\"workflow\":\"<string>\",\"system_prompt\":\"<string>\",\"guardrails\":\"<string>\",\"shape\":{},\"model\":\"<string>\",\"models\":[\"<string>\"],\"tools\":[\"<string>\"],\"stream\":true,\"background\":true,\"timeout_seconds\":123,\"max_steps\":123,\"thinking_level\":\"<string>\",\"compose\":true,\"retrieval_only\":true,\"pointers_only\":true,\"chunks_only\":true,\"artifacts_only\":true,\"max_retrieved\":123,\"max_retrieved_chars\":123,\"comparison_group\":\"<string>\"}")
.asString();require 'uri'
require 'net/http'
url = URI("https://{host}/api/query")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["X-Pinecone-Api-Version"] = '<x-pinecone-api-version>'
request["Content-Type"] = 'application/json'
request.body = "{\"ask\":\"<string>\",\"scope\":[\"<string>\"],\"session_id\":\"<string>\",\"previous_query_id\":\"<string>\",\"workflow\":\"<string>\",\"system_prompt\":\"<string>\",\"guardrails\":\"<string>\",\"shape\":{},\"model\":\"<string>\",\"models\":[\"<string>\"],\"tools\":[\"<string>\"],\"stream\":true,\"background\":true,\"timeout_seconds\":123,\"max_steps\":123,\"thinking_level\":\"<string>\",\"compose\":true,\"retrieval_only\":true,\"pointers_only\":true,\"chunks_only\":true,\"artifacts_only\":true,\"max_retrieved\":123,\"max_retrieved_chars\":123,\"comparison_group\":\"<string>\"}"
response = http.request(request)
puts response.read_body{
"id": "<string>",
"object": "<string>",
"session_id": "<string>",
"model": "<string>",
"created": 123,
"status": "<string>",
"error": "<string>",
"previous_query_id": "<string>",
"comparison": [
{
"workflow": "<string>",
"query_id": "<string>",
"model": "<string>"
}
],
"feedback": {
"rating": "<string>",
"comment": "<string>"
},
"input": [
{
"role": "<string>",
"content": "<string>"
}
],
"output": [
{
"role": "<string>",
"content": [
{
"type": null,
"text": null
}
]
}
],
"output_json": {},
"citations": [
{
"source": "<string>",
"section_paths": [
[
null
]
],
"pages": [
123
],
"score": 123,
"grounding": "<string>",
"kind": "<string>",
"source_path": "<string>",
"query_id": "<string>",
"steps": [
"<string>"
],
"artifact_name": "<string>",
"sources": [
"<string>"
]
}
],
"max_steps": 123,
"thinking_level": "<string>",
"steps": [
{
"step_id": "<string>",
"status": "<string>",
"commentary": "<string>",
"fns": [
"<string>"
],
"strategy": {
"kind": "<string>",
"fns": [
null
],
"label": "<string>",
"scope": [
null
]
},
"tool_calls": [
null
],
"code": "<string>",
"trace_truncated": {
"truncated": true,
"dropped_documents": 123,
"dropped_tool_calls": 123
},
"cost": {
"tokens_in": 123,
"tokens_out": 123,
"decide_ms": 123,
"execute_ms": 123,
"tokens_in_cached": 123,
"tokens_in_cache_write": 123,
"tokens_in_fresh": 123
},
"input_tokens": 123,
"output_tokens": 123,
"total_tokens": 123,
"cum_input_tokens": 123,
"cum_output_tokens": 123
}
],
"rollup": {
"type": "<string>",
"query_id": "<string>",
"n_steps": 123,
"n_tool_calls": 123,
"by_category": {},
"total_hits": 123,
"duration_ms": 123,
"cache_read_tokens": 123,
"cache_write_tokens": 123
},
"synthesis": {
"type": "<string>",
"query_id": "<string>",
"status": "<string>",
"tokens_in": 123,
"tokens_out": 123,
"ms": 123,
"answer_preview": "<string>",
"tokens_in_cached": 123,
"tokens_in_cache_write": 123,
"tokens_in_fresh": 123
},
"trace_ref": "<string>",
"usage": {
"input_tokens": 123,
"output_tokens": 123,
"total_tokens": 123
},
"runtime_ms": 123
}{
"id": "<string>",
"object": "<string>",
"session_id": "<string>",
"model": "<string>",
"created": 123,
"status": "<string>",
"error": "<string>",
"previous_query_id": "<string>",
"comparison": [
{
"workflow": "<string>",
"query_id": "<string>",
"model": "<string>"
}
],
"feedback": {
"rating": "<string>",
"comment": "<string>"
},
"input": [
{
"role": "<string>",
"content": "<string>"
}
],
"output": [
{
"role": "<string>",
"content": [
{
"type": null,
"text": null
}
]
}
],
"output_json": {},
"citations": [
{
"source": "<string>",
"section_paths": [
[
null
]
],
"pages": [
123
],
"score": 123,
"grounding": "<string>",
"kind": "<string>",
"source_path": "<string>",
"query_id": "<string>",
"steps": [
"<string>"
],
"artifact_name": "<string>",
"sources": [
"<string>"
]
}
],
"max_steps": 123,
"thinking_level": "<string>",
"steps": [
{
"step_id": "<string>",
"status": "<string>",
"commentary": "<string>",
"fns": [
"<string>"
],
"strategy": {
"kind": "<string>",
"fns": [
null
],
"label": "<string>",
"scope": [
null
]
},
"tool_calls": [
null
],
"code": "<string>",
"trace_truncated": {
"truncated": true,
"dropped_documents": 123,
"dropped_tool_calls": 123
},
"cost": {
"tokens_in": 123,
"tokens_out": 123,
"decide_ms": 123,
"execute_ms": 123,
"tokens_in_cached": 123,
"tokens_in_cache_write": 123,
"tokens_in_fresh": 123
},
"input_tokens": 123,
"output_tokens": 123,
"total_tokens": 123,
"cum_input_tokens": 123,
"cum_output_tokens": 123
}
],
"rollup": {
"type": "<string>",
"query_id": "<string>",
"n_steps": 123,
"n_tool_calls": 123,
"by_category": {},
"total_hits": 123,
"duration_ms": 123,
"cache_read_tokens": 123,
"cache_write_tokens": 123
},
"synthesis": {
"type": "<string>",
"query_id": "<string>",
"status": "<string>",
"tokens_in": 123,
"tokens_out": 123,
"ms": 123,
"answer_preview": "<string>",
"tokens_in_cached": 123,
"tokens_in_cache_write": 123,
"tokens_in_fresh": 123
},
"trace_ref": "<string>",
"usage": {
"input_tokens": 123,
"output_tokens": 123,
"total_tokens": 123
},
"runtime_ms": 123
}Authorizations
Section titled “Authorizations”AuthorizationstringrequiredSession token from POST /auth/login, sent as Authorization: Bearer <token>.
Headers
Section titled “Headers”X-Pinecone-Api-Version?stringDate-based contract version, echoed back on the same header. Omit for the default (2026-07); send unstable for the in-development surface. An unrecognized value is rejected with 400 unsupported_api_version.
askstringrequiredThe natural-language question
scope?string[]Context slugs/UUIDs. New session only; pinned for the session's life. A scope may not mix work and search contexts.
session_id?stringContinue an existing session
previous_query_id?stringContinue the session this query belongs to
workflow?stringSearch workflow for this turn. Ignored for work contexts, which always run the work runtime.
system_prompt?stringInstructions pinned to a new session
guardrails?stringGuardrails pinned to a new session
shape?objectJSON Schema subset for structured output; result in output_json
model?stringA catalog model id from GET /models, or a tier name (lite, standard, pro) to let the deployment resolve one. New session only; pinned for the session's life.
models?string[]Ordered fallback list, tried in turn. Takes precedence over model.
tools?string[]Tool names the turn may call. New session only; pinned for the session's life.
stream?booleanSSE streaming. Mutually exclusive with background.
background?booleanFire-and-forget: 202 + in_progress query; poll GET /queries/id. Mutually exclusive with stream.
timeout_seconds?integerMay only LOWER the 15-minute (900s) cap
max_steps?integerCap the agent's tool-loop steps for this turn
thinking_level?stringGemini reasoning depth. Default low. Gemini-backed workflows only; ignored for search_cc.
compose?booleanSet false to skip synthesis, the same as retrieval_only.
retrieval_only?booleanSkip synthesis; return retrieved hits in output_json
pointers_only?booleanSkip synthesis; return just pointers in output_json
chunks_only?booleanRetrieval-only, narrowed to chunks
artifacts_only?booleanRetrieval-only, narrowed to artifacts
max_retrieved?integerCap the item count for retrieval-only turns
max_retrieved_chars?integerCap per-item verbatim text length for retrieval-only turns
comparison_group?stringClient-generated id shared by the turns of one Compare run, so a query cap counts them as one action rather than several. Where the cap applies, a group is limited to 3 turns and a fourth is refused with 409. Omit for a normal query.
Response
Section titled “Response”200 — The completed query turn (synchronous), or the SSE stream when stream=true
One query turn. Read the answer from output[].content[].text.
idstringrequiredTurn id, qry_<uuid>.
objectstringrequiredAlways query, so a caller can tell this document apart from a session.
session_idstringrequiredThe session this turn belongs to. A turn that started a new one names it here.
model?string | nullThe model that actually answered. Null before the turn resolves one.
createdintegerrequiredUnix seconds
statusstringrequiredWhere the turn is in its lifecycle.
error?string | nullFailure detail. Set on a failed turn.
previous_query_id?string | nullThe turn this one continues. Null on a session's first turn.
comparison?object[] | nullThe sibling turns of the same Compare run. Null on a normal single query.
Show child attributes
workflow?stringThe workflow that turn ran.
query_id?stringThat turn's id. Fetch it with GET /queries/id.
model?stringThe model that answered it.
feedback?object | nullThumbs-up/down recorded on this turn. Null until a caller submits some.
Show child attributes
rating?stringThumbs up or down.
comment?string | nullFree-text note. Null when none was given.
inputobject[]requiredThe stored message array (plain-text content).
Show child attributes
rolestringrequiredWho sent the message, e.g. user.
contentstringrequiredThe message text.
outputobject[]requiredOutput items; assistant text is role, content:[type: output_text, text]
Show child attributes
rolestringrequiredWho produced it, e.g. assistant.
contentobject[]requiredThe item's content parts.
Show child attributes
typestringrequiredPart kind. output_text carries the answer.
textstringrequiredThe text of this part.
output_json?object | nullPresent when a shape was used, or on a retrieval-only turn
citationsobject[]requiredWhat the answer was grounded in. Empty on a turn that cited nothing.
Show child attributes
source?stringPath of the cited corpus file, relative to the source-tree root.
section_paths?string[][]Heading paths within the source, when known. One inner array per cited section, outermost heading first.
pages?integer[]1-based page numbers, for sources that paginate.
score?numberRetrieval score — how well this source answered the ask. Comparable only within one turn.
grounding?stringThe quoted span the answer rests on.
kind?stringWhich index the citation came out of, e.g. chunk or artifact.
source_path?stringWork contexts — path of the artifact the fact was consolidated into.
query_id?stringWork contexts — the earlier turn the fact was learned from.
steps?string[]Work contexts — step ids within that earlier turn.
artifact_name?stringWork contexts — name of the cited artifact.
sources?string[]The documents a cited artifact was derived from.
max_steps?integer | nullThe agentic tool-loop step cap this turn actually ran with, whether the request set it or the deployment default supplied it.
thinking_level?string | nullThe thinking level this turn actually ran at. Null on the Claude-backed search_cc, which has no equivalent knob.
stepsobject[]requiredThe turn's reasoning steps, reduced from its response.step events.
Show child attributes
step_idstringrequiredIdentifies the step within the turn. Clients merge repeated events by it.
statusstringrequiredHow far the step has got.
commentarystringrequiredThe model's own one-line account of what this step is doing.
fns?string[]Tool functions this step called.
strategy?objectThe retrieval approach the step picked.
Show child attributes
kind?stringStrategy family, e.g. artifacts_first.
fns?string[]Tool functions the strategy calls.
label?stringDisplay label for the strategy.
scope?string[]Context ids the strategy searched.
tool_calls?object[]Per-call detail for the step's tool calls.
Show child attributes
fn?stringName of the function this call invoked.
category?stringWhich family the tool belongs to, as tallied in the rollup's by_category.
args?objectCompact, redacted summary of the call's arguments. The key set is tool-specific and deliberately open.
result?anyCompact summary of the call's result (never the payload). The shape varies by category and is runtime-extensible.
duration_ms?integerWall time for this one call.
ok?booleanWhether the call succeeded.
score_space?stringWhich scoring space the returned scores live in, for calls that retrieve.
error?stringFailure detail. Set when ok is false.
code?stringThe source string alone, as later rows store it.
trace_truncated?objectWhat a clamp shed from an oversized step event.
Show child attributes
truncated?booleanWhether the clamp fired at all.
dropped_documents?integerRetrieved documents shed from the event.
dropped_tool_calls?integerHow many calls the clamp discarded.
cost?objectThe step's incremental token and latency cost — deltas against the running turn cursor, not totals. The cache fields are tracked only on the search-as-code path.
Show child attributes
tokens_in?integerThe step's input tokens.
tokens_out?integerThe step's output tokens.
decide_ms?integerTime spent choosing what to do.
execute_ms?integerTime spent running the tool calls it chose.
tokens_in_cached?integerInput tokens served from the prompt cache.
tokens_in_cache_write?integerInput tokens written into the prompt cache.
tokens_in_fresh?integerInput tokens neither cached nor cache-written.
input_tokens?integerPrompt tokens billed by this one step.
output_tokens?integerCompletion tokens billed by this one step.
total_tokens?integerPrompt plus completion for this one step.
cum_input_tokens?integerTurn-to-date input tokens, including this step.
cum_output_tokens?integerTurn-to-date output tokens, including this step.
rollup?object | nullEnd-of-turn counters. Null on a turn the runtime never closed.
Show child attributes
type?stringEvent form only.
query_id?stringEvent form only.
n_steps?integerReasoning steps the turn ran.
n_tool_calls?integerHow many calls the whole turn made.
by_category?objectTool calls tallied by category.
total_hits?integerRetrieved items across all calls, before dedup.
duration_ms?integerWall time for the whole turn.
cache_read_tokens?integerInput tokens served from the prompt cache.
cache_write_tokens?integerInput tokens written into the prompt cache.
synthesis?object | nullThe answer completion's cost. Null on a turn that skipped synthesis, as retrieval-only turns do.
Show child attributes
type?stringEvent form only.
query_id?stringEvent form only.
status?stringHow synthesis ended, e.g. completed.
tokens_in?integerThe completion's input tokens.
tokens_out?integerThe completion's output tokens.
ms?integerWall time of the completion.
answer_preview?stringLeading characters of the answer, for a progress display that has no full text yet.
tokens_in_cached?integerInput tokens served from the prompt cache.
tokens_in_cache_write?integerInput tokens written into the prompt cache.
tokens_in_fresh?integerInput tokens neither cached nor cache-written.
trace_ref?string | nullBlob key of the persisted trace. Fetch the trace itself from GET /queries/id/trace.
usageobjectrequiredThe turn's token totals. Embed and rerank bill on their own seam and are not counted here.
Show child attributes
input_tokensintegerrequiredInput tokens the turn billed.
output_tokensintegerrequiredOutput tokens the turn billed.
total_tokensintegerrequiredInput plus output tokens.
runtime_msintegerrequiredWall time from turn start to terminal state.