Cancel an in-progress query turn
Idempotent — an already-terminal query is returned unchanged. On success the turn's status becomes cancelled.
POST /queries/{id}/cancel
curl --request POST \
--url https://{host}/api/queries/{id}/cancel \
--header 'Authorization: Bearer <token>' \
--header 'X-Pinecone-Api-Version: <x-pinecone-api-version>'import requests
url = "https://{host}/api/queries/{id}/cancel"
headers = {
"Authorization": "Bearer <token>",
"X-Pinecone-Api-Version": "<x-pinecone-api-version>"
}
response = requests.post(url, headers=headers)
print(response.text)const options = {method: "POST", headers: {"Authorization": "Bearer <token>", "X-Pinecone-Api-Version": "<x-pinecone-api-version>"}};
fetch("https://{host}/api/queries/{id}/cancel", options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://{host}/api/queries/{id}/cancel",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"X-Pinecone-Api-Version: <x-pinecone-api-version>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://{host}/api/queries/{id}/cancel"
req, _ := http.NewRequest("POST", url, nil)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("X-Pinecone-Api-Version", "<x-pinecone-api-version>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://{host}/api/queries/{id}/cancel")
.header("Authorization", "Bearer <token>")
.header("X-Pinecone-Api-Version", "<x-pinecone-api-version>")
.asString();require 'uri'
require 'net/http'
url = URI("https://{host}/api/queries/{id}/cancel")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["X-Pinecone-Api-Version"] = '<x-pinecone-api-version>'
response = http.request(request)
puts response.read_body{
"id": "<string>",
"object": "<string>",
"session_id": "<string>",
"model": "<string>",
"created": 123,
"status": "<string>",
"error": "<string>",
"previous_query_id": "<string>",
"comparison": [
{
"workflow": "<string>",
"query_id": "<string>",
"model": "<string>"
}
],
"feedback": {
"rating": "<string>",
"comment": "<string>"
},
"input": [
{
"role": "<string>",
"content": "<string>"
}
],
"output": [
{
"role": "<string>",
"content": [
{
"type": null,
"text": null
}
]
}
],
"output_json": {},
"citations": [
{
"source": "<string>",
"section_paths": [
[
null
]
],
"pages": [
123
],
"score": 123,
"grounding": "<string>",
"kind": "<string>",
"source_path": "<string>",
"query_id": "<string>",
"steps": [
"<string>"
],
"artifact_name": "<string>",
"sources": [
"<string>"
]
}
],
"max_steps": 123,
"thinking_level": "<string>",
"steps": [
{
"step_id": "<string>",
"status": "<string>",
"commentary": "<string>",
"fns": [
"<string>"
],
"strategy": {
"kind": "<string>",
"fns": [
null
],
"label": "<string>",
"scope": [
null
]
},
"tool_calls": [
null
],
"code": "<string>",
"trace_truncated": {
"truncated": true,
"dropped_documents": 123,
"dropped_tool_calls": 123
},
"cost": {
"tokens_in": 123,
"tokens_out": 123,
"decide_ms": 123,
"execute_ms": 123,
"tokens_in_cached": 123,
"tokens_in_cache_write": 123,
"tokens_in_fresh": 123
},
"input_tokens": 123,
"output_tokens": 123,
"total_tokens": 123,
"cum_input_tokens": 123,
"cum_output_tokens": 123
}
],
"rollup": {
"type": "<string>",
"query_id": "<string>",
"n_steps": 123,
"n_tool_calls": 123,
"by_category": {},
"total_hits": 123,
"duration_ms": 123,
"cache_read_tokens": 123,
"cache_write_tokens": 123
},
"synthesis": {
"type": "<string>",
"query_id": "<string>",
"status": "<string>",
"tokens_in": 123,
"tokens_out": 123,
"ms": 123,
"answer_preview": "<string>",
"tokens_in_cached": 123,
"tokens_in_cache_write": 123,
"tokens_in_fresh": 123
},
"trace_ref": "<string>",
"usage": {
"input_tokens": 123,
"output_tokens": 123,
"total_tokens": 123
},
"runtime_ms": 123
}{
"message": "<string>",
"code": "<string>"
}Authorizations
Section titled “Authorizations”AuthorizationstringrequiredSession token from POST /auth/login, sent as Authorization: Bearer <token>.
Headers
Section titled “Headers”X-Pinecone-Api-Version?stringDate-based contract version, echoed back on the same header. Omit for the default (2026-07); send unstable for the in-development surface. An unrecognized value is rejected with 400 unsupported_api_version.
Path Parameters
Section titled “Path Parameters”idstringrequiredQuery turn id.
Response
Section titled “Response”200 — The (now cancelled) query
One query turn. Read the answer from output[].content[].text.
idstringrequiredTurn id, qry_<uuid>.
objectstringrequiredAlways query, so a caller can tell this document apart from a session.
session_idstringrequiredThe session this turn belongs to. A turn that started a new one names it here.
model?string | nullThe model that actually answered. Null before the turn resolves one.
createdintegerrequiredUnix seconds
statusstringrequiredWhere the turn is in its lifecycle.
error?string | nullFailure detail. Set on a failed turn.
previous_query_id?string | nullThe turn this one continues. Null on a session's first turn.
comparison?object[] | nullThe sibling turns of the same Compare run. Null on a normal single query.
Show child attributes
workflow?stringThe workflow that turn ran.
query_id?stringThat turn's id. Fetch it with GET /queries/id.
model?stringThe model that answered it.
feedback?object | nullThumbs-up/down recorded on this turn. Null until a caller submits some.
Show child attributes
rating?stringThumbs up or down.
comment?string | nullFree-text note. Null when none was given.
inputobject[]requiredThe stored message array (plain-text content).
Show child attributes
rolestringrequiredWho sent the message, e.g. user.
contentstringrequiredThe message text.
outputobject[]requiredOutput items; assistant text is role, content:[type: output_text, text]
Show child attributes
rolestringrequiredWho produced it, e.g. assistant.
contentobject[]requiredThe item's content parts.
Show child attributes
typestringrequiredPart kind. output_text carries the answer.
textstringrequiredThe text of this part.
output_json?object | nullPresent when a shape was used, or on a retrieval-only turn
citationsobject[]requiredWhat the answer was grounded in. Empty on a turn that cited nothing.
Show child attributes
source?stringPath of the cited corpus file, relative to the source-tree root.
section_paths?string[][]Heading paths within the source, when known. One inner array per cited section, outermost heading first.
pages?integer[]1-based page numbers, for sources that paginate.
score?numberRetrieval score — how well this source answered the ask. Comparable only within one turn.
grounding?stringThe quoted span the answer rests on.
kind?stringWhich index the citation came out of, e.g. chunk or artifact.
source_path?stringWork contexts — path of the artifact the fact was consolidated into.
query_id?stringWork contexts — the earlier turn the fact was learned from.
steps?string[]Work contexts — step ids within that earlier turn.
artifact_name?stringWork contexts — name of the cited artifact.
sources?string[]The documents a cited artifact was derived from.
max_steps?integer | nullThe agentic tool-loop step cap this turn actually ran with, whether the request set it or the deployment default supplied it.
thinking_level?string | nullThe thinking level this turn actually ran at. Null on the Claude-backed search_cc, which has no equivalent knob.
stepsobject[]requiredThe turn's reasoning steps, reduced from its response.step events.
Show child attributes
step_idstringrequiredIdentifies the step within the turn. Clients merge repeated events by it.
statusstringrequiredHow far the step has got.
commentarystringrequiredThe model's own one-line account of what this step is doing.
fns?string[]Tool functions this step called.
strategy?objectThe retrieval approach the step picked.
Show child attributes
kind?stringStrategy family, e.g. artifacts_first.
fns?string[]Tool functions the strategy calls.
label?stringDisplay label for the strategy.
scope?string[]Context ids the strategy searched.
tool_calls?object[]Per-call detail for the step's tool calls.
Show child attributes
fn?stringName of the function this call invoked.
category?stringWhich family the tool belongs to, as tallied in the rollup's by_category.
args?objectCompact, redacted summary of the call's arguments. The key set is tool-specific and deliberately open.
result?anyCompact summary of the call's result (never the payload). The shape varies by category and is runtime-extensible.
duration_ms?integerWall time for this one call.
ok?booleanWhether the call succeeded.
score_space?stringWhich scoring space the returned scores live in, for calls that retrieve.
error?stringFailure detail. Set when ok is false.
code?stringThe source string alone, as later rows store it.
trace_truncated?objectWhat a clamp shed from an oversized step event.
Show child attributes
truncated?booleanWhether the clamp fired at all.
dropped_documents?integerRetrieved documents shed from the event.
dropped_tool_calls?integerHow many calls the clamp discarded.
cost?objectThe step's incremental token and latency cost — deltas against the running turn cursor, not totals. The cache fields are tracked only on the search-as-code path.
Show child attributes
tokens_in?integerThe step's input tokens.
tokens_out?integerThe step's output tokens.
decide_ms?integerTime spent choosing what to do.
execute_ms?integerTime spent running the tool calls it chose.
tokens_in_cached?integerInput tokens served from the prompt cache.
tokens_in_cache_write?integerInput tokens written into the prompt cache.
tokens_in_fresh?integerInput tokens neither cached nor cache-written.
input_tokens?integerPrompt tokens billed by this one step.
output_tokens?integerCompletion tokens billed by this one step.
total_tokens?integerPrompt plus completion for this one step.
cum_input_tokens?integerTurn-to-date input tokens, including this step.
cum_output_tokens?integerTurn-to-date output tokens, including this step.
rollup?object | nullEnd-of-turn counters. Null on a turn the runtime never closed.
Show child attributes
type?stringEvent form only.
query_id?stringEvent form only.
n_steps?integerReasoning steps the turn ran.
n_tool_calls?integerHow many calls the whole turn made.
by_category?objectTool calls tallied by category.
total_hits?integerRetrieved items across all calls, before dedup.
duration_ms?integerWall time for the whole turn.
cache_read_tokens?integerInput tokens served from the prompt cache.
cache_write_tokens?integerInput tokens written into the prompt cache.
synthesis?object | nullThe answer completion's cost. Null on a turn that skipped synthesis, as retrieval-only turns do.
Show child attributes
type?stringEvent form only.
query_id?stringEvent form only.
status?stringHow synthesis ended, e.g. completed.
tokens_in?integerThe completion's input tokens.
tokens_out?integerThe completion's output tokens.
ms?integerWall time of the completion.
answer_preview?stringLeading characters of the answer, for a progress display that has no full text yet.
tokens_in_cached?integerInput tokens served from the prompt cache.
tokens_in_cache_write?integerInput tokens written into the prompt cache.
tokens_in_fresh?integerInput tokens neither cached nor cache-written.
trace_ref?string | nullBlob key of the persisted trace. Fetch the trace itself from GET /queries/id/trace.
usageobjectrequiredThe turn's token totals. Embed and rerank bill on their own seam and are not counted here.
Show child attributes
input_tokensintegerrequiredInput tokens the turn billed.
output_tokensintegerrequiredOutput tokens the turn billed.
total_tokensintegerrequiredInput plus output tokens.
runtime_msintegerrequiredWall time from turn start to terminal state.