Skip to main content
Pinecone Docs
current

Search documentation

Type to search this documentation.

Trigger an on-demand self-tuning optimize

Tunes the manifest from real query traffic, then chains a forced re-curate. Every field of the required body is optional, so {} is valid — and a no-op without candidate_queries. The context must be curated, and only one optimize runs at a time.

POST /contexts/{slug}/optimize

cURL
curl --request POST \
  --url https://{host}/api/contexts/{slug}/optimize \
  --header 'Authorization: Bearer <token>' \
  --header 'X-Pinecone-Api-Version: <x-pinecone-api-version>' \
  --header 'Content-Type: application/json' \
  --data '{
  "eval_pass_rate_threshold": 123,
  "retrieval_p90_latency_ms": 123,
  "optimize_timeout_seconds": 123,
  "max_iterations": 123,
  "max_tool_turns": 123,
  "models": [
    "<string>"
  ],
  "candidate_queries": [
    {
      "ask": "<string>",
      "answer": "<string>",
      "fallback_to_chunks": true,
      "latency_ms": 123
    }
  ],
  "pinecone_api_key": "<string>"
}'
Python
import requests

url = "https://{host}/api/contexts/{slug}/optimize"

payload = {
  "eval_pass_rate_threshold": 123,
  "retrieval_p90_latency_ms": 123,
  "optimize_timeout_seconds": 123,
  "max_iterations": 123,
  "max_tool_turns": 123,
  "models": [
    "<string>"
  ],
  "candidate_queries": [
    {
      "ask": "<string>",
      "answer": "<string>",
      "fallback_to_chunks": True,
      "latency_ms": 123
    }
  ],
  "pinecone_api_key": "<string>"
}
headers = {
    "Authorization": "Bearer <token>",
    "X-Pinecone-Api-Version": "<x-pinecone-api-version>",
    "Content-Type": "application/json"
}

response = requests.post(url, json=payload, headers=headers)

print(response.text)
JavaScript
const options = {method: "POST", headers: {"Authorization": "Bearer <token>", "X-Pinecone-Api-Version": "<x-pinecone-api-version>", "Content-Type": "application/json"}, body: JSON.stringify({
  "eval_pass_rate_threshold": 123,
  "retrieval_p90_latency_ms": 123,
  "optimize_timeout_seconds": 123,
  "max_iterations": 123,
  "max_tool_turns": 123,
  "models": [
    "<string>"
  ],
  "candidate_queries": [
    {
      "ask": "<string>",
      "answer": "<string>",
      "fallback_to_chunks": true,
      "latency_ms": 123
    }
  ],
  "pinecone_api_key": "<string>"
})};

fetch("https://{host}/api/contexts/{slug}/optimize", options)
  .then(res => res.json())
  .then(res => console.log(res))
  .catch(err => console.error(err));
PHP
<?php

$curl = curl_init();

curl_setopt_array($curl, [
  CURLOPT_URL => "https://{host}/api/contexts/{slug}/optimize",
  CURLOPT_RETURNTRANSFER => true,
  CURLOPT_CUSTOMREQUEST => "POST",
  CURLOPT_POSTFIELDS => "{\"eval_pass_rate_threshold\":123,\"retrieval_p90_latency_ms\":123,\"optimize_timeout_seconds\":123,\"max_iterations\":123,\"max_tool_turns\":123,\"models\":[\"<string>\"],\"candidate_queries\":[{\"ask\":\"<string>\",\"answer\":\"<string>\",\"fallback_to_chunks\":true,\"latency_ms\":123}],\"pinecone_api_key\":\"<string>\"}",
  CURLOPT_HTTPHEADER => [
    "Authorization: Bearer <token>",
    "X-Pinecone-Api-Version: <x-pinecone-api-version>",
    "Content-Type: application/json"
  ],
]);

$response = curl_exec($curl);
$err = curl_error($curl);

curl_close($curl);

if ($err) {
  echo "cURL Error #:" . $err;
} else {
  echo $response;
}
Go
package main

import (
	"fmt"
	"strings"
	"net/http"
	"io"
)

func main() {

	url := "https://{host}/api/contexts/{slug}/optimize"

	payload := strings.NewReader("{\"eval_pass_rate_threshold\":123,\"retrieval_p90_latency_ms\":123,\"optimize_timeout_seconds\":123,\"max_iterations\":123,\"max_tool_turns\":123,\"models\":[\"<string>\"],\"candidate_queries\":[{\"ask\":\"<string>\",\"answer\":\"<string>\",\"fallback_to_chunks\":true,\"latency_ms\":123}],\"pinecone_api_key\":\"<string>\"}")

	req, _ := http.NewRequest("POST", url, payload)

	req.Header.Add("Authorization", "Bearer <token>")
	req.Header.Add("X-Pinecone-Api-Version", "<x-pinecone-api-version>")
	req.Header.Add("Content-Type", "application/json")

	res, _ := http.DefaultClient.Do(req)

	defer res.Body.Close()
	body, _ := io.ReadAll(res.Body)

	fmt.Println(string(body))

}
Java
HttpResponse<String> response = Unirest.post("https://{host}/api/contexts/{slug}/optimize")
  .header("Authorization", "Bearer <token>")
  .header("X-Pinecone-Api-Version", "<x-pinecone-api-version>")
  .header("Content-Type", "application/json")
  .body("{\"eval_pass_rate_threshold\":123,\"retrieval_p90_latency_ms\":123,\"optimize_timeout_seconds\":123,\"max_iterations\":123,\"max_tool_turns\":123,\"models\":[\"<string>\"],\"candidate_queries\":[{\"ask\":\"<string>\",\"answer\":\"<string>\",\"fallback_to_chunks\":true,\"latency_ms\":123}],\"pinecone_api_key\":\"<string>\"}")
  .asString();
Ruby
require 'uri'
require 'net/http'

url = URI("https://{host}/api/contexts/{slug}/optimize")

http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true

request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["X-Pinecone-Api-Version"] = '<x-pinecone-api-version>'
request["Content-Type"] = 'application/json'
request.body = "{\"eval_pass_rate_threshold\":123,\"retrieval_p90_latency_ms\":123,\"optimize_timeout_seconds\":123,\"max_iterations\":123,\"max_tool_turns\":123,\"models\":[\"<string>\"],\"candidate_queries\":[{\"ask\":\"<string>\",\"answer\":\"<string>\",\"fallback_to_chunks\":true,\"latency_ms\":123}],\"pinecone_api_key\":\"<string>\"}"

response = http.request(request)
puts response.read_body
200
{
  "task_id": "<string>",
  "state": "<string>",
  "exploring": true,
  "profiling": true,
  "importing": true
}
404
{
  "message": "<string>",
  "code": "<string>"
}
Authorizationstringrequired

Session token from POST /auth/login, sent as Authorization: Bearer <token>.

X-Pinecone-Api-Version?string

Date-based contract version, echoed back on the same header. Omit for the default (2026-07); send unstable for the in-development surface. An unrecognized value is rejected with 400 unsupported_api_version.

Typestring
Default2026-07
slugstringrequired

Context slug or UUID.

Typestring
eval_pass_rate_threshold?number

Pass rate a candidate manifest must clear to count as ready.

Typenumber
retrieval_p90_latency_ms?integer

The p90 latency bar a candidate must stay under.

Typeinteger
optimize_timeout_seconds?integer

Default 7200

Typeinteger
max_iterations?integer

Default 20

Typeinteger
max_tool_turns?integer

Default 200

Typeinteger
models?string[]

Model-id overrides passed to the optimize runtime

Typestring[]
candidate_queries?object[]

Recent query records to tune toward (absent = no-op)

Typeobject[]
Show child attributes
ask?string

The question to tune retrieval against.

Typestring
answer?string

Used as eval ground truth

Typestring
fallback_to_chunks?boolean

Whether answering this query had to drill from artifacts down to chunks.

Typeboolean
latency_ms?number

Latency the query showed in real traffic, as a tuning baseline.

Typenumber
pinecone_api_key?string

Task-token override; normally not needed

Typestring

200 — Optimize enqueued (state: optimizing)

What every context-workflow trigger returns. The work is asynchronous: poll GET /tasks/{task_id}. The per-workflow booleans duplicate state and appear only where a workflow's contract names one.

task_idstringrequired

The enqueued task. Poll it at GET /tasks/id.

Typestring
statestringrequired

The forward-looking status the trigger put the context into.

Typestring
exploring?boolean

Duplicates state; present only for explore

Typeboolean
profiling?boolean

Duplicates state; present only for profile

Typeboolean
importing?boolean

Duplicates state; present only for the import/curate triggers

Typeboolean
Suggest an edit

Propose a replacement for this page. The site team reviews it before applying any changes.

Export
Documentation menu