Docs / Technical guides
Cache-first
Enable cache via a governance policy (kind=cache), then repeat prompts with cache_similarity_threshold on ask().
from tokensaver_sdk import TokenSaver
ts = TokenSaver(api_key="ts_...")
ts.create_governance_policy(
"Repeat prompts",
kind="cache",
config={"exact_cache": True, "semantic_cache": True, "similarity_threshold": 0.85},
)
prompt = "Summarize the Q1 support incidents in exactly 4 bullets."
first = ts.ask(
prompt,
provider="openai",
model="gpt-4o",
cache_similarity_threshold=0.85,
)
second = ts.ask(
prompt,
provider="openai",
model="gpt-4o",
cache_similarity_threshold=0.85,
)
print("first cache_hit:", first.metrics.cache_hit)
print("second cache_hit:", second.metrics.cache_hit)
print("second cost:", second.metrics.cost_usd)