Docs / Technical guides

Cache-first

Enable cache via a governance policy (kind=cache), then repeat prompts with cache_similarity_threshold on ask().

from tokensaver_sdk import TokenSaver

ts = TokenSaver(api_key="ts_...")
ts.create_governance_policy(
    "Repeat prompts",
    kind="cache",
    config={"exact_cache": True, "semantic_cache": True, "similarity_threshold": 0.85},
)

prompt = "Summarize the Q1 support incidents in exactly 4 bullets."

first = ts.ask(
    prompt,
    provider="openai",
    model="gpt-4o",
    cache_similarity_threshold=0.85,
)
second = ts.ask(
    prompt,
    provider="openai",
    model="gpt-4o",
    cache_similarity_threshold=0.85,
)

print("first cache_hit:", first.metrics.cache_hit)
print("second cache_hit:", second.metrics.cache_hit)
print("second cost:", second.metrics.cost_usd)