Shows above the prompt how long the prompt cache has left, the session's cache hit rate and misses, and what a cold next message would re-cache. Adds /cache.

The complete skill library that Cure Consulting Group uses to build apps, platforms, and products. These skills encode our standards, frameworks, and processes — so every project ships with the same level of rigor.
Now available as a Claude Code Plugin — install once, get auto-updates across all projects.
ProductEngineeringSkills/
├── .claude-plugin/ # Plugin manifest
│ └── plugin.json
├── skills/{domain}/ # 80 skills, organized by domain (engineering/platform/product/business/marketing/security/legal)
│ ├── sdlc/
│ ├── android-feature-scaffold/
│ ├── incident-response/ # NEW
│ ├── accessibility-audit/ # NEW
│ ├── performance-review/ # NEW
│ ├── database-architect/ # NEW
│ ├── infrastructure-scaffold/ # NEW
│ ├── project-bootstrap/
│ ├── e2e-testing/
│ ├── test-accounts/
│ ├── uat/
│ ├── compliance-architect/
│ ├── data-migration/
│ ├── feature-flags/
│ ├── release-management/
│ ├── observability/
│ ├── client-handoff/
│ ├── llmops/
│ ├── disaster-recovery/
│ ├── dora-metrics/
│ ├── design-system/
│ ├── client-communication/
│ ├── i18n/
│ ├── notification-architect/
│ ├── offline-first/
│ ├── chaos-engineering/
│ ├── edge-computing/
│ ├── finops/
│ ├── micro-frontends/
│ ├── growth-engineering/
│ ├── green-software/
│ ├── proposal-generator/
│ ├── api-gateway/
│ ├── ... (75 total — see docs/OVERVIEW.md for full inventory)
│ └── legal-doc-scaffold/
├── agents/ # 35 custom subagent definitions
├── personas/ # 4 cross-domain engagement archetypes
│ ├── code-reviewer.md # Security + quality review agent
│ ├── project-bootstrapper.md # New project setup agent
│ ├── test-runner.md # Execute test suites, report coverage
│ ├── pr-reviewer.md # Automated PR diff review
│ ├── refactor-assistant.md # Safe refactoring with test validation
│ ├── ci-debugger.md # Diagnose failed CI/CD runs
│ ├── release-coordinator.md # Version bump, changelog, deploy validation
│ ├── doc-generator.md # API docs, ADRs, changelogs from code
│ ├── codebase-explainer.md # Onboarding — explain architecture, trace flows
│ ├── migration-validator.md # Database migration safety checks
│ ├── deployment-validator.md # Pre-deployment checklist validation
│ ├── dependency-auditor.md # Vulnerability and outdated package audit
│ ├── api-validator.md # OpenAPI spec and contract validation
│ ├── product-analyst.md # Feature adoption, analytics instrumentation
│ ├── ux-researcher.md # Usability analysis, friction mapping
│ ├── roadmap-strategist.md # RICE scoring, dependency mapping, roadmaps
│ ├── competitive-intel.md # Feature matrices, positioning, moat analysis
│ ├── content-strategist.md # Editorial calendars, SEO, content briefs
│ ├── campaign-analyst.md # Attribution, funnel analysis, channel ROI
│ ├── brand-guardian.md # Voice/tone, visual identity, microcopy audit
│ ├── growth-analyst.md # Activation, retention, viral mechanics
│ ├── financial-analyst.md # Revenue forecasts, unit economics, scenarios
│ ├── market-intelligence.md # TAM/SAM/SOM, trends, market timing
│ ├── investor-relations.md # Board updates, KPIs, fundraising narratives
│ ├── contract-reviewer.md # SOW/contract risk, terms, IP review
│ ├── data-analyst.md # Schema exploration, queries, data quality
│ ├── metrics-dashboard.md # KPI definitions, SLOs, dashboard wireframes
│ ├── ab-test-analyst.md # Experiment design, statistical analysis
│ ├── qa-engineer.md # Test planning, edge cases, regression, quality gates
│ ├── accessibility-checker.md # WCAG 2.2 automated compliance
│ └── firebase-security-auditor.md # Firestore rules and Functions audit
├── hooks/ # Multi-layer automated enforcement
│ └── hooks.json # Command + Prompt hooks (9 event types)
├── rules/ # 11 path-specific coding standards
│ ├── android.md # Loads for *.kt files
│ ├── ios.md # Loads for *.swift files
│ ├── web.md # Loads for *.ts/*.tsx files
│ ├── firebase.md # Loads for functions/**
│ ├── python.md # Loads for *.py files
│ ├── go.md # Loads for *.go files
│ ├── rust.md # Loads for *.rs files
│ ├── sql.md # Loads for *.sql, migrations/**
│ ├── docker.md # Loads for Dockerfile, *.dockerfile
│ ├── terraform.md # Loads for *.tf, *.tfvars
│ └── cicd.md # Loads for .github/workflows/**
├── output-styles/ # 9 custom output formatting styles
│ ├── prd/ # Product docs (PRDs, GTM, research)
│ ├── code-generation/ # Code scaffolds and implementations
│ ├── financial-analysis/ # Cost models, SaaS metrics
│ ├── audit-report/ # Audits, reviews, compliance
│ ├── api-specification/ # OpenAPI specs, endpoint docs
│ ├── architecture-decision/ # ADRs, RFCs, trade-off matrices
│ ├── runbook/ # Incident runbooks, DR procedures
│ ├── test-plan/ # Test plans, coverage reports
│ └── monitoring-alert/ # Alert definitions, thresholds
├── .mcp.json # MCP server configs (GitHub, Sentry, Firestore, PostgreSQL)
├── .lsp.json # LSP server configs (TypeScript, Python/Pyright)
├── marketplace.json # Plugin marketplace manifest
├── settings.json # Default permission rules
├── claude-commands/ # Legacy format (backwards compat, 64 files)
├── gemini skills/ # Google Gemini skills (.skill ZIP)
├── CLAUDE.md # Project instructions (Claude)
├── GEMINI.md # Project instructions (Gemini CLI)
├── AGENT-GUIDE.md # How to structure prompts for agents & skills
├── setup.sh # Setup script for Antigravity & other projects
└── README.md
Install the plugin as an npm package from GitHub Packages. This is the easiest way to keep all your projects up to date.
1. Authenticate with GitHub Packages (one-time setup):
# Create a Personal Access Token (PAT) with read:packages scope at
# https://github.com/settings/tokens, then:
npm login --scope=@cure-consulting-group --registry=https://npm.pkg.github.com
Or add to your project's .npmrc:
@cure-consulting-group:registry=https://npm.pkg.github.com
//npm.pkg.github.com/:_authToken=${GITHUB_TOKEN}
2. Install in your project:
npm install @cure-consulting-group/product-engineering-skills
The postinstall script automatically:
~/.claude/plugins/ProductEngineeringSkills~/.claude/settings.jsonAll 80 skills, 39 agents, 4 personas, hooks, rules, and output styles are immediately available.
3. Enable auto-updates with Dependabot (recommended):
Add .github/dependabot.yml to your project (or run setup.sh which does this automatically):
version: 2
updates:
- package-ecosystem: "npm"
directory: "/"
schedule:
interval: "daily"
allow:
- dependency-name: "@cure-consulting-group/product-engineering-skills"
labels:
- "dependencies"
- "skills-update"
commit-message:
prefix: "chore"
include: "scope"
Dependabot will open a PR in your project whenever a new version is published. Merge it and every agent on that project gets the updated skills.
4. Manual update:
npm update @cure-consulting-group/product-engineering-skills
# Load the plugin during a session
claude --plugin-dir /path/to/ProductEngineeringSkills
# Or for development/testing
claude --plugin-dir ./ProductEngineeringSkills
Once loaded, all skills are available as namespaced commands:
/cure-product-engineering:sdlc
/cure-product-engineering:feature-audit
/cure-product-engineering:android-feature-scaffold
/cure-product-engineering:incident-response
/cure-product-engineering:accessibility-audit
Hooks, agents, rules, output styles, and MCP servers are all included automatically.
The fastest way to onboard any project:
# From the target project directory
/path/to/ProductEngineeringSkills/setup.sh
# Or specify the project path
/path/to/ProductEngineeringSkills/setup.sh /path/to/antigravity-app
# Install globally for ALL projects
/path/to/ProductEngineeringSkills/setup.sh --global
# Legacy mode (just copy skills, no hooks/agents)
/path/to/ProductEngineeringSkills/setup.sh --legacy
The setup script will:
~/.claude/plugins/.claude/settings.local.json to .gitignore# Add the Cure Consulting marketplace
claude marketplace add https://github.com/Cure-Consulting-Group/ProductEngineeringSkills/marketplace.json
# Install the plugin
claude plugin install cure-product-engineering
Copy the claude-commands/ files into your project's .claude/commands/ directory:
cp claude-commands/*.md /path/to/your/project/.claude/commands/
Then use them as slash commands:
/sdlc — Generate SDLC artifacts
/android-feature-scaffold — Scaffold an Android feature module
/feature-audit — Audit a completed feature
Import the .skill files from gemini skills/ into your Gemini workspace. Each .skill file is a ZIP archive containing:
SKILL.md — The main skill definitionreferences/ — Supporting documents and templates| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| product-manager | OKRs, roadmaps, RICE prioritization, feature briefs | Yes |
| product-design | Apple HIG, Material Design 3, design tokens, accessibility-first | Yes |
| market-research | TAM/SAM/SOM, competitive analysis, ICP definition (read-only) | Yes |
| go-to-market | GTM plans, launch strategy, channel selection, growth playbooks | Yes |
| product-marketing | Brand strategy, messaging frameworks, campaigns | Yes |
| customer-onboarding | Activation flows, empty states, email sequences, retention | Yes |
| seo-content-engine | Technical SEO, structured data, content strategy | Yes |
| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| sdlc | PRDs, ADRs, RFCs, Epics, Stories, Task specs — full SDLC | Yes |
| android-feature-scaffold | Clean Architecture Android scaffolding (MVI, Compose, Hilt) | Yes |
| ios-architect | Swift/SwiftUI Clean Architecture, MVVM, structured concurrency | Yes |
| nextjs-feature-scaffold | App Router, Server/Client components, Tailwind patterns | Yes |
| firebase-architect | Firestore schema, security rules, Cloud Functions | Yes |
| api-architect | REST/GraphQL design, versioning, auth, rate limiting | Yes |
| api-gateway | API gateway and BFF layers, rate limiting, GraphQL federation | Yes |
| stripe-integration | Stripe payments + subscriptions via Firebase Functions | Yes |
| ai-feature-builder | LLM integration, RAG pipelines, prompt engineering | Yes |
| llmops | LLM operationalization — prompt versioning, eval pipelines, cost optimization, guardrails | Yes |
| database-architect | Schema design, migrations, indexing for Firestore/PostgreSQL/SQLite | Yes |
| data-migration | ETL pipelines, zero-downtime cutover, validation, rollback strategies | Yes |
| infrastructure-scaffold | Cloud infra configs for Firebase, GCP, Vercel, Docker | Yes |
| edge-computing | Edge functions, CDN strategies, cache invalidation, edge middleware | Yes |
| micro-frontends | Module federation, monorepo management, independent deployments | Yes |
| offline-first | Offline-first architecture, sync strategies, conflict resolution, optimistic UI | Yes |
| i18n | Internationalization — string extraction, RTL, locale-aware formatting, translation workflows | Yes |
| notification-architect | Push (FCM/APNs), in-app messaging, email, preference management | Yes |
| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| feature-audit | 5-phase post-completion audit with scored gap report | Yes (read-only, forked) |
| testing-strategy | Testing pyramid, platform standards, coverage rules | Yes |
| e2e-testing | E2E test suites with page objects, visual regression, CI integration | Yes |
| test-accounts | Test user personas, seed data scripts, environment credentials | Yes |
| uat | UAT plans, acceptance criteria checklists, go/no-go release gates | Yes |
| security-review | OWASP checklist, auth/data/API/mobile/web security | Yes (read-only, forked) |
| compliance-architect | HIPAA, COPPA, GDPR, PCI compliance frameworks, consent flows, audit trails | Yes |
| accessibility-audit | WCAG 2.2 compliance, screen readers, inclusive design | Yes (read-only, forked) |
| performance-review | Performance budgets, load testing, optimization strategies | Yes |
| chaos-engineering | Resilience testing, failure injection, graceful degradation, game days | Yes |
| green-software | Sustainable software practices, carbon-aware computing, energy efficiency | Yes |
| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| project-bootstrap | Bootstrap repo with CLAUDE.md + STATE.md via codebase inspection and developer interview | Yes |
| project-manager | Sprint planning, RACI, risk registers, retrospectives | Yes |
| ci-cd-pipeline | GitHub Actions, build/test/deploy, environments, secrets | Yes |
| release-management | App store submissions, staged rollouts, versioning, ASO, changelogs | Yes |
| feature-flags | Progressive rollouts, A/B testing, kill switches, experimentation frameworks | Yes |
| observability | Structured logging, distributed tracing, alerting, SLO/SLI, dashboards | Yes |
| dora-metrics | DORA and SPACE metrics — deployment frequency, lead time, MTTR, developer experience | Yes |
| analytics-implementation | Event taxonomy, tracking plans, funnels, dashboards | Yes |
| incident-response | Runbooks, severity classification, post-mortems, escalation | Yes |
| disaster-recovery | DR and business continuity — RTO/RPO, backup strategies, failover, DR testing | Yes |
| growth-engineering | Activation funnels, referral programs, lifecycle automation, PLG patterns | Yes |
| design-system | Design tokens, component libraries, Storybook/Catalog, cross-platform consistency | Yes |
| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| engineering-cost-model | Project estimates, infrastructure costs, build vs buy | Yes (read-only) |
| saas-financial-model | Unit economics, MRR/ARR, pricing tiers, break-even | Yes (read-only) |
| finops | Cloud cost optimization, budget alerts, resource right-sizing, FinOps practices | Yes |
| investor-reporting | Investor updates, board decks, portfolio financials, cap table, runway modeling | Yes |
| fundraising-materials | Pitch decks, data rooms, investor updates, cap table scenarios, fundraising pipeline | Yes |
| burn-rate-tracker | Burn rates, runway scenarios, break-even analysis, cash flow projections | Yes |
| legal-doc-scaffold | ToS, Privacy Policy, SOW, NDA scaffolds | No (manual only) |
| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| portfolio-registry | Product portfolio registry — single source of truth for all products, stacks, teams, stages | Yes |
| technology-radar | ThoughtWorks-style technology radar — Adopt/Trial/Assess/Hold across the portfolio | Yes |
| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| client-handoff | Handoff packages, runbooks, credential transfers, maintenance SLAs, knowledge transfer | Yes |
| client-communication | Sprint demo scripts, stakeholder updates, risk escalation, executive summaries | Yes |
| proposal-generator | Consulting proposals, SOWs, milestone pricing, engagement structure | No (manual only) |
| Skill | What It Does | Auto-Invoked? |
|---|---|---|
| android-design-expert | Material Design 3 — dynamic color, component tokens, adaptive layouts, motion, Compose patterns | Yes |
| ios-design-expert | Apple HIG — SF Symbols, Dynamic Type, navigation patterns, SwiftUI components | Yes |
| web-design-expert | Responsive design, CSS architecture, design tokens, container queries, accessibility-first, Tailwind | Yes |
| stitch-design | AI-native UI design via Stitch MCP — vibe design, mockups, screen generation, design tokens, component export | Yes |
The library ships a standard maintenance loop (self-provisioned to .claude/loop.md on first session start — run bare /loop to use it), Recurring Mode sections in the goal-shaped skills (finops, burn-rate-tracker, investor-reporting, security-review, and others), and copy-paste cloud-routine recipes in docs/AUTOMATION.md. Library upkeep cadence: docs/MAINTENANCE.md. Mechanism selection and unattended-run guardrails: /cure-product-engineering:engagement-automation.
The plugin ships command and prompt hooks across 9 event types: SessionStart, PreCompact, PostCompact, ConfigChange, PostToolUseFailure, UserPromptSubmit, PreToolUse, Stop, SubagentStop. Highlights: a Stop-hook quality gate (blocks "done" without verification), a PreToolUse static security guard on skill/agent/persona files, and a ConfigChange audit trigger when skill files change mid-session.
New in v4.0: Hooks now suggest and auto-trigger agents based on context. Every code edit, test run, deployment, and PR action recommends the most relevant agent(s).
| Hook | Event | What It Does | Agent Integration |
|---|---|---|---|
| Welcome | SessionStart | Confirms plugin loaded with counts; full inventory stays in docs/OVERVIEW.md | Points to inventory |
| Git status | SessionStart | Reports current branch, uncommitted changes, last commit | — |
| Dependency check | SessionStart | Detects outdated packages | Suggests dependency-auditor |
| Code edit advisor | PostToolUse (Edit/Write) | Context-aware suggestions based on file type (.kt, .swift, .ts, .sql, .tf, etc.) | Suggests code-reviewer, test-runner, brand-guardian, migration-validator |
| Command advisor | PostToolUse (Bash) | Post-action guidance for tests, installs, deploys, PRs, releases | Suggests ci-debugger, dependency-auditor, pr-reviewer, release-coordinator |
| Failure recovery | PostToolUseFailure | Diagnoses failure type and suggests fix approach | Auto-suggests ci-debugger, deployment-validator, dependency-auditor |
| Destructive prompt guard | UserPromptSubmit | Detects destructive operations in prompts | Blocks and confirms |
| Protected files | PreToolUse (Edit/Write) | Blocks edits to .env, lock files, credentials, tfstate | — |
| Dangerous commands | PreToolUse (Bash) | Blocks force push, destructive rm, DROP TABLE, prod deploys | — |
| Context re-injection | PreCompact | Re-injects all 80 skills, 39 agents, 4 personas, and Cure standards | Full inventory preserved |
| Post-compact restore | PostCompact | Confirms context restored with agent availability | — |
| Subagent start banner | SubagentStart | Announces agent with role, standards, and companion agents | Lists companion agents |
| Subagent completion | SubagentStop | Suggests follow-up agents (test-runner, code-reviewer, pr-reviewer) | Agent chaining |
| Task quality check | TaskCompleted | Validates tests, security, docs, brand consistency | Suggests test-runner, code-reviewer, doc-generator, brand-guardian |
| Hook | Event | What It Does | Agent Integration |
|---|---|---|---|
| Code quality gate | PreToolUse (Edit/Write) | Haiku validates: no secrets, no debug logs, no disabled tests, no any types | — |
| Deployment safety | PreToolUse (Bash) | Haiku validates: blocks production deployments outside CI/CD | — |
| Intent classifier | UserPromptSubmit | Haiku classifies prompt intent and suggests the most relevant agent(s) from all 30 | Maps prompts → agents with confidence scores |
| Hook | Event | What It Does | Agent Integration |
|---|---|---|---|
| Completion validator | Stop | Validates: tests for new code, security review for sensitive changes, rollback for migrations, docs for features, brand consistency for UI, analytics for events, API contracts | Suggests specific agents for each gap found |
Pre-configured MCP servers in .mcp.json:
| Server | Type | What It Does |
|---|---|---|
| GitHub | HTTP | PR management, issue tracking, code search |
| Sentry | HTTP | Error monitoring, issue tracking, release health |
| Firestore | stdio | Direct database queries, schema inspection |
| PostgreSQL | stdio | Database queries, schema inspection, migrations |
Pre-configured LSP servers in .lsp.json:
| Server | Language | What It Provides |
|---|---|---|
| TypeScript | .ts, .tsx, .js | Type checking, auto-imports, refactoring, go-to-definition |
| Python (Pyright) | *.py | Static type analysis, import resolution, error diagnostics |
Custom output formatting for different artifact types:
| Style | Used By | Key Rules |
|---|---|---|
| prd | Product skills (PRDs, GTM, research) | Numbered sections, decision matrices, executive summaries |
| code-generation | Engineering skills (scaffolds) | File tree first, dependency order, complete runnable code |
| financial-analysis | Business skills (costs, models) | ASCII tables, explicit assumptions, sensitivity analysis |
| audit-report | Quality skills (audits, reviews) | Severity scoring, checklists, remediation with effort estimates |
| api-specification | API design skills | OpenAPI 3.0 blocks, endpoint tables, request/response examples |
| architecture-decision | ADR and RFC skills | Context/decision/consequences format, trade-off matrices |
| runbook | Incident response, disaster recovery | Numbered steps, command blocks, decision trees, escalation paths |
| test-plan | Testing strategy, QA skills | Coverage tables, test case templates, pass/fail criteria |
| monitoring-alert | Observability, incident response | Alert definition tables, threshold rationale, runbook links |
| Agent | Purpose | Tools | Auto-Triggered By |
|---|---|---|---|
| code-reviewer | Security + quality review against Cure standards | Read-only | Stop hook, SubagentStop |
| project-bootstrapper | Set up new projects with correct architecture |
hooks/register.tsx 104 lines1import { atom, read, update } from 'claude-code'
2import type { EngineInterface, Register } from 'claude-code'
3
4import { EMPTY, bandText, bar, inferTtl, leftText, recordStep, reportText, tokensText, view, writeFeeText } from './cache'
5
6/**
7 * cure-cache-band: a row above the prompt showing how long the prompt cache
8 * has left, the session's hit rate and misses, and, once it has lapsed, how
9 * many tokens the next message will pay to re-cache. /cache gives the detail.
10 *
11 * It counts the main conversation's requests only: a subagent keeps a cache
12 * of its own. Nothing is fetched; every figure comes from the usage each
13 * model response already reports.
14 *
15 * Setting, as an environment variable:
16 * CURE_CACHE_TTL `5m` or `1h`, when the inferred TTL is wrong for your plan
17 *
18 * The TTL is inferred, not reported by the API (see inferTtl): the countdown
19 * is an estimate, the hit rate and misses are measured.
20 */
21
22const TICK_MS = 1000
23
24const cache = atom({ plugin: 'cure-cache-band', key: 'cache' } as const, EMPTY)
25
26let timer: { cancel: () => void } | undefined
27let drawn = ''
28
29export const register: Register = on => {
30 on('session.start', async ($, e, next) => {
31 await $.command.register({ name: 'cache', description: 'Show prompt cache time left, hit rate and misses for this session' })
32 timer?.cancel()
33 // Redraw only when the row's text would change: once a second in the last minute, once a minute before it.
34 timer = $.clock.every(TICK_MS, async () => {
35 const text = bandText(view(await read($, cache), await $.clock.now()))
36 if (text !== drawn) $.ui.invalidate('ui.render')
37 })
38 return next(e)
39 })
40
41 on('session.end', async ($, e, next) => {
42 timer?.cancel()
43 timer = undefined
44 return next(e)
45 })
46
47 on('turn.step', async function* ($, e, next) {
48 const result = yield* next(e)
49 const usage = result.usage
50 if (e.agentId === undefined && usage) {
51 const [now, ttl] = [await $.clock.now(), (await ttlNow($)).ttl]
52 await update($, cache, state => recordStep(state, usage, now, ttl))
53 }
54 return result
55 })
56
57 on('command.run', { command: 'cache' }, async $ => {
58 const { source } = await ttlNow($)
59 return { text: reportText(await read($, cache), await $.clock.now(), source) }
60 })
61
62 on('ui.render', { component: 'AbovePrompt' }, async ($, e, next) => {
63 const v = view(await read($, cache), await $.clock.now())
64 drawn = bandText(v)
65 if (v.kind === 'none' || e.props.hasSurvey) return next(e)
66
67 const { Box, Text } = $.ui.resolve(e)
68
69 if (v.kind === 'cold') {
70 const fee = writeFeeText(v.tokens)
71 return (
72 <Box>
73 <Text color="red">cache ○ cold</Text>
74 <Text dimColor> · next message re-caches {tokensText(v.tokens)} tokens{fee ? ` (${fee})` : ''}</Text>
75 </Box>
76 )
77 }
78
79 const color = v.level === 'low' ? 'yellow' : 'green'
80 const b = bar(v.fraction)
81 return (
82 <Box>
83 <Text color={color}>
84 cache ● {v.ttl} {b.filled}
85 </Text>
86 <Text dimColor>{b.empty}</Text>
87 <Text color={color}> {leftText(v.leftMs)}</Text>
88 <Text dimColor>
89 {' '}
90 · hit {v.hitPercent}% · misses {v.misses}
91 </Text>
92 </Box>
93 )
94 })
95}
96
97async function ttlNow($: EngineInterface) {
98 const override = (await $.env.get('CURE_CACHE_TTL')) || undefined
99 const { rateLimits } = await $.session.usage()
100 const ttl = inferTtl(override, rateLimits)
101 const source = override === ttl ? 'CURE_CACHE_TTL' : rateLimits.length === 0 ? 'inferred: no subscription limits reported' : 'inferred from subscription limits'
102 return { ttl, source }
103}
104hooks/cache.ts 144 lines1import type { CacheState, CacheTtl } from '../types'
2
3/** What one model response reported, as `turn.step`'s `usage` carries it. */
4export type StepUsage = {
5 input_tokens: number
6 output_tokens: number
7 cache_read_input_tokens: number
8 cache_creation_input_tokens: number
9}
10
11export type RateLimit = { kind: string; percentUsed: number }
12
13export type View =
14 | { kind: 'none' }
15 | { kind: 'warm'; level: 'ok' | 'low'; ttl: CacheTtl; leftMs: number; fraction: number; hitPercent: number; misses: number }
16 | { kind: 'cold'; tokens: number }
17
18export const TTL_MS: Record<CacheTtl, number> = { '5m': 5 * 60 * 1000, '1h': 60 * 60 * 1000 }
19
20/** Under this share of the TTL left, the band turns amber. */
21export const LOW_FRACTION = 0.25
22/** A prefix shorter than this is below the API's minimum cacheable length; reading none of it is not a miss. */
23const MIN_CACHEABLE = 1024
24export const BAR_CELLS = 16
25
26export const EMPTY: CacheState = {
27 lastAt: null,
28 ttl: '1h',
29 prefixTokens: 0,
30 nextTokens: 0,
31 read: 0,
32 written: 0,
33 uncached: 0,
34 requests: 0,
35 misses: 0,
36}
37
38/**
39 * The TTL the session's cache entries are assumed to have. The API does not
40 * report it per response, so this is inferred: CURE_CACHE_TTL when set;
41 * otherwise 1h on a subscription inside its limits, 5m on an API key (no
42 * rate-limit windows) or once a window is exhausted.
43 */
44export function inferTtl(override: string | undefined, rateLimits: readonly RateLimit[]): CacheTtl {
45 if (override === '5m' || override === '1h') return override
46 if (rateLimits.length === 0) return '5m'
47 return rateLimits.some(l => l.percentUsed >= 100) ? '5m' : '1h'
48}
49
50/**
51 * Folds one main-thread response into the session's figures. A miss is a
52 * request, after the first, that read less than half of the prefix the
53 * request before it left in the cache: an expiry, a model switch, a
54 * compaction or an edited prefix all land here, and all cost a re-write.
55 */
56export function recordStep(state: CacheState, usage: StepUsage, now: number, ttl: CacheTtl): CacheState {
57 const prefix = usage.input_tokens + usage.cache_read_input_tokens + usage.cache_creation_input_tokens
58 // The first request has no prefix before it, so it is never a miss.
59 const isMiss = state.prefixTokens >= MIN_CACHEABLE && usage.cache_read_input_tokens < state.prefixTokens / 2
60 return {
61 lastAt: now,
62 ttl,
63 prefixTokens: prefix,
64 nextTokens: prefix + usage.output_tokens,
65 read: state.read + usage.cache_read_input_tokens,
66 written: state.written + usage.cache_creation_input_tokens,
67 uncached: state.uncached + usage.input_tokens,
68 requests: state.requests + 1,
69 misses: state.misses + (isMiss ? 1 : 0),
70 }
71}
72
73/** Share of all prompt tokens this session that the cache served, as a whole percentage. */
74export function hitPercent(state: CacheState): number {
75 const total = state.read + state.written + state.uncached
76 return total === 0 ? 0 : Math.round((state.read / total) * 100)
77}
78
79export function view(state: CacheState, now: number): View {
80 if (state.lastAt === null) return { kind: 'none' }
81 const total = TTL_MS[state.ttl]
82 const leftMs = state.lastAt + total - now
83 if (leftMs <= 0) return { kind: 'cold', tokens: state.nextTokens }
84 const fraction = Math.min(1, leftMs / total)
85 return {
86 kind: 'warm',
87 level: fraction < LOW_FRACTION ? 'low' : 'ok',
88 ttl: state.ttl,
89 leftMs,
90 fraction,
91 hitPercent: hitPercent(state),
92 misses: state.misses,
93 }
94}
95
96/** `59m left` above a minute, `53s left` under it; never rounds up to the full TTL's next unit. */
97export function leftText(leftMs: number): string {
98 const seconds = Math.max(1, Math.ceil(leftMs / 1000))
99 return seconds > 60 ? `${Math.floor(seconds / 60)}m left` : `${seconds}s left`
100}
101
102export function tokensText(tokens: number): string {
103 if (tokens >= 1_000_000) return `${(tokens / 1_000_000).toFixed(1)}M`
104 return tokens >= 1000 ? `${Math.round(tokens / 1000)}k` : String(tokens)
105}
106
107/** Estimated write fee for re-caching tokens at Claude 3.5 Sonnet's write rate ($3.75 / M tokens). */
108export function writeFeeText(tokens: number): string {
109 if (tokens <= 0) return ''
110 const fee = (tokens / 1_000_000) * 3.75
111 return fee < 0.01 ? '<$0.01 fee' : `~$${fee.toFixed(2)} fee`
112}
113
114/** The bar's two runs: at least one filled cell while any time is left. */
115export function bar(fraction: number, cells: number = BAR_CELLS): { filled: string; empty: string } {
116 const n = Math.min(cells, Math.max(1, Math.round(fraction * cells)))
117 return { filled: '█'.repeat(n), empty: '░'.repeat(cells - n) }
118}
119
120/** The band as one plain string: what the timer compares to know a redraw is due, and what /cache prints first. */
121export function bandText(v: View): string {
122 if (v.kind === 'none') return ''
123 if (v.kind === 'cold') {
124 const fee = writeFeeText(v.tokens)
125 return `cache ○ cold · next message re-caches ${tokensText(v.tokens)} tokens${fee ? ` (${fee})` : ''}`
126 }
127 const b = bar(v.fraction)
128 return `cache ● ${v.ttl} ${b.filled}${b.empty} ${leftText(v.leftMs)} · hit ${v.hitPercent}% · misses ${v.misses}`
129}
130
131export function reportText(state: CacheState, now: number, ttlSource: string): string {
132 const v = view(state, now)
133 if (v.kind === 'none') return 'cure-cache-band: no model response yet this session, so nothing is cached that the band has seen.'
134 return [
135 bandText(v),
136 ` TTL assumed ${state.ttl} (${ttlSource})`,
137 ` Requests (main) ${state.requests}, ${state.misses} missed`,
138 ` Read from cache ${tokensText(state.read)} tokens`,
139 ` Written to cache ${tokensText(state.written)} tokens`,
140 ` Uncached input ${tokensText(state.uncached)} tokens`,
141 ` Next message re-sends ${tokensText(state.nextTokens)} tokens${v.kind === 'cold' ? `, all at the cache-write rate (${writeFeeText(state.nextTokens)})` : ''}`,
142 ].join('\n')
143}
144types/index.d.ts 25 lines1export type CacheTtl = '5m' | '1h'
2
3export type CacheState = {
4 /** When the last main-thread response arrived, in `$.clock.now()`'s milliseconds; null before the first. */
5 lastAt: number | null
6 /** The TTL assumed for the entry that response wrote or refreshed. */
7 ttl: CacheTtl
8 /** Prompt tokens the last request was answered over: what a warm cache holds. */
9 prefixTokens: number
10 /** Prompt tokens the next request re-sends: the prefix plus the last response's output. */
11 nextTokens: number
12 /** Session totals over main-thread requests. */
13 read: number
14 written: number
15 uncached: number
16 requests: number
17 misses: number
18}
19
20declare module 'claude-code' {
21 interface PluginState {
22 'cure-cache-band': { cache: CacheState }
23 }
24}
25