SLOPSHOPPER

cure-cache-band

Shows above the prompt how long the prompt cache has left, the session's cache hit rate and misses, and what a cold next message would re-cache. Adds /cache.

newbandcommandtimer
A shopper browsing a rack in a slop shop
Preview · a replayed session in a sandbox
claude · ~/work/app · cure-cache-band
› fix the failing auth test and add an audit log call ⏺ Read(src/auth.ts) ⎿ Read 6 lines ⏺ Update(src/auth.ts) ⎿ Added 2 lines, removed 1 line ⏺ Bash(bun test) ⎿ 3 pass, 1 fail ● Done. refresh now rejects expired claims and logs an audit event. ✻ Worked for 42s · done 4:20 PM › /cache ⎿ cure-cache-band: cure-cache-band: no model response yet this session, so nothing is cached that the band has seen. ────────────────────────────────────────────────────────────────────────────────────────────────────────────────────── › ? for shortcuts
README

Product Engineering Skills

The complete skill library that Cure Consulting Group uses to build apps, platforms, and products. These skills encode our standards, frameworks, and processes — so every project ships with the same level of rigor.

Now available as a Claude Code Plugin — install once, get auto-updates across all projects.

How It's Organized

ProductEngineeringSkills/
├── .claude-plugin/           # Plugin manifest
│   └── plugin.json
├── skills/{domain}/          # 80 skills, organized by domain (engineering/platform/product/business/marketing/security/legal)
│   ├── sdlc/
│   ├── android-feature-scaffold/
│   ├── incident-response/     # NEW
│   ├── accessibility-audit/   # NEW
│   ├── performance-review/    # NEW
│   ├── database-architect/    # NEW
│   ├── infrastructure-scaffold/ # NEW
│   ├── project-bootstrap/
│   ├── e2e-testing/
│   ├── test-accounts/
│   ├── uat/
│   ├── compliance-architect/
│   ├── data-migration/
│   ├── feature-flags/
│   ├── release-management/
│   ├── observability/
│   ├── client-handoff/
│   ├── llmops/
│   ├── disaster-recovery/
│   ├── dora-metrics/
│   ├── design-system/
│   ├── client-communication/
│   ├── i18n/
│   ├── notification-architect/
│   ├── offline-first/
│   ├── chaos-engineering/
│   ├── edge-computing/
│   ├── finops/
│   ├── micro-frontends/
│   ├── growth-engineering/
│   ├── green-software/
│   ├── proposal-generator/
│   ├── api-gateway/
│   ├── ... (75 total — see docs/OVERVIEW.md for full inventory)
│   └── legal-doc-scaffold/
├── agents/                   # 35 custom subagent definitions
├── personas/                 # 4 cross-domain engagement archetypes
│   ├── code-reviewer.md      # Security + quality review agent
│   ├── project-bootstrapper.md  # New project setup agent
│   ├── test-runner.md        # Execute test suites, report coverage
│   ├── pr-reviewer.md        # Automated PR diff review
│   ├── refactor-assistant.md # Safe refactoring with test validation
│   ├── ci-debugger.md        # Diagnose failed CI/CD runs
│   ├── release-coordinator.md # Version bump, changelog, deploy validation
│   ├── doc-generator.md      # API docs, ADRs, changelogs from code
│   ├── codebase-explainer.md # Onboarding — explain architecture, trace flows
│   ├── migration-validator.md # Database migration safety checks
│   ├── deployment-validator.md # Pre-deployment checklist validation
│   ├── dependency-auditor.md # Vulnerability and outdated package audit
│   ├── api-validator.md      # OpenAPI spec and contract validation
│   ├── product-analyst.md    # Feature adoption, analytics instrumentation
│   ├── ux-researcher.md      # Usability analysis, friction mapping
│   ├── roadmap-strategist.md # RICE scoring, dependency mapping, roadmaps
│   ├── competitive-intel.md  # Feature matrices, positioning, moat analysis
│   ├── content-strategist.md # Editorial calendars, SEO, content briefs
│   ├── campaign-analyst.md   # Attribution, funnel analysis, channel ROI
│   ├── brand-guardian.md     # Voice/tone, visual identity, microcopy audit
│   ├── growth-analyst.md     # Activation, retention, viral mechanics
│   ├── financial-analyst.md  # Revenue forecasts, unit economics, scenarios
│   ├── market-intelligence.md # TAM/SAM/SOM, trends, market timing
│   ├── investor-relations.md # Board updates, KPIs, fundraising narratives
│   ├── contract-reviewer.md  # SOW/contract risk, terms, IP review
│   ├── data-analyst.md       # Schema exploration, queries, data quality
│   ├── metrics-dashboard.md  # KPI definitions, SLOs, dashboard wireframes
│   ├── ab-test-analyst.md    # Experiment design, statistical analysis
│   ├── qa-engineer.md         # Test planning, edge cases, regression, quality gates
│   ├── accessibility-checker.md # WCAG 2.2 automated compliance
│   └── firebase-security-auditor.md # Firestore rules and Functions audit
├── hooks/                    # Multi-layer automated enforcement
│   └── hooks.json            # Command + Prompt hooks (9 event types)
├── rules/                    # 11 path-specific coding standards
│   ├── android.md             # Loads for *.kt files
│   ├── ios.md                 # Loads for *.swift files
│   ├── web.md                 # Loads for *.ts/*.tsx files
│   ├── firebase.md            # Loads for functions/**
│   ├── python.md              # Loads for *.py files
│   ├── go.md                  # Loads for *.go files
│   ├── rust.md                # Loads for *.rs files
│   ├── sql.md                 # Loads for *.sql, migrations/**
│   ├── docker.md              # Loads for Dockerfile, *.dockerfile
│   ├── terraform.md           # Loads for *.tf, *.tfvars
│   └── cicd.md                # Loads for .github/workflows/**
├── output-styles/            # 9 custom output formatting styles
│   ├── prd/                   # Product docs (PRDs, GTM, research)
│   ├── code-generation/       # Code scaffolds and implementations
│   ├── financial-analysis/    # Cost models, SaaS metrics
│   ├── audit-report/          # Audits, reviews, compliance
│   ├── api-specification/     # OpenAPI specs, endpoint docs
│   ├── architecture-decision/ # ADRs, RFCs, trade-off matrices
│   ├── runbook/               # Incident runbooks, DR procedures
│   ├── test-plan/             # Test plans, coverage reports
│   └── monitoring-alert/      # Alert definitions, thresholds
├── .mcp.json                 # MCP server configs (GitHub, Sentry, Firestore, PostgreSQL)
├── .lsp.json                 # LSP server configs (TypeScript, Python/Pyright)
├── marketplace.json          # Plugin marketplace manifest
├── settings.json             # Default permission rules
├── claude-commands/           # Legacy format (backwards compat, 64 files)
├── gemini skills/             # Google Gemini skills (.skill ZIP)
├── CLAUDE.md                  # Project instructions (Claude)
├── GEMINI.md                  # Project instructions (Gemini CLI)
├── AGENT-GUIDE.md             # How to structure prompts for agents & skills
├── setup.sh                  # Setup script for Antigravity & other projects
└── README.md

Installation

Via GitHub Package (Recommended)

Install the plugin as an npm package from GitHub Packages. This is the easiest way to keep all your projects up to date.

1. Authenticate with GitHub Packages (one-time setup):

# Create a Personal Access Token (PAT) with read:packages scope at
# https://github.com/settings/tokens, then:
npm login --scope=@cure-consulting-group --registry=https://npm.pkg.github.com

Or add to your project's .npmrc:

@cure-consulting-group:registry=https://npm.pkg.github.com
//npm.pkg.github.com/:_authToken=${GITHUB_TOKEN}

2. Install in your project:

npm install @cure-consulting-group/product-engineering-skills

The postinstall script automatically:

  • Symlinks the package to ~/.claude/plugins/ProductEngineeringSkills
  • Registers the plugin in ~/.claude/settings.json

All 80 skills, 39 agents, 4 personas, hooks, rules, and output styles are immediately available.

3. Enable auto-updates with Dependabot (recommended):

Add .github/dependabot.yml to your project (or run setup.sh which does this automatically):

version: 2
updates:
  - package-ecosystem: "npm"
    directory: "/"
    schedule:
      interval: "daily"
    allow:
      - dependency-name: "@cure-consulting-group/product-engineering-skills"
    labels:
      - "dependencies"
      - "skills-update"
    commit-message:
      prefix: "chore"
      include: "scope"

Dependabot will open a PR in your project whenever a new version is published. Merge it and every agent on that project gets the updated skills.

4. Manual update:

npm update @cure-consulting-group/product-engineering-skills

As a Claude Code Plugin (Manual)

# Load the plugin during a session
claude --plugin-dir /path/to/ProductEngineeringSkills

# Or for development/testing
claude --plugin-dir ./ProductEngineeringSkills

Once loaded, all skills are available as namespaced commands:

/cure-product-engineering:sdlc
/cure-product-engineering:feature-audit
/cure-product-engineering:android-feature-scaffold
/cure-product-engineering:incident-response
/cure-product-engineering:accessibility-audit

Hooks, agents, rules, output styles, and MCP servers are all included automatically.

Setup Script (Antigravity & Other Projects)

The fastest way to onboard any project:

# From the target project directory
/path/to/ProductEngineeringSkills/setup.sh

# Or specify the project path
/path/to/ProductEngineeringSkills/setup.sh /path/to/antigravity-app

# Install globally for ALL projects
/path/to/ProductEngineeringSkills/setup.sh --global

# Legacy mode (just copy skills, no hooks/agents)
/path/to/ProductEngineeringSkills/setup.sh --legacy

The setup script will:

  1. Clone/update the plugin to ~/.claude/plugins/
  2. Detect your project type (Android/iOS/Web/Firebase)
  3. Install only the relevant path-specific rules
  4. Create a project CLAUDE.md with skill references
  5. Add .claude/settings.local.json to .gitignore

Via Marketplace

# Add the Cure Consulting marketplace
claude marketplace add https://github.com/Cure-Consulting-Group/ProductEngineeringSkills/marketplace.json

# Install the plugin
claude plugin install cure-product-engineering

Legacy Method (Copy Commands)

Copy the claude-commands/ files into your project's .claude/commands/ directory:

cp claude-commands/*.md /path/to/your/project/.claude/commands/

Then use them as slash commands:

/sdlc — Generate SDLC artifacts
/android-feature-scaffold — Scaffold an Android feature module
/feature-audit — Audit a completed feature

Using with Google Gemini

Import the .skill files from gemini skills/ into your Gemini workspace. Each .skill file is a ZIP archive containing:

  • SKILL.md — The main skill definition
  • references/ — Supporting documents and templates

Skill Inventory (64 Skills)

Product & Strategy (7)

SkillWhat It DoesAuto-Invoked?
product-managerOKRs, roadmaps, RICE prioritization, feature briefsYes
product-designApple HIG, Material Design 3, design tokens, accessibility-firstYes
market-researchTAM/SAM/SOM, competitive analysis, ICP definition (read-only)Yes
go-to-marketGTM plans, launch strategy, channel selection, growth playbooksYes
product-marketingBrand strategy, messaging frameworks, campaignsYes
customer-onboardingActivation flows, empty states, email sequences, retentionYes
seo-content-engineTechnical SEO, structured data, content strategyYes

Engineering & Architecture (18)

SkillWhat It DoesAuto-Invoked?
sdlcPRDs, ADRs, RFCs, Epics, Stories, Task specs — full SDLCYes
android-feature-scaffoldClean Architecture Android scaffolding (MVI, Compose, Hilt)Yes
ios-architectSwift/SwiftUI Clean Architecture, MVVM, structured concurrencyYes
nextjs-feature-scaffoldApp Router, Server/Client components, Tailwind patternsYes
firebase-architectFirestore schema, security rules, Cloud FunctionsYes
api-architectREST/GraphQL design, versioning, auth, rate limitingYes
api-gatewayAPI gateway and BFF layers, rate limiting, GraphQL federationYes
stripe-integrationStripe payments + subscriptions via Firebase FunctionsYes
ai-feature-builderLLM integration, RAG pipelines, prompt engineeringYes
llmopsLLM operationalization — prompt versioning, eval pipelines, cost optimization, guardrailsYes
database-architectSchema design, migrations, indexing for Firestore/PostgreSQL/SQLiteYes
data-migrationETL pipelines, zero-downtime cutover, validation, rollback strategiesYes
infrastructure-scaffoldCloud infra configs for Firebase, GCP, Vercel, DockerYes
edge-computingEdge functions, CDN strategies, cache invalidation, edge middlewareYes
micro-frontendsModule federation, monorepo management, independent deploymentsYes
offline-firstOffline-first architecture, sync strategies, conflict resolution, optimistic UIYes
i18nInternationalization — string extraction, RTL, locale-aware formatting, translation workflowsYes
notification-architectPush (FCM/APNs), in-app messaging, email, preference managementYes

Quality & Security (11)

SkillWhat It DoesAuto-Invoked?
feature-audit5-phase post-completion audit with scored gap reportYes (read-only, forked)
testing-strategyTesting pyramid, platform standards, coverage rulesYes
e2e-testingE2E test suites with page objects, visual regression, CI integrationYes
test-accountsTest user personas, seed data scripts, environment credentialsYes
uatUAT plans, acceptance criteria checklists, go/no-go release gatesYes
security-reviewOWASP checklist, auth/data/API/mobile/web securityYes (read-only, forked)
compliance-architectHIPAA, COPPA, GDPR, PCI compliance frameworks, consent flows, audit trailsYes
accessibility-auditWCAG 2.2 compliance, screen readers, inclusive designYes (read-only, forked)
performance-reviewPerformance budgets, load testing, optimization strategiesYes
chaos-engineeringResilience testing, failure injection, graceful degradation, game daysYes
green-softwareSustainable software practices, carbon-aware computing, energy efficiencyYes

Operations & Delivery (12)

SkillWhat It DoesAuto-Invoked?
project-bootstrapBootstrap repo with CLAUDE.md + STATE.md via codebase inspection and developer interviewYes
project-managerSprint planning, RACI, risk registers, retrospectivesYes
ci-cd-pipelineGitHub Actions, build/test/deploy, environments, secretsYes
release-managementApp store submissions, staged rollouts, versioning, ASO, changelogsYes
feature-flagsProgressive rollouts, A/B testing, kill switches, experimentation frameworksYes
observabilityStructured logging, distributed tracing, alerting, SLO/SLI, dashboardsYes
dora-metricsDORA and SPACE metrics — deployment frequency, lead time, MTTR, developer experienceYes
analytics-implementationEvent taxonomy, tracking plans, funnels, dashboardsYes
incident-responseRunbooks, severity classification, post-mortems, escalationYes
disaster-recoveryDR and business continuity — RTO/RPO, backup strategies, failover, DR testingYes
growth-engineeringActivation funnels, referral programs, lifecycle automation, PLG patternsYes
design-systemDesign tokens, component libraries, Storybook/Catalog, cross-platform consistencyYes

Business & Finance (7)

SkillWhat It DoesAuto-Invoked?
engineering-cost-modelProject estimates, infrastructure costs, build vs buyYes (read-only)
saas-financial-modelUnit economics, MRR/ARR, pricing tiers, break-evenYes (read-only)
finopsCloud cost optimization, budget alerts, resource right-sizing, FinOps practicesYes
investor-reportingInvestor updates, board decks, portfolio financials, cap table, runway modelingYes
fundraising-materialsPitch decks, data rooms, investor updates, cap table scenarios, fundraising pipelineYes
burn-rate-trackerBurn rates, runway scenarios, break-even analysis, cash flow projectionsYes
legal-doc-scaffoldToS, Privacy Policy, SOW, NDA scaffoldsNo (manual only)

Portfolio Management (2) — NEW

SkillWhat It DoesAuto-Invoked?
portfolio-registryProduct portfolio registry — single source of truth for all products, stacks, teams, stagesYes
technology-radarThoughtWorks-style technology radar — Adopt/Trial/Assess/Hold across the portfolioYes

Consulting Operations (3)

SkillWhat It DoesAuto-Invoked?
client-handoffHandoff packages, runbooks, credential transfers, maintenance SLAs, knowledge transferYes
client-communicationSprint demo scripts, stakeholder updates, risk escalation, executive summariesYes
proposal-generatorConsulting proposals, SOWs, milestone pricing, engagement structureNo (manual only)

Platform Design (4)

SkillWhat It DoesAuto-Invoked?
android-design-expertMaterial Design 3 — dynamic color, component tokens, adaptive layouts, motion, Compose patternsYes
ios-design-expertApple HIG — SF Symbols, Dynamic Type, navigation patterns, SwiftUI componentsYes
web-design-expertResponsive design, CSS architecture, design tokens, container queries, accessibility-first, TailwindYes
stitch-designAI-native UI design via Stitch MCP — vibe design, mockups, screen generation, design tokens, component exportYes

Recurring Automation (Loops & Routines)

The library ships a standard maintenance loop (self-provisioned to .claude/loop.md on first session start — run bare /loop to use it), Recurring Mode sections in the goal-shaped skills (finops, burn-rate-tracker, investor-reporting, security-review, and others), and copy-paste cloud-routine recipes in docs/AUTOMATION.md. Library upkeep cadence: docs/MAINTENANCE.md. Mechanism selection and unattended-run guardrails: /cure-product-engineering:engagement-automation.

Hooks (Multi-Layer Automated Enforcement)

The plugin ships command and prompt hooks across 9 event types: SessionStart, PreCompact, PostCompact, ConfigChange, PostToolUseFailure, UserPromptSubmit, PreToolUse, Stop, SubagentStop. Highlights: a Stop-hook quality gate (blocks "done" without verification), a PreToolUse static security guard on skill/agent/persona files, and a ConfigChange audit trigger when skill files change mid-session.

New in v4.0: Hooks now suggest and auto-trigger agents based on context. Every code edit, test run, deployment, and PR action recommends the most relevant agent(s).

Command Hooks (Deterministic)

HookEventWhat It DoesAgent Integration
WelcomeSessionStartConfirms plugin loaded with counts; full inventory stays in docs/OVERVIEW.mdPoints to inventory
Git statusSessionStartReports current branch, uncommitted changes, last commit—
Dependency checkSessionStartDetects outdated packagesSuggests dependency-auditor
Code edit advisorPostToolUse (Edit/Write)Context-aware suggestions based on file type (.kt, .swift, .ts, .sql, .tf, etc.)Suggests code-reviewer, test-runner, brand-guardian, migration-validator
Command advisorPostToolUse (Bash)Post-action guidance for tests, installs, deploys, PRs, releasesSuggests ci-debugger, dependency-auditor, pr-reviewer, release-coordinator
Failure recoveryPostToolUseFailureDiagnoses failure type and suggests fix approachAuto-suggests ci-debugger, deployment-validator, dependency-auditor
Destructive prompt guardUserPromptSubmitDetects destructive operations in promptsBlocks and confirms
Protected filesPreToolUse (Edit/Write)Blocks edits to .env, lock files, credentials, tfstate—
Dangerous commandsPreToolUse (Bash)Blocks force push, destructive rm, DROP TABLE, prod deploys—
Context re-injectionPreCompactRe-injects all 80 skills, 39 agents, 4 personas, and Cure standardsFull inventory preserved
Post-compact restorePostCompactConfirms context restored with agent availability—
Subagent start bannerSubagentStartAnnounces agent with role, standards, and companion agentsLists companion agents
Subagent completionSubagentStopSuggests follow-up agents (test-runner, code-reviewer, pr-reviewer)Agent chaining
Task quality checkTaskCompletedValidates tests, security, docs, brand consistencySuggests test-runner, code-reviewer, doc-generator, brand-guardian

Prompt Hooks (LLM-Validated)

HookEventWhat It DoesAgent Integration
Code quality gatePreToolUse (Edit/Write)Haiku validates: no secrets, no debug logs, no disabled tests, no any types—
Deployment safetyPreToolUse (Bash)Haiku validates: blocks production deployments outside CI/CD—
Intent classifierUserPromptSubmitHaiku classifies prompt intent and suggests the most relevant agent(s) from all 30Maps prompts → agents with confidence scores

Agent Hooks (Multi-Turn Verification)

HookEventWhat It DoesAgent Integration
Completion validatorStopValidates: tests for new code, security review for sensitive changes, rollback for migrations, docs for features, brand consistency for UI, analytics for events, API contractsSuggests specific agents for each gap found

MCP Server Integrations

Pre-configured MCP servers in .mcp.json:

ServerTypeWhat It Does
GitHubHTTPPR management, issue tracking, code search
SentryHTTPError monitoring, issue tracking, release health
FirestorestdioDirect database queries, schema inspection
PostgreSQLstdioDatabase queries, schema inspection, migrations

LSP Server Integrations

Pre-configured LSP servers in .lsp.json:

ServerLanguageWhat It Provides
TypeScript.ts, .tsx, .jsType checking, auto-imports, refactoring, go-to-definition
Python (Pyright)*.pyStatic type analysis, import resolution, error diagnostics

Output Styles

Custom output formatting for different artifact types:

StyleUsed ByKey Rules
prdProduct skills (PRDs, GTM, research)Numbered sections, decision matrices, executive summaries
code-generationEngineering skills (scaffolds)File tree first, dependency order, complete runnable code
financial-analysisBusiness skills (costs, models)ASCII tables, explicit assumptions, sensitivity analysis
audit-reportQuality skills (audits, reviews)Severity scoring, checklists, remediation with effort estimates
api-specificationAPI design skillsOpenAPI 3.0 blocks, endpoint tables, request/response examples
architecture-decisionADR and RFC skillsContext/decision/consequences format, trade-off matrices
runbookIncident response, disaster recoveryNumbered steps, command blocks, decision trees, escalation paths
test-planTesting strategy, QA skillsCoverage tables, test case templates, pass/fail criteria
monitoring-alertObservability, incident responseAlert definition tables, threshold rationale, runbook links

Custom Agents (30)

Engineering Agents (14)

AgentPurposeToolsAuto-Triggered By
code-reviewerSecurity + quality review against Cure standardsRead-onlyStop hook, SubagentStop
project-bootstrapperSet up new projects with correct architecture
Source 3 files
hooks/register.tsx 104 lines
1import { atom, read, update } from 'claude-code'
2import type { EngineInterface, Register } from 'claude-code'
3
4import { EMPTY, bandText, bar, inferTtl, leftText, recordStep, reportText, tokensText, view, writeFeeText } from './cache'
5
6/**
7 * cure-cache-band: a row above the prompt showing how long the prompt cache
8 * has left, the session's hit rate and misses, and, once it has lapsed, how
9 * many tokens the next message will pay to re-cache. /cache gives the detail.
10 *
11 * It counts the main conversation's requests only: a subagent keeps a cache
12 * of its own. Nothing is fetched; every figure comes from the usage each
13 * model response already reports.
14 *
15 * Setting, as an environment variable:
16 *   CURE_CACHE_TTL  `5m` or `1h`, when the inferred TTL is wrong for your plan
17 *
18 * The TTL is inferred, not reported by the API (see inferTtl): the countdown
19 * is an estimate, the hit rate and misses are measured.
20 */
21
22const TICK_MS = 1000
23
24const cache = atom({ plugin: 'cure-cache-band', key: 'cache' } as const, EMPTY)
25
26let timer: { cancel: () => void } | undefined
27let drawn = ''
28
29export const register: Register = on => {
30  on('session.start', async ($, e, next) => {
31    await $.command.register({ name: 'cache', description: 'Show prompt cache time left, hit rate and misses for this session' })
32    timer?.cancel()
33    // Redraw only when the row's text would change: once a second in the last minute, once a minute before it.
34    timer = $.clock.every(TICK_MS, async () => {
35      const text = bandText(view(await read($, cache), await $.clock.now()))
36      if (text !== drawn) $.ui.invalidate('ui.render')
37    })
38    return next(e)
39  })
40
41  on('session.end', async ($, e, next) => {
42    timer?.cancel()
43    timer = undefined
44    return next(e)
45  })
46
47  on('turn.step', async function* ($, e, next) {
48    const result = yield* next(e)
49    const usage = result.usage
50    if (e.agentId === undefined && usage) {
51      const [now, ttl] = [await $.clock.now(), (await ttlNow($)).ttl]
52      await update($, cache, state => recordStep(state, usage, now, ttl))
53    }
54    return result
55  })
56
57  on('command.run', { command: 'cache' }, async $ => {
58    const { source } = await ttlNow($)
59    return { text: reportText(await read($, cache), await $.clock.now(), source) }
60  })
61
62  on('ui.render', { component: 'AbovePrompt' }, async ($, e, next) => {
63    const v = view(await read($, cache), await $.clock.now())
64    drawn = bandText(v)
65    if (v.kind === 'none' || e.props.hasSurvey) return next(e)
66
67    const { Box, Text } = $.ui.resolve(e)
68
69    if (v.kind === 'cold') {
70      const fee = writeFeeText(v.tokens)
71      return (
72        <Box>
73          <Text color="red">cache ○ cold</Text>
74          <Text dimColor> · next message re-caches {tokensText(v.tokens)} tokens{fee ? ` (${fee})` : ''}</Text>
75        </Box>
76      )
77    }
78
79    const color = v.level === 'low' ? 'yellow' : 'green'
80    const b = bar(v.fraction)
81    return (
82      <Box>
83        <Text color={color}>
84          cache ● {v.ttl} {b.filled}
85        </Text>
86        <Text dimColor>{b.empty}</Text>
87        <Text color={color}> {leftText(v.leftMs)}</Text>
88        <Text dimColor>
89          {' '}
90          · hit {v.hitPercent}% · misses {v.misses}
91        </Text>
92      </Box>
93    )
94  })
95}
96
97async function ttlNow($: EngineInterface) {
98  const override = (await $.env.get('CURE_CACHE_TTL')) || undefined
99  const { rateLimits } = await $.session.usage()
100  const ttl = inferTtl(override, rateLimits)
101  const source = override === ttl ? 'CURE_CACHE_TTL' : rateLimits.length === 0 ? 'inferred: no subscription limits reported' : 'inferred from subscription limits'
102  return { ttl, source }
103}
104
hooks/cache.ts 144 lines
1import type { CacheState, CacheTtl } from '../types'
2
3/** What one model response reported, as `turn.step`'s `usage` carries it. */
4export type StepUsage = {
5  input_tokens: number
6  output_tokens: number
7  cache_read_input_tokens: number
8  cache_creation_input_tokens: number
9}
10
11export type RateLimit = { kind: string; percentUsed: number }
12
13export type View =
14  | { kind: 'none' }
15  | { kind: 'warm'; level: 'ok' | 'low'; ttl: CacheTtl; leftMs: number; fraction: number; hitPercent: number; misses: number }
16  | { kind: 'cold'; tokens: number }
17
18export const TTL_MS: Record<CacheTtl, number> = { '5m': 5 * 60 * 1000, '1h': 60 * 60 * 1000 }
19
20/** Under this share of the TTL left, the band turns amber. */
21export const LOW_FRACTION = 0.25
22/** A prefix shorter than this is below the API's minimum cacheable length; reading none of it is not a miss. */
23const MIN_CACHEABLE = 1024
24export const BAR_CELLS = 16
25
26export const EMPTY: CacheState = {
27  lastAt: null,
28  ttl: '1h',
29  prefixTokens: 0,
30  nextTokens: 0,
31  read: 0,
32  written: 0,
33  uncached: 0,
34  requests: 0,
35  misses: 0,
36}
37
38/**
39 * The TTL the session's cache entries are assumed to have. The API does not
40 * report it per response, so this is inferred: CURE_CACHE_TTL when set;
41 * otherwise 1h on a subscription inside its limits, 5m on an API key (no
42 * rate-limit windows) or once a window is exhausted.
43 */
44export function inferTtl(override: string | undefined, rateLimits: readonly RateLimit[]): CacheTtl {
45  if (override === '5m' || override === '1h') return override
46  if (rateLimits.length === 0) return '5m'
47  return rateLimits.some(l => l.percentUsed >= 100) ? '5m' : '1h'
48}
49
50/**
51 * Folds one main-thread response into the session's figures. A miss is a
52 * request, after the first, that read less than half of the prefix the
53 * request before it left in the cache: an expiry, a model switch, a
54 * compaction or an edited prefix all land here, and all cost a re-write.
55 */
56export function recordStep(state: CacheState, usage: StepUsage, now: number, ttl: CacheTtl): CacheState {
57  const prefix = usage.input_tokens + usage.cache_read_input_tokens + usage.cache_creation_input_tokens
58  // The first request has no prefix before it, so it is never a miss.
59  const isMiss = state.prefixTokens >= MIN_CACHEABLE && usage.cache_read_input_tokens < state.prefixTokens / 2
60  return {
61    lastAt: now,
62    ttl,
63    prefixTokens: prefix,
64    nextTokens: prefix + usage.output_tokens,
65    read: state.read + usage.cache_read_input_tokens,
66    written: state.written + usage.cache_creation_input_tokens,
67    uncached: state.uncached + usage.input_tokens,
68    requests: state.requests + 1,
69    misses: state.misses + (isMiss ? 1 : 0),
70  }
71}
72
73/** Share of all prompt tokens this session that the cache served, as a whole percentage. */
74export function hitPercent(state: CacheState): number {
75  const total = state.read + state.written + state.uncached
76  return total === 0 ? 0 : Math.round((state.read / total) * 100)
77}
78
79export function view(state: CacheState, now: number): View {
80  if (state.lastAt === null) return { kind: 'none' }
81  const total = TTL_MS[state.ttl]
82  const leftMs = state.lastAt + total - now
83  if (leftMs <= 0) return { kind: 'cold', tokens: state.nextTokens }
84  const fraction = Math.min(1, leftMs / total)
85  return {
86    kind: 'warm',
87    level: fraction < LOW_FRACTION ? 'low' : 'ok',
88    ttl: state.ttl,
89    leftMs,
90    fraction,
91    hitPercent: hitPercent(state),
92    misses: state.misses,
93  }
94}
95
96/** `59m left` above a minute, `53s left` under it; never rounds up to the full TTL's next unit. */
97export function leftText(leftMs: number): string {
98  const seconds = Math.max(1, Math.ceil(leftMs / 1000))
99  return seconds > 60 ? `${Math.floor(seconds / 60)}m left` : `${seconds}s left`
100}
101
102export function tokensText(tokens: number): string {
103  if (tokens >= 1_000_000) return `${(tokens / 1_000_000).toFixed(1)}M`
104  return tokens >= 1000 ? `${Math.round(tokens / 1000)}k` : String(tokens)
105}
106
107/** Estimated write fee for re-caching tokens at Claude 3.5 Sonnet's write rate ($3.75 / M tokens). */
108export function writeFeeText(tokens: number): string {
109  if (tokens <= 0) return ''
110  const fee = (tokens / 1_000_000) * 3.75
111  return fee < 0.01 ? '<$0.01 fee' : `~$${fee.toFixed(2)} fee`
112}
113
114/** The bar's two runs: at least one filled cell while any time is left. */
115export function bar(fraction: number, cells: number = BAR_CELLS): { filled: string; empty: string } {
116  const n = Math.min(cells, Math.max(1, Math.round(fraction * cells)))
117  return { filled: '█'.repeat(n), empty: '░'.repeat(cells - n) }
118}
119
120/** The band as one plain string: what the timer compares to know a redraw is due, and what /cache prints first. */
121export function bandText(v: View): string {
122  if (v.kind === 'none') return ''
123  if (v.kind === 'cold') {
124    const fee = writeFeeText(v.tokens)
125    return `cache ○ cold · next message re-caches ${tokensText(v.tokens)} tokens${fee ? ` (${fee})` : ''}`
126  }
127  const b = bar(v.fraction)
128  return `cache ● ${v.ttl} ${b.filled}${b.empty} ${leftText(v.leftMs)} · hit ${v.hitPercent}% · misses ${v.misses}`
129}
130
131export function reportText(state: CacheState, now: number, ttlSource: string): string {
132  const v = view(state, now)
133  if (v.kind === 'none') return 'cure-cache-band: no model response yet this session, so nothing is cached that the band has seen.'
134  return [
135    bandText(v),
136    `  TTL assumed           ${state.ttl} (${ttlSource})`,
137    `  Requests (main)       ${state.requests}, ${state.misses} missed`,
138    `  Read from cache       ${tokensText(state.read)} tokens`,
139    `  Written to cache      ${tokensText(state.written)} tokens`,
140    `  Uncached input        ${tokensText(state.uncached)} tokens`,
141    `  Next message re-sends ${tokensText(state.nextTokens)} tokens${v.kind === 'cold' ? `, all at the cache-write rate (${writeFeeText(state.nextTokens)})` : ''}`,
142  ].join('\n')
143}
144
types/index.d.ts 25 lines
1export type CacheTtl = '5m' | '1h'
2
3export type CacheState = {
4  /** When the last main-thread response arrived, in `$.clock.now()`'s milliseconds; null before the first. */
5  lastAt: number | null
6  /** The TTL assumed for the entry that response wrote or refreshed. */
7  ttl: CacheTtl
8  /** Prompt tokens the last request was answered over: what a warm cache holds. */
9  prefixTokens: number
10  /** Prompt tokens the next request re-sends: the prefix plus the last response's output. */
11  nextTokens: number
12  /** Session totals over main-thread requests. */
13  read: number
14  written: number
15  uncached: number
16  requests: number
17  misses: number
18}
19
20declare module 'claude-code' {
21  interface PluginState {
22    'cure-cache-band': { cache: CacheState }
23  }
24}
25