SLOPSHOPPER

jev-model-router

Corrects the model tier of every subagent spawn (agent.spawn) with a live Jev judgment (haiku / sonnet / opus), instead of relying on /delegate's static…

newnetworkagents
A shopper browsing a rack in a slop shop
README

dotfiles

Configs for my machine.

AI tooling

All agent-related configuration lives under ai/: shared agent rules, Claude defaults, MCP definitions, and reusable skills. Run ./setup to install the shared rules, MCP configuration, and skills. Run ./ai/skills/install when you only need to refresh the skills.

Inspiration

Source 2 files
hooks/jev-model-router.ts 106 lines
1/**
2 * jev-model-router — corrects the model of every subagent spawn with
3 * TypeSafe's Jev, so /delegate's routing table doesn't have to be recalled
4 * and applied by hand each time a subagent is dispatched.
5 *
6 * Only `agent.spawn` is hooked. This does not make Claude decide to
7 * delegate in the first place — that decision still belongs to /delegate
8 * (or a separate nudge hook). It only fixes the model once a spawn happens,
9 * regardless of which subagent_type or model the caller picked.
10 *
11 * Three tiers: mechanical (haiku), ordinary (sonnet), deep (opus). A task
12 * Jev reads as risky (touches production, money, or irreversible data) is
13 * always forced to deep, regardless of confidence.
14 *
15 * Fail-open: a classification that errors, fails to parse, or runs past the
16 * latency budget leaves the spawn exactly as the caller built it.
17 *
18 * Needs CLAUDE_CODE_ENABLE_FUNCTION_HOOKS=1 (Claude Code >= 2.1.259).
19 * Typed against https://github.com/anthropics/claude-code/tree/main/mods
20 */
21import type { Register } from 'claude-code'
22import {
23  DEFAULT_BASE_URL,
24  DEFAULT_MODEL,
25  endpoint,
26  readDecision,
27  requestBody,
28  requestHeaders,
29  route,
30} from './policy.ts'
31import type { Decision, PolicyConfig } from './policy.ts'
32
33export const register: Register = (on, options) => {
34  const text = (key: string, fallback: string) =>
35    typeof options[key] === 'string' && options[key] ? (options[key] as string) : fallback
36  const number = (key: string, fallback: number) =>
37    typeof options[key] === 'number' ? (options[key] as number) : fallback
38  const flag = (key: string, fallback: boolean) =>
39    typeof options[key] === 'boolean' ? (options[key] as boolean) : fallback
40
41  const timeoutMs = number('timeoutMs', 800)
42  const logDecisions = flag('logDecisions', true)
43  const policy: PolicyConfig = {
44    minUpgradeConfidence: number('minUpgradeConfidence', 0.3),
45    minDowngradeConfidence: number('minDowngradeConfidence', 0.7),
46  }
47
48  let announced = false
49
50  on('agent.spawn', async ($, e, next) => {
51    // Resolve the key lazily and once: the plugin's own userConfig field
52    // first, then the environment variable the shell already exports,
53    // since this mod is personal and not distributed with a real key.
54    const apiKey = text('typesafeApiKey', '') || (await $.env.get('TYPESAFE_API_KEY')) || ''
55    const baseUrl = text('typesafeBaseUrl', DEFAULT_BASE_URL)
56    const modelId = text('typesafeModel', DEFAULT_MODEL)
57    const url = endpoint(baseUrl)
58
59    if (!announced) {
60      announced = true
61      if (logDecisions) {
62        $.ui.log(
63          apiKey
64            ? `[jev-model-router] ready on typesafe (${url})`
65            : '[jev-model-router] no TypeSafe key found (userConfig.typesafeApiKey or $TYPESAFE_API_KEY); every spawn left unchanged',
66        )
67      }
68    }
69
70    // A fork inherits its parent's model; `model` is ignored for it.
71    if (!apiKey || e.fork) return next(e)
72
73    const startedAt = await $.clock.now()
74    let decision: Decision | null = null
75    try {
76      const response = await Promise.race([
77        $.http.fetch(url, {
78          method: 'POST',
79          headers: requestHeaders(apiKey),
80          body: requestBody(
81            { prompt: e.prompt, description: e.description, subagentType: e.subagentType },
82            modelId,
83          ),
84        }),
85        $.clock.sleep(timeoutMs),
86      ])
87      if (response && response.ok) decision = readDecision(response.text)
88      else if (response) $.ui.log(`[jev-model-router] typesafe responded ${response.status}`)
89      else $.ui.log(`[jev-model-router] classification passed ${timeoutMs}ms; leaving the spawn alone`)
90    } catch (error) {
91      $.ui.log(`[jev-model-router] classification failed: ${String(error)}`)
92    }
93
94    const current = e.model ?? e.parentModel
95    const { model, reason } = route(decision, { model: current }, policy)
96
97    if (logDecisions) {
98      const ms = (await $.clock.now()) - startedAt
99      $.ui.log(`[jev-model-router] ${e.subagentType ?? 'agent'} (${ms}ms): ${reason}`)
100    }
101
102    if (!model) return next(e)
103    return next({ ...e, model })
104  })
105}
106
hooks/policy.ts 181 lines
1/**
2 * jev-model-router — pure decision logic. No `$`, no I/O: builds the
3 * TypeSafe request, reads its answer, and turns that answer into a model.
4 *
5 * Three tiers: `/delegate` itself only ever hands work to haiku or sonnet,
6 * but that was a rule about what *it* delegates, not a ceiling this router
7 * has to keep. A task Jev reads as genuinely hard (open-ended debugging,
8 * architecture, something with no decided approach yet) gets opus.
9 */
10
11export type Tier = 'mechanical' | 'ordinary' | 'deep'
12
13export const TIER_ORDER: readonly Tier[] = ['mechanical', 'ordinary', 'deep']
14
15export const TIER_MODEL: Record<Tier, string> = {
16  mechanical: 'haiku',
17  ordinary: 'sonnet',
18  deep: 'opus',
19}
20
21/** Descriptions taken from /delegate's routing table, plus a deep tier it doesn't have. */
22const TIER_CRITERIA = {
23  mechanical:
24    'Well-specified edit, bulk rename, boilerplate, formatting, or an already-diagnosed one-line fix. The root cause or exact change is already known; no investigation or design left to do.',
25  ordinary:
26    'A feature, bug fix, or refactor whose approach is decided but whose implementation still takes judgment across a few files.',
27  deep: 'Open-ended: the cause is unknown, the approach is undecided, or it is architecture, security, a data migration, or something whose shape has to be discovered while doing it.',
28}
29
30export interface Decision {
31  tier: Tier
32  /** Confidence in the tier, or null when the backend reported none. */
33  confidence: number | null
34  /** P(true) that getting this task wrong would be costly or hard to reverse. */
35  risky: number | null
36}
37
38const DEFAULT_BASE_URL = 'https://api.typesafe.ai'
39const DEFAULT_MODEL = 'jev-latest'
40
41export function endpoint(baseUrl: string): string {
42  return `${baseUrl.replace(/\/+$/, '')}/v1/systemone`
43}
44
45export function requestBody(
46  state: Record<string, unknown>,
47  model: string,
48): string {
49  return JSON.stringify({
50    model,
51    state,
52    questions: {
53      tier: {
54        type: 'choice',
55        instructions:
56          'Which is the cheapest tier of engineer that can complete this coding task well, unsupervised?',
57        criteria: TIER_CRITERIA,
58      },
59      risky: {
60        type: 'noul',
61        instructions:
62          'Carrying out this task wrong would itself be costly or hard to reverse (touches production, money, credentials, or data that cannot be restored). Writing or testing code that deals with such things, without running it against the real system, does not count.',
63      },
64    },
65  })
66}
67
68export function requestHeaders(apiKey: string): Record<string, string> {
69  return { 'content-type': 'application/json', authorization: `Bearer ${apiKey}` }
70}
71
72function isTier(value: unknown): value is Tier {
73  return value === 'mechanical' || value === 'ordinary' || value === 'deep'
74}
75
76function confidenceOf(answer: Record<string, unknown>): number | null {
77  if (typeof answer.confidence === 'number') return answer.confidence
78  const probabilities = answer.probabilities as Record<string, number> | undefined
79  const values = probabilities ? Object.values(probabilities) : []
80  return values.length > 0 ? Math.max(...values) : null
81}
82
83export function readDecision(responseText: string): Decision | null {
84  let parsed: unknown
85  try {
86    parsed = JSON.parse(responseText)
87  } catch {
88    return null
89  }
90  const answers = (parsed as { answers?: Record<string, Record<string, unknown>> }).answers
91  if (!answers) return null
92
93  const tierAnswer = answers.tier
94  if (!tierAnswer || !isTier(tierAnswer.choice)) return null
95
96  const riskyAnswer = answers.risky
97  const risky = typeof riskyAnswer?.noul === 'number' ? riskyAnswer.noul : null
98
99  return { tier: tierAnswer.choice, confidence: confidenceOf(tierAnswer), risky }
100}
101
102/** Where a model id/alias sits on the tier ladder, or null when it matches none. */
103export function rankOf(model: string | undefined): number | null {
104  if (!model) return null
105  const lowered = model.toLowerCase()
106  if (lowered.includes('haiku')) return 0
107  if (lowered.includes('sonnet')) return 1
108  if (lowered.includes('opus')) return 2
109  return null
110}
111
112export interface PolicyConfig {
113  minUpgradeConfidence: number
114  minDowngradeConfidence: number
115}
116
117export interface Routing {
118  model: string | null
119  reason: string
120}
121
122/**
123 * Whether a change of rank clears its threshold. The two mistakes do not
124 * cost the same: under-powering a task (downgrade) needs a high bar, paying
125 * a bit more for a task that didn't need it (upgrade) needs a low one. An
126 * unrecognised current value (a subagent_type this router doesn't know) is
127 * treated as an upgrade, the gentler threshold, never guessed at as risky.
128 */
129function allowed(
130  wanted: number,
131  current: number | null,
132  confidence: number | null,
133  config: PolicyConfig,
134): boolean {
135  if (current !== null && wanted === current) return false
136  const isDowngrade = current !== null && wanted < current
137  const bar = isDowngrade ? config.minDowngradeConfidence : config.minUpgradeConfidence
138  if (confidence === null) return !isDowngrade
139  return confidence >= bar
140}
141
142/**
143 * Turns a decision into a model, or null to leave the spawn as it is.
144 * Risky forces `ordinary` regardless of confidence — it raises the floor,
145 * never lowers one already above it (a task above `ordinary` cannot happen
146 * here since `ordinary` is the ceiling).
147 */
148export function route(
149  decision: Decision | null,
150  current: { model?: string },
151  config: PolicyConfig,
152): Routing {
153  if (!decision) return { model: null, reason: 'no decision' }
154
155  let tier = decision.tier
156  let forced = false
157  if (decision.risky !== null && decision.risky > 0.7 && tier !== 'deep') {
158    tier = 'deep'
159    forced = true
160  }
161
162  const wantedRank = TIER_ORDER.indexOf(tier)
163  const currentRank = rankOf(current.model)
164  const wantedModel = TIER_MODEL[tier]
165
166  const said = decision.confidence === null ? 'confidence n/d' : `confidence ${decision.confidence.toFixed(2)}`
167
168  if (wantedModel === current.model) {
169    return { model: null, reason: `kept ${current.model ?? 'default'}, already ${tier} (${said})` }
170  }
171  if (!forced && !allowed(wantedRank, currentRank, decision.confidence, config)) {
172    return {
173      model: null,
174      reason: `kept ${current.model ?? 'default'}, wanted ${wantedModel} (${said}, below threshold)`,
175    }
176  }
177  return { model: wantedModel, reason: forced ? `${tier}, forced by risk` : `${tier} (${said})` }
178}
179
180export { DEFAULT_BASE_URL, DEFAULT_MODEL }
181