Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 5 additions & 3 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -20,9 +20,11 @@ jobs:
ci:
runs-on: ubuntu-latest
steps:
# D13: SHA-pin these actions as a go-live hardening step (recorded in the runbook); version tags for now.
- uses: actions/checkout@v4
- uses: actions/setup-node@v4
# D13 (done): SHA-pinned. A tag is mutable - the action owner can silently repoint v5 at new code, which
# is how the trivy-action and kics-github-action compromises landed. Bump these deliberately, never by
# drift. Newer majors exist (checkout v7, setup-node v7); moving to them is a separate, tested change.
- uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5
- uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5
with:
node-version: '22'
cache: 'npm'
Expand Down
21 changes: 17 additions & 4 deletions .github/workflows/refresh-pricing.yml
Original file line number Diff line number Diff line change
Expand Up @@ -58,15 +58,18 @@ jobs:
fi

- id: app-token
uses: actions/create-github-app-token@v2
# SHA-pinned: this step mints a Contents:RW + PullRequests:RW token, and the job that follows can
# push to a branch that auto-merges to main and auto-deploys. A repointed tag here is a direct path
# to production, so it is the single most important pin in the repo.
uses: actions/create-github-app-token@fee1f7d63c2ff003460e3d139729b119787bc349 # v2
with:
app-id: ${{ secrets.REFRESH_BOT_APP_ID }}
private-key: ${{ secrets.REFRESH_BOT_PRIVATE_KEY }}

- uses: actions/checkout@v5
- uses: actions/checkout@fbc6f3992d24b796d5a048ff273f7fcc4a7b6c09 # v5
with:
token: ${{ steps.app-token.outputs.token }} # so the branch push is authored by the App
- uses: actions/setup-node@v5
- uses: actions/setup-node@a0853c24544627f65ddf259abe73b1d18a591444 # v5
with:
node-version: '22'
cache: npm
Expand All @@ -91,6 +94,7 @@ jobs:
R_SHORT: ${{ steps.refresh.outputs.short }}
R_DATE: ${{ steps.refresh.outputs.date }}
R_ANCHOR: ${{ steps.refresh.outputs.anchor_changed }}
R_TIER_REVIEW: ${{ steps.refresh.outputs.tier_review }}
BODY_PATH: ${{ runner.temp }}/refresh-pr-body.md
run: |
set -euo pipefail
Expand All @@ -113,7 +117,13 @@ jobs:
AUTOMERGE=false
fi
if [ "$R_ANCHOR" = "true" ]; then
echo "::warning::Anchor price changed; the E2E math oracles need a hand edit. Leaving the PR for a human."
echo "::warning::Anchor price changed. The E2E math oracles need a hand edit. Leaving the PR for a human."
AUTOMERGE=false
fi
# A tier change is invisible to the anchor fingerprint but reprices long-context forecasts by up
# to 2x, so it must never ride the unattended path.
if [ "$R_TIER_REVIEW" = "true" ]; then
echo "::warning::Price-tier structure changed on already-shipped models, or a threshold key became unreadable. Leaving the PR for a human."
AUTOMERGE=false
fi

Expand All @@ -122,6 +132,9 @@ jobs:
if [ "$R_ANCHOR" = "true" ]; then
TITLE="$TITLE [ANCHOR PRICE CHANGED - update the E2E oracles]"
fi
if [ "$R_TIER_REVIEW" = "true" ]; then
TITLE="$TITLE [TIER CHANGE - review before merge]"
fi
# An App-token PR fires `pull_request` normally, so CI starts on its own (no dispatch needed).
gh pr create --base main --head "$BRANCH" --title "$TITLE" --body-file "$BODY_PATH"

Expand Down
52 changes: 26 additions & 26 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

10 changes: 8 additions & 2 deletions package.json
Original file line number Diff line number Diff line change
Expand Up @@ -61,7 +61,7 @@
"globals": "14.0.0",
"jsdom": "29.1.1",
"node-fetch": "3.3.2",
"postcss": "8.5.16",
"postcss": "8.5.18",
"size-limit": "11.2.0",
"tailwindcss": "3.4.15",
"tsx": "4.19.2",
Expand All @@ -71,7 +71,13 @@
},
"overrides": {
"base64-js": "1.5.1",
"esbuild": "0.25.12"
"esbuild": "0.25.12",
"dompurify": "3.4.13",
"brace-expansion@1": "1.1.18",
"brace-expansion@2": "2.1.4",
"brace-expansion@3": "3.0.3",
"brace-expansion@>=4": "5.0.9",
"postcss": "8.5.18"
},
"engines": {
"node": ">=22"
Expand Down
7 changes: 5 additions & 2 deletions scripts/registry/buildRegistry.ts
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@

import { writeFileSync, renameSync, readFileSync } from 'node:fs';
import { createHash } from 'node:crypto';
import { normalizeEntry, dedupeRecords, type RawEntry } from '../../src/registry/normalize';
import { normalizeEntry, dedupeRecords, countUnparsedTierKeys, type RawEntry } from '../../src/registry/normalize';
import type { RegistrySnapshot, ModelRecord } from '../../src/types/registry';

// A4/P2-A3: pin to a specific upstream commit SHA, never `main`. The raw upstream body is VENDORED into the
Expand Down Expand Up @@ -49,8 +49,10 @@ export function buildSnapshot(
if (!isPlainObject(raw)) throw new Error('registry snapshot root must be a plain object');
const models: ModelRecord[] = [];
let droppedCount = 0;
let unparsedTierKeyCount = 0;
for (const [rawKey, entry] of Object.entries(raw)) {
if (rawKey === 'sample_spec') continue; // A4: meta/example entry, not a model; not counted
if (isPlainObject(entry)) unparsedTierKeyCount += countUnparsedTierKeys(entry);
const rec = isPlainObject(entry) ? normalizeEntry(rawKey, entry) : null;
if (rec === null) droppedCount++;
else models.push(rec);
Expand All @@ -61,6 +63,7 @@ export function buildSnapshot(
snapshotDate: date,
droppedCount,
conflictCount,
unparsedTierKeyCount,
models: [...deduped].sort(compareRecords),
};
}
Expand Down Expand Up @@ -97,7 +100,7 @@ async function main(): Promise<void> {
writeFileSync(tmp, JSON.stringify(snap));
renameSync(tmp, target);
console.log(
`registry: ${snap.models.length} models, ${snap.droppedCount} dropped, ${snap.conflictCount} conflicts`,
`registry: ${snap.models.length} models, ${snap.droppedCount} dropped, ${snap.conflictCount} conflicts, ${snap.unparsedTierKeyCount} unparsed tier keys`,
);
}

Expand Down
28 changes: 27 additions & 1 deletion scripts/registry/refresh.mjs
Original file line number Diff line number Diff line change
Expand Up @@ -93,19 +93,45 @@ const removed = [...oldKeys].filter((k) => !newKeys.has(k)).sort();
const newAnchors = anchorPrices(newSnap.models);
const anchorChanged = ANCHORS.filter((k) => oldAnchors[k] !== newAnchors[k]);

// Tier deltas are invisible to the anchor fingerprint (neither gpt-4o nor gpt-4o-mini has a tier), yet a
// changed above-threshold rate reprices a long-context forecast by up to 2x. Since the refresh now
// auto-merges and auto-deploys, any tier change on a model we already shipped, or any newly unreadable
// threshold key, has to stop the unattended path and wait for a human. Models that are NEW this refresh are
// not flagged: they are already listed in the added section and were never priced before.
const tierFingerprint = (snap) =>
new Map(snap.models.map((m) => [`${m.canonicalId}|${m.deployment}`, JSON.stringify(m.tiers ?? [])]));
const oldTiers = tierFingerprint(oldSnap);
const newTiers = tierFingerprint(newSnap);
const tierChanged = [...newTiers.entries()]
.filter(([k, v]) => oldTiers.has(k) && oldTiers.get(k) !== v)
.map(([k]) => k)
.sort();
const unparsedBefore = oldSnap.unparsedTierKeyCount ?? 0;
const unparsedNow = newSnap.unparsedTierKeyCount ?? 0;
// Latching, not edge-triggered. Gating on an INCREASE meant that once any unreadable key merged, every
// later refresh would auto-merge again while the snapshot still carried keys we cannot price. The
// invariant is "zero unreadable threshold keys", so any non-zero count holds the PR for a human.
const unparsedUp = unparsedNow > 0;
const tierReview = tierChanged.length > 0 || unparsedUp;

const headline = `Refreshed to LiteLLM @ \`${sha.slice(0, 8)}\` (${date}). ${newSnap.models.length} models (${oldSnap.models.length} before): ${added.length} added, ${removed.length} removed.`;
const anchorLine = anchorChanged.length
? `WARNING: anchor price changed for ${anchorChanged.join(', ')}. The hand-computed E2E math oracles (chatbot $143.75, etc.) will FAIL and must be updated by hand before merge. old ${JSON.stringify(oldAnchors)} new ${JSON.stringify(newAnchors)}`
: `Anchor prices unchanged (${ANCHORS.join(', ')}), so the E2E math oracles still hold.`;
const tierLine = tierReview
? `REVIEW REQUIRED: ${tierChanged.length} already-shipped model(s) changed their price-tier structure${unparsedUp ? `, and the snapshot carries ${unparsedNow} unreadable threshold key(s) (was ${unparsedBefore}) that may be pricing a model flat above a real cliff` : ''}. Auto-merge is disabled for this PR because a tier change silently reprices long-context forecasts. Check these before merging:\n${tierChanged.slice(0, 40).map((k) => `- ${k}`).join('\n')}${tierChanged.length > 40 ? `\n...and ${tierChanged.length - 40} more` : ''}`
: `No tier changes on already-shipped models, and no unreadable threshold keys (${unparsedNow}).`;
const list = (arr) => (arr.length ? arr.slice(0, 300).map((k) => `- ${k}`).join('\n') + (arr.length > 300 ? `\n…and ${arr.length - 300} more` : '') : '_none_');
const body = `${headline}\n\n${anchorLine}\n\n<details><summary>${added.length} added</summary>\n\n${list(added)}\n</details>\n\n<details><summary>${removed.length} removed</summary>\n\n${list(removed)}\n</details>\n\nAuto-generated by the weekly \`refresh-pricing\` workflow, which waits for this PR's \`ci\` run and then **merges it automatically once that run is green** — the full run, not the pre-PR quick gate, is what executes the hand-computed E2E math oracles, so it is the gate that catches a broken anchor price. Auto-merge is skipped and this PR waits for a human if the diff touches anything outside the pricing artifact allowlist, if an anchor price changed, or if \`ci\` is red.`;
const body = `${headline}\n\n${anchorLine}\n\n${tierLine}\n\n<details><summary>${added.length} added</summary>\n\n${list(added)}\n</details>\n\n<details><summary>${removed.length} removed</summary>\n\n${list(removed)}\n</details>\n\nAuto-generated by the weekly \`refresh-pricing\` workflow, which waits for this PR's \`ci\` run and then **merges it automatically once that run is green** — the full run, not the pre-PR quick gate, is what executes the hand-computed E2E math oracles, so it is the gate that catches a broken anchor price. Auto-merge is skipped and this PR waits for a human if the diff touches anything outside the pricing artifact allowlist, if an anchor price changed, or if \`ci\` is red.`;

console.log(headline);
console.log(anchorLine);
console.log(tierLine);
const bodyPath = process.env.REFRESH_BODY_PATH ?? '.refresh-pr-body.md';
writeFileSync(bodyPath, body);
setOutput('changed', 'true');
setOutput('sha', sha);
setOutput('short', sha.slice(0, 8));
setOutput('date', date);
setOutput('anchor_changed', anchorChanged.length ? 'true' : 'false');
setOutput('tier_review', tierReview ? 'true' : 'false');
2 changes: 1 addition & 1 deletion src/config/registry.generated.json

Large diffs are not rendered by default.

25 changes: 25 additions & 0 deletions src/engine/caching/__tests__/policy.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -44,4 +44,29 @@ describe('cache policy constants + write-rate-by-TTL (C6/C7/C14)', () => {
expect(typeof tEffFor('__proto__')).toBe('number');
expect(tEffFor('__proto__')).toBe(tEffFor('some-unlisted-provider')); // both hit the default
});

// Review finding: above a price cliff the CacheSpec field holds the BASE write rate, so the tier-aware
// rate must be passed in or every above-cliff cache write is billed at base.
it('uses the tier-aware 5-minute write rate when one is supplied', () => {
// base spec: 5-min write 3.75 (= base input 3.0 * 1.25). Above the cliff both double.
expect(writeRateForTtl(bSpec(), 'min5', 6.0, 7.5)).toBeCloseTo(7.5, 10);
// hr1 conformance now compares the TIER 5-min (7.5) against tier input 6.0 * 1.25 = 7.5, so it
// derives rather than bailing to null the way the base 3.75 did.
expect(writeRateForTtl(bSpec(), 'hr1', 6.0, 7.5)).toBeCloseTo(12.0, 10); // 6.0 * 2.0
});

// Deferred-item fix: upstream publishes a real 1-hour write rate for 124 SKUs. Prefer it over the
// UNVERIFIED WRITE_MULT.hr1 derivation, and prefer the tier-aware one above a cliff.
it('prefers upstream\'s published 1-hour write rate over the derivation', () => {
const spec = bSpec({ cacheWriteHr1PerMToken: 6.0 });
expect(writeRateForTtl(spec, 'hr1', 3.0)).toBeCloseTo(6.0, 10);
// A non-conforming 5-min rate would have made the derivation bail to null; the published rate stands.
expect(writeRateForTtl(bSpec({ cacheWritePerMToken: 3.9, cacheWriteHr1PerMToken: 6.0 }), 'hr1', 3.0)).toBeCloseTo(6.0, 10);
// Above a cliff the tier-aware published rate wins (claude-sonnet-4-5 ships $12/M above 200k).
expect(writeRateForTtl(spec, 'hr1', 6.0, 7.5, 12.0)).toBeCloseTo(12.0, 10);
});

it('still derives when upstream publishes no 1-hour rate', () => {
expect(writeRateForTtl(bSpec(), 'hr1', 3.0)).toBeCloseTo(6.0, 10); // 3.0 * WRITE_MULT.hr1
});
});
20 changes: 18 additions & 2 deletions src/engine/caching/policy.ts
Original file line number Diff line number Diff line change
Expand Up @@ -30,10 +30,26 @@ export function tEffFor(provider: string): number {
// C6: only trust a derived 1-hr write rate when the registry 5-min rate conforms to base*1.25.
const WRITE_RATE_TOLERANCE = 0.02; // relative tolerance on the (5-min == base*1.25) invariant

export function writeRateForTtl(cache: CacheSpec, ttl: CacheTtl, inputPrice: number): number | null {
const fiveMin = cache.cacheWritePerMToken;
// `effectiveFiveMin` is the tier-aware 5-minute write rate. Above a price cliff the CacheSpec's own field
// is the BASE rate, so reading it here billed every above-cliff write at base: 2x under on min5, and on hr1
// the base 5-min rate no longer matches base*1.25 against the tier-aware input, so the C6 conformance check
// returned null and the caller fell back to base again (3.2x under against upstream's real above-cliff
// 1-hour rate). Callers that know the tier pass it; the parameter is optional so the non-tiered callers and
// the unit tests keep the plain base-rate behaviour.
export function writeRateForTtl(
cache: CacheSpec,
ttl: CacheTtl,
inputPrice: number,
effectiveFiveMin?: number | null,
effectiveHr1?: number | null,
): number | null {
const fiveMin = effectiveFiveMin ?? cache.cacheWritePerMToken;
if (fiveMin === undefined) return null; // no write cost (Archetype A)
if (ttl === 'min5') return fiveMin; // the real data
// Upstream publishes a real 1-hour write rate for many SKUs. Prefer it over the WRITE_MULT derivation
// below, which is an explicitly UNVERIFIED modeling assumption. Real data beats a multiplier.
const publishedHr1 = effectiveHr1 ?? cache.cacheWriteHr1PerMToken;
if (publishedHr1 !== undefined && publishedHr1 !== null) return publishedHr1;
// C6: derive the 1-hr rate from the BASE input rate, not by scaling the stored 5-min field - but only
// when the stored 5-min rate actually equals base*1.25; a non-conforming SKU returns null (unknown),
// never a fabricated guess.
Expand Down
Loading