From 148c9af393611947e03d296fa1461a0e6497d89f Mon Sep 17 00:00:00 2001 From: GrillerGeek Date: Tue, 29 Sep 2026 15:06:00 -0400 Subject: [PATCH] fix(routing): guard input delivery and preserve ordinary-work evidence (v0.17.3) --- CHANGELOG.md | 6 + docs/contributing-agents.md | 2 +- docs/installation.md | 2 +- ...2026-09-29-dynamic-routing-verification.md | 67 ++++++++ plugin/.claude-plugin/plugin.json | 2 +- plugin/.codex-plugin/plugin.json | 2 +- plugin/README.md | 2 +- plugin/commands/quest.md | 6 + plugin/plugin.json | 2 +- plugin/portable/references/global-routing.md | 14 +- plugin/portable/references/hosts.md | 6 + plugin/portable/references/input-delivery.md | 104 ++++++++++++ .../references/qualification-study.md | 6 + plugin/portable/references/quest.md | 6 + plugin/portable/references/routing.md | 6 + plugin/portable/scripts/routing_delivery.py | 109 +++++++++++++ plugin/portable/scripts/routing_evidence.py | 2 +- plugin/portable/scripts/routing_feedback.py | 64 ++++++++ plugin/portable/scripts/study_runner.py | 75 ++++++++- plugin/routing-setup/references/setup-flow.md | 6 + .../references/global-routing.md | 14 +- .../guildhall-quest/references/hosts.md | 6 + .../references/input-delivery.md | 104 ++++++++++++ .../references/qualification-study.md | 6 + .../guildhall-quest/references/quest.md | 6 + .../guildhall-quest/references/routing.md | 6 + .../scripts/routing_delivery.py | 109 +++++++++++++ .../scripts/routing_evidence.py | 2 +- .../scripts/routing_feedback.py | 64 ++++++++ .../guildhall-quest/scripts/study_runner.py | 75 ++++++++- .../references/global-routing.md | 14 +- .../references/input-delivery.md | 104 ++++++++++++ .../references/qualification-study.md | 6 + .../references/routing.md | 6 + .../references/setup-flow.md | 6 + .../scripts/routing_delivery.py | 109 +++++++++++++ .../scripts/routing_evidence.py | 2 +- .../scripts/routing_feedback.py | 64 ++++++++ .../scripts/study_runner.py | 75 ++++++++- scripts/build_portable.py | 2 +- tests/test_routing_delivery.py | 154 ++++++++++++++++++ tests/test_routing_study.py | 6 + 42 files changed, 1377 insertions(+), 52 deletions(-) create mode 100644 docs/reviews/2026-09-29-dynamic-routing-verification.md create mode 100644 plugin/portable/references/input-delivery.md create mode 100644 plugin/portable/scripts/routing_delivery.py create mode 100644 plugin/portable/scripts/routing_feedback.py create mode 100644 plugin/skills/guildhall-quest/references/input-delivery.md create mode 100644 plugin/skills/guildhall-quest/scripts/routing_delivery.py create mode 100644 plugin/skills/guildhall-quest/scripts/routing_feedback.py create mode 100644 plugin/skills/guildhall-routing-setup/references/input-delivery.md create mode 100644 plugin/skills/guildhall-routing-setup/scripts/routing_delivery.py create mode 100644 plugin/skills/guildhall-routing-setup/scripts/routing_feedback.py create mode 100644 tests/test_routing_delivery.py diff --git a/CHANGELOG.md b/CHANGELOG.md index 4b71a24..06e88d7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,11 @@ # Changelog +## 0.17.3 — Unreleased + +- Add bounded handoff chunks and host-visible completeness checks with budgeted recovery. +- Exclude incomplete/unknown live inputs before optional grading and model comparisons. +- Record ordinary-work feedback without extra model calls, automatic training or policy changes. + ## 0.17.2 — Unreleased - Make Dynamic routing the ordinary automatic-selection wizard path on Codex and Claude. diff --git a/docs/contributing-agents.md b/docs/contributing-agents.md index a655c08..04e4e4f 100644 --- a/docs/contributing-agents.md +++ b/docs/contributing-agents.md @@ -147,7 +147,7 @@ workers or claim synthetic savings as observed results. See the ## Distribution and versioning -The current package version is 0.17.2; increment all three manifests for further +The current package version is 0.17.3; increment all three manifests for further installer-visible changes. Existing native Claude agents/commands/hooks remain preserved. Never rewrite historical plans to claim newer evidence. No repository change implicitly installs personally, publishes a release or edits the separate diff --git a/docs/installation.md b/docs/installation.md index b60d719..a965523 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -1,6 +1,6 @@ # Guildhall installation and host support -Version **0.17.2 is a release candidate**, available through the `main` commands +Version **0.17.3 is a release candidate**, available through the `main` commands below after merge. It includes all-role routing eligibility, usage/evidence/study tools, independent `guildhall-quest` and `guildhall-routing-setup` skills and native Codex metadata. Claude's `/guildhall:quest`, nineteen agent definitions and hooks remain available. Choose one route per quest to avoid duplicate entry diff --git a/docs/reviews/2026-09-29-dynamic-routing-verification.md b/docs/reviews/2026-09-29-dynamic-routing-verification.md new file mode 100644 index 0000000..c44ba9b --- /dev/null +++ b/docs/reviews/2026-09-29-dynamic-routing-verification.md @@ -0,0 +1,67 @@ +# Practical dynamic routing verification + +The four-PR stack implements task-level Jev routing on Codex, native Claude and +standalone Claude. Final package version is 0.17.3. Installation remains off; +users can explicitly approve Dynamic routing without a qualification study. +Advanced adaptive retains its benchmark requirements. No personal settings, +marketplaces or saved study results were changed during this implementation. + +## Stack and behavior + +1. [PR 47](https://github.com/GrillerGeek/guildhall/pull/47): v5 dynamic contracts, + reviewed controls, versioned approvals, fallback and router-identity handling. +2. [PR 48](https://github.com/GrillerGeek/guildhall/pull/48): pinned host catalogs, + controlled task briefs, meaningful choice criteria and privacy boundaries. +3. [PR 49](https://github.com/GrillerGeek/guildhall/pull/49): wizard and all three + host routes, explicit role locks versus Claude/default fallbacks, migration. +4. `routing/delivery-integrity`: bounded handoffs, study delivery guards and local + feedback from ordinary work. This branch targets PR 49's branch for review. + +Review and merge in order. The first three PRs have passing GitHub validation and +pinned/latest installer checks. The final PR should ship with the feature, while +benchmark qualification remains optional. These intermediate package versions +are a review stack, not a request to publish or install each step personally. + +## Delivery and feedback + +Required references become bounded identified chunks. Host-visible content is +checked for omissions, changed bytes, explicit/recognized truncation, duplicate +attempts and cumulative read budgets. Recovery appends only missing authorized +chunks without resetting time/attempt counts or replaying a worker. All three +host routes share the checker; instructions specify actual host output controls. + +New live studies cannot claim workers without a frozen expected-input manifest. +Invalid prior live delivery stops further claims before spending on later trials. +Incomplete/unknown live inputs are excluded from blind grading; grading refuses +them, and export withholds the whole model comparison rather than cherry-picking +valid-looking outputs. Original outcomes, grades and consumed usage remain. +Legacy live delivery without evidence stays unknown; historical synthetic +arithmetic remains explicitly synthetic. + +Ordinary-work feedback reuses existing test/review outcomes, requested/trusted +observed settings and normalized usage. It correlates task, host, policy and +worker scope, keeps cache/reasoning accounting, leaves unknown subscription +allowance/cost unknown, and proposes reviewed catalog updates only. Known forced +substitution suspends choices without replay; unknown served identity does not. +No helper launches models, adds graders, trains or automatically edits policy. + +## Validation and limits + +Offline suite: 178 tests on Python 3.12 and 3.14, including all-role/all-host +routing, distinct same-role tasks, approval/scope/lock/fallback behavior, installed +wizard operations, privacy, truncation/recovery, invalid study exclusion and +normal-work feedback. Native validator: 19 agents, no errors/warnings. Portable +validator, generated drift checks, skill validation and whitespace checks pass. + +Isolated native Codex and Claude installs passed exact bytes/modes and removal of +the source fixture. The pinned skills 1.5.25 installer passed both skills +independently, for Codex and Claude, from explicit bundle paths and repository-root +discovery, with source removal. Installer receipts are local temporary artifacts; +no personal profile or credentials were copied into those fixtures. + +Fake providers and supplied synthetic host records establish wiring and failure +handling, not live recommendation quality or savings. No paid workers, Jev calls, +qualification studies or regrading runs were started. Delivery checks validate +supplied host-visible evidence; they cannot authenticate a fabricated source +label or prove comprehension. Hosts lacking visibility report unknown. Codex +catalog setup uses the actual exposed roster; no universal model list is shipped. diff --git a/plugin/.claude-plugin/plugin.json b/plugin/.claude-plugin/plugin.json index 944214c..55224a4 100644 --- a/plugin/.claude-plugin/plugin.json +++ b/plugin/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "guildhall", - "version": "0.17.2", + "version": "0.17.3", "description": "The Guildhall \u2014 a gathering place for adventurers. A TDD-ordered coding agent harness for Claude Code, tuned for Opus-tier orchestration (Opus 5 recommended seat). The /quest slash command runs Mordain the Guildmaster, who writes a durable plan file, then dispatches 18 specialist adventurers across three tiers: Opus (architecture-reviewer, security-reviewer, reliability-reviewer, migration-safety-reviewer), Sonnet (test-author, feature-implementer, ui-test-author, docs-writer, pr-author, prototype-builder, debug-investigator, observability-reviewer, performance-reviewer, ops-readiness-reviewer, accessibility-reviewer), and Haiku (refactorer, plugin-validator, fog-cartographer). Post-green reviews fan out in parallel \u2014 two always-on (security, docs) plus six gated production-readiness reviewers (observability, reliability, performance, ops-readiness, migration-safety, accessibility) that fire only when their trigger matches the diff. Rook (pr-author) closes the quest with a platform-agnostic PR draft and folds the runbook into the body. Integrates with IDD-framework specs.", "author": { "name": "GrillerGeek" diff --git a/plugin/.codex-plugin/plugin.json b/plugin/.codex-plugin/plugin.json index cd574e7..8bb08d3 100644 --- a/plugin/.codex-plugin/plugin.json +++ b/plugin/.codex-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "guildhall", - "version": "0.17.2", + "version": "0.17.3", "author": { "name": "GrillerGeek" }, diff --git a/plugin/README.md b/plugin/README.md index 09db736..a4eb597 100644 --- a/plugin/README.md +++ b/plugin/README.md @@ -44,7 +44,7 @@ Full character sheets in [`CHARACTERS.md`](CHARACTERS.md). ## Installation -Version **0.17.2 includes the installed routing tools**. The [installation guide](https://github.com/GrillerGeek/guildhall/blob/main/docs/installation.md) +Version **0.17.3 includes the installed routing tools**. The [installation guide](https://github.com/GrillerGeek/guildhall/blob/main/docs/installation.md) covers this repository's Codex, Claude and standalone routes, updates and removal. The `main` route receives 0.16.0 after merge; the separate marketplace below is not updated by this change. diff --git a/plugin/commands/quest.md b/plugin/commands/quest.md index 9485fcd..41a6ec4 100644 --- a/plugin/commands/quest.md +++ b/plugin/commands/quest.md @@ -405,3 +405,9 @@ The tone is a Guildmaster's fireside account, not a machine's log. Keep it truth For schema-v3/v4 routing, read `${CLAUDE_PLUGIN_ROOT}/skills/guildhall-quest/references/host-evidence.md`. Run preflight before paid studies; compare reviewed observations after each worker and suspend on drift without replay. Schema v4 supports all 18 specialist roles under explicit allowlists and scoped qualification. Follow the [role matrix and migration guide](../skills/guildhall-quest/references/role-eligibility.md); upgrading never enables a role automatically. + + +Use [bounded input delivery and feedback](../skills/guildhall-quest/references/input-delivery.md) +for required worker references on every path. Record host-visible completeness +or unknown, repair only missing authorized chunks within the existing budget, +and retain actual outcomes/usage in the plan without extra grading workers. diff --git a/plugin/plugin.json b/plugin/plugin.json index 265f745..7a9be65 100644 --- a/plugin/plugin.json +++ b/plugin/plugin.json @@ -1,7 +1,7 @@ { "$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json", "name": "guildhall", - "version": "0.17.2", + "version": "0.17.3", "author": { "name": "GrillerGeek" }, diff --git a/plugin/portable/references/global-routing.md b/plugin/portable/references/global-routing.md index 2ecc50e..9e4f0d1 100644 --- a/plugin/portable/references/global-routing.md +++ b/plugin/portable/references/global-routing.md @@ -63,8 +63,8 @@ Use an ordinary setup conversation outside quest execution. For example: > Configure Guildhall specialist model routing globally for this host. Read the > installed global and model-routing guides. Discover supported model/effort > settings without paid probes. Prepare an off global policy, keep unknown metrics -> and qualifications unknown, and show the proposed shadow policy, all-project -> scope, objective, candidates, budgets, outbound fields and host evidence. +> and qualifications unknown, and show the proposed Dynamic policy, all-project +> scope, objective, candidates, budgets, outbound fields and worker controls. > Activate only after I explicitly approve that exact proposal. Retain any > existing project policy and explain which source currently wins. @@ -83,16 +83,17 @@ different source do not copy consent. Credentials stay in the host environment, using the policy's `key_env`; never store the secret value in either file. Each approval binds the effective canonical policy hash, source and host route, -current host configuration fingerprint, and reviewed evidence hashes. Changing +current host configuration fingerprint, and either v5 reviewed control facts or +legacy/qualified reviewed evidence hashes. Changing another host's global entry or JSON formatting does not invalidate it. Changing the selected policy or host configuration requires renewed activation. Expiry, -revocation, missing reviewed host evidence or corrupt approval state prevents +revocation, missing required control/evidence review or corrupt approval state prevents reuse. Approval expiry may be omitted; qualification still has its own expiry. When reviewed host evidence has an expiry, do not approve beyond that expiry. The helper fingerprints the supplied host request fields (sorting supported settings), excluding the evidence hash itself; that hash must separately appear -in approved evidence. Supply fresh, truthful host metadata each time, not a stale +in approved evidence. V5 uses the control fingerprint described below instead. Supply fresh, truthful host metadata each time, not a stale snapshot to keep approval working. Host/role qualification checks still happen in the routing engine. `ready` means reusable consent, not adaptive qualification or proof of which model executed. @@ -165,7 +166,8 @@ recognize a valid opt-out or valid off policy, and skip both Python helpers. Check global defaults when the project file is absent. If selected configuration cannot be validated, stop configuration resolution; never guess that it is off. -For enabled routing, call `status` with current host evidence. Use its `policy` +For enabled routing, call `status` with current host controls/evidence shaped to +the selected request version. V5 dynamic requires no execution-evidence hashes. Use its `policy` and `activation` verbatim in the existing routing request, setting request schema version to the selected policy's schema version. Add a summary hash only after the exact summary's approval. Non-ready results do not authorize external calls; diff --git a/plugin/portable/references/hosts.md b/plugin/portable/references/hosts.md index 984921f..9a5c8d6 100644 --- a/plugin/portable/references/hosts.md +++ b/plugin/portable/references/hosts.md @@ -105,3 +105,9 @@ worker and bundled role body without assuming native registration or hooks. Codex uses actual exposed model/effort arguments and a fresh independent context; missing served identity alone is not a reason to demand a study. Never copy effort names between hosts. Reuse the same control approval across role fallback choices. + + +For required reference material, use [bounded input delivery](input-delivery.md) +and record ordinary-work feedback from existing tests/reviews/usage only. Repair +missing chunks within the existing budget; never treat source hashes or worker +assertions as delivery proof or run extra workers to collect feedback. diff --git a/plugin/portable/references/input-delivery.md b/plugin/portable/references/input-delivery.md new file mode 100644 index 0000000..6e94ef5 --- /dev/null +++ b/plugin/portable/references/input-delivery.md @@ -0,0 +1,104 @@ +# Bounded input delivery and ordinary-work feedback + +A correct source file and its hash do not prove that a worker received the whole +file. Treat delivery as a separate fact from model identity and output quality. +Use this procedure for required role/reference material on all three hosts and +for optional study inputs. It does not authorize another worker or paid grader. + +## Freeze and read bounded material + +1. Inventory only the role contract and references authorized by the handoff. + Identify required sections with stable material IDs. Test-author inventory + comes only from its Spec/API/test handoff, never implementation context. + Preserve the source revision and permitted read scope. Do not load unrelated + documents just to create an inventory. +2. `scripts/routing_delivery.py` accepts JSON stdin operations `plan`, `emit` and + `assess`. Plan takes `materials: [{id, text}]`, `max_read_attempts` and + `max_read_ms`; derive these limits from the task's remaining authorized budget, + not a new retry allowance. It returns chunk IDs, byte lengths and SHA256 values. + Store a large manifest in task-owned temporary storage; do not print it through + an undersized tool output. The helper reads no files and launches no models. +3. Emit takes the same authorized `materials` and one `chunk_id`. It returns only + that identified chunk: at most 512 Unicode characters / 2048 UTF-8 bytes. Read + one chunk at a time with enough actual tool output budget, including the JSON + wrapper. A prose request for a larger output limit does not set the limit. +4. On Codex, set the actual shell tool's `max_output_tokens` and the orchestration + tool's own output cap. When functions.exec is present, use a literal first-line + pragma, e.g. `// @exec: {"max_output_tokens": 3000}`, plus the nested command's + `max_output_tokens: 3000` for a single chunk. Do not combine a whole reference + bundle into that one capped call. On Claude native/skill routes, use the actual + Read offset/limit or bounded shell output supported by that worker tool; names + and controls vary, so do not copy Codex parameters into Claude. +5. Inspect the visible result for omissions, truncation markers and the complete + identified content. Source-side hashes alone and a worker saying “read it all” + are insufficient. If trusted host output is unavailable, record `unknown`. + Do not relabel source-file bytes as the observed tool output. + +## Assess, recover and preserve limits + +Assess takes `packet: {schema_version: 1, host, worker_id, manifest, observations}`. +Each observation has `chunk_id`, zero-based `attempt`, the actual visible `text`, +`truncated` (boolean or null), `source` (`host_tool_output`, `worker_assertion`, +`unavailable`), local evidence reference (or null), and `elapsed_ms` (or null). +Use `host_tool_output` only for host-visible evidence you actually inspected. +These are supplied records: the helper cannot authenticate a fabricated source +label or prove worker comprehension. Inspect the underlying evidence honestly. + +The helper checks visible content/length/hash, duplicate attempts, truncation +flags/markers, omissions and cumulative read budgets. It returns complete, +incomplete, unknown or budget_exceeded, missing IDs and the recoverable subset. +Append only missing chunk reads within the remaining budget; retain all earlier +attempts. Unknown timing does not authorize budgeted recovery. A complete later +read can repair a truncated chunk, but does not erase consumed attempts/time. +A duplicate attempt is invalid input. Never automatically restart the worker. + +For ordinary work, repair required missing inputs within scope or report the +input-delivery failure. Unknown completeness must be reported as unknown; it +is not proof of complete delivery, and does not itself reinstate a study gate +for Dynamic routing. Existing tests/review still determine the work's outcome. +Keep these receipts in the permitted quest plan (or final fast-lane report). + +## Optional study integration + +New study state uses version 2; existing manifests/state remain readable and +original outcomes/grades stay unchanged. New live trials cannot be claimed without a frozen delivery manifest, and an +invalid prior live trial stops further claims before more model usage. Before claiming a trial, freeze the +expected manifest with `study_runner.py delivery_plan --directory … --run … +--input manifest.json`. All trials for one fixture must use the same manifest. +After actual reads, append the cumulative assess packet with the `delivery` +operation. It must match that manifest, host and worker; earlier observations +cannot be removed. Record the ordinary outcome and actual consumption normally. + +`blind` excludes incomplete/unknown live delivery before paying for grading. +`grade` refuses those trials. `export` returns `invalid_input_delivery` and no +comparison dataset if any trial is invalid, even if an old grade exists; it does +not cherry-pick a favorable subset or manufacture quality failures. Missing +legacy live delivery is unknown. Historical synthetic fixtures keep explicitly +synthetic arithmetic compatibility; this is not live delivery evidence. +Keep prior exported reports unchanged and annotate their limits separately. +No new study or regrading run is required for ordinary Dynamic activation. + +## Feedback from work already performed + +`scripts/routing_feedback.py` accepts the v5 request, its decision, a worker +outcome, observation and optional existing usage packet. Outcome fields are +worker_id, status (completed/failed/interrupted), tests +(passed/failed/unknown/not_applicable), review +(accepted/rejected/unknown/not_applicable), retries, local evidence references +and delivery status. Observation fields are source +(host_metadata/worker_assertion/unknown), evidence, requested_resolved and +observed model/effort pairs (nullable), and configuration_supported (nullable). + +Reuse trusted host metadata, actual tests/reviews/retries and complete usage from +that work. Compare resolved identities only when known; requested aliases are not +served names. Worker assertions never become observed identity. Known forced +substitution/unsupported configuration suspends future choices without resetting +counters or replaying work. Unknown identity alone does not suspend Dynamic mode. + +Optional usage input uses the existing routing_usage schema, correlated to this +worker and host/role/category. Cache/reasoning semantics and incomplete counts +remain intact. Raw tokens are a subscription proxy; account-wide percentages do +not establish per-task allowance or monetary cost. The result leaves subscription +allowance percent null, requests zero extra runs and proposes catalog review only. +It never trains, mutates a policy/catalog, gives qualification or writes files. +Do not schedule duplicate runs, independent grading or extra retries for feedback. diff --git a/plugin/portable/references/qualification-study.md b/plugin/portable/references/qualification-study.md index a4337c4..487e21d 100644 --- a/plugin/portable/references/qualification-study.md +++ b/plugin/portable/references/qualification-study.md @@ -1,5 +1,11 @@ # Run a bounded routing study +This advanced workflow is optional. Dynamic routing does not require a study. +Before a paid trial, follow [bounded delivery](input-delivery.md), freeze its +expected input manifest and verify actual visible chunks. Exclude incomplete or +unknown live delivery before grading or comparing model quality/efficiency. +A failed study does not authorize another run or prevent Dynamic activation. + Guildhall supplies a development headroom analyzer and a study controller. They run no models themselves. The controller prepares isolated Git worktrees and explicit dispatch packets for the actual host, then imports measured outcomes diff --git a/plugin/portable/references/quest.md b/plugin/portable/references/quest.md index 2284776..2151d5e 100644 --- a/plugin/portable/references/quest.md +++ b/plugin/portable/references/quest.md @@ -211,3 +211,9 @@ routing solely from a role's own assertion. Use [host preflight and evidence](host-evidence.md) for optional schema-v3 routing; retain worker scopes, gates, state and explicit activation. + + +For required reference material, use [bounded input delivery](input-delivery.md) +and record ordinary-work feedback from existing tests/reviews/usage only. Repair +missing chunks within the existing budget; never treat source hashes or worker +assertions as delivery proof or run extra workers to collect feedback. diff --git a/plugin/portable/references/routing.md b/plugin/portable/references/routing.md index a9dd9c6..2d3bb3b 100644 --- a/plugin/portable/references/routing.md +++ b/plugin/portable/references/routing.md @@ -313,3 +313,9 @@ or automatically promotes a profile. Follow the [role matrix and migration guide](role-eligibility.md). Eligibility never changes role contracts, test-author handoffs, review membership or lifecycle gates. Upgrading does not add roles to existing allowlists or qualify profiles. + + +For required reference material, use [bounded input delivery](input-delivery.md) +and record ordinary-work feedback from existing tests/reviews/usage only. Repair +missing chunks within the existing budget; never treat source hashes or worker +assertions as delivery proof or run extra workers to collect feedback. diff --git a/plugin/portable/scripts/routing_delivery.py b/plugin/portable/scripts/routing_delivery.py new file mode 100644 index 0000000..798135f --- /dev/null +++ b/plugin/portable/scripts/routing_delivery.py @@ -0,0 +1,109 @@ +#!/usr/bin/env python3 +"""Bounded handoff chunks and supplied host-visible delivery checks; no model calls.""" +from __future__ import annotations +import hashlib +import json +from pathlib import Path +import re +import runpy +import sys + +_R=runpy.run_path(str(Path(__file__).with_name('route_model.py')),run_name='_delivery_router') +obj,enum,array=_R['obj'],_R['enum'],_R['array'] +LIMIT=2*1024*1024 +MATERIAL=obj(id=_R['CID'],text=dict(type='string',maxLength=262144)) +CHUNK=obj(id=dict(type='string',pattern=r'^[a-z][a-z0-9_-]{0,31}:[0-9]+$',maxLength=40), + bytes=dict(type='integer',minimum=0,maximum=4096),sha256=_R['HASH']) +MANIFEST=obj(schema_version=enum([1]),chunks=array(CHUNK,1024,1), + max_read_attempts=dict(type='integer',minimum=1,maximum=4096), + max_read_ms=dict(type='integer',minimum=1,maximum=3600000)) +OBSERVATION=obj(chunk_id=CHUNK['properties']['id'],attempt=dict(type='integer',minimum=0), + text=dict(type='string',maxLength=8192),truncated=_R['nullable'](_R['BOOL']), + source=enum(['host_tool_output','worker_assertion','unavailable']),evidence=_R['nullable'](_R['STRING']), + elapsed_ms=_R['nullable'](_R['NUMBER'])) +PACKET=obj(schema_version=enum([1]),host=enum(_R['ROUTES']),worker_id=_R['STRING'], + manifest=MANIFEST,observations=array(OBSERVATION,4096)) +TRUNCATED=re.compile(r'(?:output (?:was |is )?truncated|truncated output|\[\.\.\.\s*truncated\s*\.\.\.\]|omitted \d+ lines)',re.I) + + +def need(value): + if not value:raise ValueError('invalid_delivery_input') + + +def chunks(materials): + _R['validate'](materials,array(MATERIAL,64,1)) + need(len({m['id'] for m in materials})==len(materials)) + need(sum(len(m['text'].encode()) for m in materials)<=LIMIT//2) + result=[] + for material in materials: + # At most 512 Unicode characters / 2048 UTF-8 bytes per visible read. + pieces=[material['text'][n:n+512] for n in range(0,len(material['text']),512)] or [''] + result.extend(dict(id=f"{material['id']}:{i}",text=text) for i,text in enumerate(pieces)) + need(len(result)<=1024) + return result + + +def plan(materials,max_read_attempts,max_read_ms): + result=dict(schema_version=1,chunks=[dict(id=c['id'],bytes=len(c['text'].encode()), + sha256=hashlib.sha256(c['text'].encode()).hexdigest()) for c in chunks(materials)], + max_read_attempts=max_read_attempts,max_read_ms=max_read_ms) + _R['validate'](result,MANIFEST) + need(max_read_attempts>=len(result['chunks'])) + return result + + +def emit(materials,chunk_id): + result=next((c for c in chunks(materials) if c['id']==chunk_id),None) + need(result is not None) + return result + + +def assess(packet): + _R['validate'](packet,PACKET) + manifest=packet['manifest'];expected={c['id']:c for c in manifest['chunks']} + need(len(expected)==len(manifest['chunks'])) + observations=packet['observations'];seen=set();complete=set();failed=set();unknown=set() + elapsed=0;timing_known=True + for o in observations: + cid=o['chunk_id'];identity=(cid,o['attempt']) + need(cid in expected and identity not in seen);seen.add(identity) + if o['elapsed_ms'] is None:timing_known=False + else:elapsed+=o['elapsed_ms'] + if o['source']!='host_tool_output' or o['truncated'] is None or o['evidence'] is None: + unknown.add(cid);continue + raw=o['text'].encode() + if (o['truncated'] or TRUNCATED.search(o['text']) or len(raw)!=expected[cid]['bytes'] or + hashlib.sha256(raw).hexdigest()!=expected[cid]['sha256']): + failed.add(cid) + else:complete.add(cid) + missing=sorted(set(expected)-complete) + exhausted=len(observations)>manifest['max_read_attempts'] or elapsed>manifest['max_read_ms'] + status=('budget_exceeded' if exhausted else 'complete' if not missing else + 'incomplete' if failed-set(complete) or set(missing)-unknown else 'unknown') + remaining=max(0,manifest['max_read_attempts']-len(observations)) + return dict(schema_version=1,host=packet['host'],worker_id=packet['worker_id'], + status=status,comparison_eligible=status=='complete',missing_chunks=missing, + manifest_hash=_R['policy_hash'](manifest),evidence_hash=_R['policy_hash'](packet), + attempts_used=len(observations),attempts_remaining=remaining,read_ms=elapsed if timing_known else None, + recoverable_chunks=missing[:remaining] if not exhausted and timing_known and elapsed=len(manifest['chunks'])) + need(len({c['id'] for c in manifest['chunks']})==len(manifest['chunks'])) + need(manifest['max_read_ms']<=state['manifest']['timeout_seconds']*1000) + peers=[r['delivery_manifest'] for r in state['runs'] if r['fixture_id']==run['fixture_id'] and 'delivery_manifest' in r] + need(all(p==manifest for p in peers)) + run['delivery_manifest']=manifest;save(root,state) + return dict(run_id=run_id,manifest_hash=digest(manifest),dispatch=False) + + +def delivery(directory,run_id,packet): + """Append visible-read evidence; never erase earlier attempts or reset budgets.""" + assessment=_D['assess'](packet) + with locked(directory) as (root,state): + run,_=selected(state,run_id) + need(run['status']=='running' and run.get('delivery_manifest')==packet['manifest']) + need(packet['host']==state['manifest']['host']['route']) + hashes=[digest(o) for o in packet['observations']] + prior=run.get('delivery_observations',[]) + need(hashes[:len(prior)]==prior) + need('delivery' not in run or run['delivery']['worker_id']==packet['worker_id']) + run['delivery']=assessment;run['delivery_observations']=hashes;save(root,state) + return assessment + + +def delivery_status(state,run): + assessment=run.get('delivery') + if assessment is not None: + return assessment['status'] + # Historical synthetic fixtures retain arithmetic coverage, not live proof. + return 'synthetic_unchecked' if state['manifest']['synthetic'] else 'unknown' + + +def comparable(state,run): + return delivery_status(state,run) in ('complete','synthetic_unchecked') + + def claim(directory,run_id): with locked(directory) as (root,state): run,fixture=selected(state,run_id) need(run['status']=='pending' and run['candidate'] is not None) need(not any(r['status']=='running' for r in state['runs'])) + if not state['manifest']['synthetic']: + need(state.get('schema_version',1)==1 or 'delivery_manifest' in run) + need(all(comparable(state,r) for r in state['runs'] if r['status']=='recorded')) recorded=[r['outcome'] for r in state['runs'] if r['status']=='recorded'] need(all(r['usage_tokens'] is not None for r in recorded)) need(state['prior_usage_tokens']+sum(r['usage_tokens'] for r in recorded)') known=state['prior_usage_tokens']+sum(r['outcome']['usage_tokens'] or 0 for r in state['runs'] if r['status']=='recorded') if known+(outcome['usage_tokens'] or 0)>state['manifest']['usage_budget_tokens']:violations.append('') + if 'delivery' in run:need(run['delivery']['worker_id']==outcome['worker_id']) run.update(status='recorded',outcome=outcome,tree_hash=tree_hash,violations=violations) - save(root,state);return dict(run_id=run_id,violations=violations,recorded=True) + save(root,state);return dict(run_id=run_id,violations=violations,recorded=True, + input_delivery=delivery_status(state,run),comparison_eligible=comparable(state,run)) def blind(directory): with locked(directory) as (root,state): - packets=[] + packets=[];excluded=[] for run in state['runs']: if run['status']!='recorded':continue + if not comparable(state,run): + excluded.append(dict(run_id=run['id'],reason='input_delivery_'+delivery_status(state,run)));continue _,fixture=selected(state,run['id']) _,snapshot_hash,artifacts=tree_evidence(verify_tree(root,state,run),include_artifacts=True) need(snapshot_hash==run['tree_hash']) @@ -226,7 +277,7 @@ def blind(directory): text=json.dumps(packet).lower() need(all(c['model'].lower() not in text for c in state['manifest']['candidates'])) packets.append(packet) - return dict(schema_version=1,packets=packets,instruction='Grade accepted/critical_misses independently; do not inspect study.json, worktree names or strategy metadata.') + return dict(schema_version=2,packets=packets,excluded=excluded,instruction='Grade accepted/critical_misses independently; do not inspect study.json, worktree names or strategy metadata.') def grade(directory,grades): @@ -237,7 +288,7 @@ def grade(directory,grades): shape(item,'blind_id accepted critical_misses reason') need(type(item['accepted']) is bool);_S['number'](item['critical_misses'],True);_S['string'](item['reason']) need(item['blind_id'] not in seen);seen.add(item['blind_id']) - run,_=selected(state,item['blind_id']);need(run['status']=='recorded' and 'grade' not in run) + run,_=selected(state,item['blind_id']);need(run['status']=='recorded' and 'grade' not in run and comparable(state,run)) need(tree_evidence(verify_tree(root,state,run))[1]==run['tree_hash']) run['grade']=item save(root,state);return dict(frozen_grades=len(grades),qualification=False) @@ -245,7 +296,15 @@ def grade(directory,grades): def export(directory): with locked(directory) as (root,state): - need(all(r['status']=='recorded' and 'grade' in r for r in state['runs'])) + need(all(r['status']=='recorded' for r in state['runs'])) + invalid=[dict(run_id=r['id'],reason='input_delivery_'+delivery_status(state,r)) for r in state['runs'] if not comparable(state,r)] + if invalid: + # Do not export a filtered, deceptively favorable comparison or regrade + # old outputs. Original outcomes/usage/grades stay unchanged on disk. + return dict(schema_version=2,status='invalid_input_delivery',qualification=False, + manifest_hash=state['manifest_hash'],invalid_trials=invalid, + reason='No model comparison exported; preserve original outcomes and consumed budget.') + need(all('grade' in r for r in state['runs'])) records=[] for run in state['runs']: need(tree_evidence(verify_tree(root,state,run))[1]==run['tree_hash']) @@ -266,7 +325,7 @@ def export(directory): def main(): parser=argparse.ArgumentParser(description=__doc__) - parser.add_argument('operation',choices=['prepare','select','claim','record','blind','grade','export']) + parser.add_argument('operation',choices=['prepare','select','claim','delivery_plan','delivery','record','blind','grade','export']) parser.add_argument('--directory',required=True,type=Path) parser.add_argument('--manifest',type=Path);parser.add_argument('--repo',type=Path) parser.add_argument('--phase',default='development',choices=['development','holdout']) diff --git a/plugin/routing-setup/references/setup-flow.md b/plugin/routing-setup/references/setup-flow.md index 308fed1..bd1c2af 100644 --- a/plugin/routing-setup/references/setup-flow.md +++ b/plugin/routing-setup/references/setup-flow.md @@ -198,3 +198,9 @@ configuration and approval only; it never resets an active quest's counters. qualification, corrupt files, unsafe permissions or an existing override. Do not silently overwrite malformed files, relax permissions or clear locks. Suggest the smallest concrete repair, then apply only the requested repair. + + +For a supplied study failure involving truncated references, read +[input delivery](input-delivery.md) and inspect only the supplied report. Preserve +its original outcomes. Offer Dynamic routing based on supported controls; do not +require regrading, extra trials or fabricated delivery evidence to enable it. diff --git a/plugin/skills/guildhall-quest/references/global-routing.md b/plugin/skills/guildhall-quest/references/global-routing.md index 2ecc50e..9e4f0d1 100644 --- a/plugin/skills/guildhall-quest/references/global-routing.md +++ b/plugin/skills/guildhall-quest/references/global-routing.md @@ -63,8 +63,8 @@ Use an ordinary setup conversation outside quest execution. For example: > Configure Guildhall specialist model routing globally for this host. Read the > installed global and model-routing guides. Discover supported model/effort > settings without paid probes. Prepare an off global policy, keep unknown metrics -> and qualifications unknown, and show the proposed shadow policy, all-project -> scope, objective, candidates, budgets, outbound fields and host evidence. +> and qualifications unknown, and show the proposed Dynamic policy, all-project +> scope, objective, candidates, budgets, outbound fields and worker controls. > Activate only after I explicitly approve that exact proposal. Retain any > existing project policy and explain which source currently wins. @@ -83,16 +83,17 @@ different source do not copy consent. Credentials stay in the host environment, using the policy's `key_env`; never store the secret value in either file. Each approval binds the effective canonical policy hash, source and host route, -current host configuration fingerprint, and reviewed evidence hashes. Changing +current host configuration fingerprint, and either v5 reviewed control facts or +legacy/qualified reviewed evidence hashes. Changing another host's global entry or JSON formatting does not invalidate it. Changing the selected policy or host configuration requires renewed activation. Expiry, -revocation, missing reviewed host evidence or corrupt approval state prevents +revocation, missing required control/evidence review or corrupt approval state prevents reuse. Approval expiry may be omitted; qualification still has its own expiry. When reviewed host evidence has an expiry, do not approve beyond that expiry. The helper fingerprints the supplied host request fields (sorting supported settings), excluding the evidence hash itself; that hash must separately appear -in approved evidence. Supply fresh, truthful host metadata each time, not a stale +in approved evidence. V5 uses the control fingerprint described below instead. Supply fresh, truthful host metadata each time, not a stale snapshot to keep approval working. Host/role qualification checks still happen in the routing engine. `ready` means reusable consent, not adaptive qualification or proof of which model executed. @@ -165,7 +166,8 @@ recognize a valid opt-out or valid off policy, and skip both Python helpers. Check global defaults when the project file is absent. If selected configuration cannot be validated, stop configuration resolution; never guess that it is off. -For enabled routing, call `status` with current host evidence. Use its `policy` +For enabled routing, call `status` with current host controls/evidence shaped to +the selected request version. V5 dynamic requires no execution-evidence hashes. Use its `policy` and `activation` verbatim in the existing routing request, setting request schema version to the selected policy's schema version. Add a summary hash only after the exact summary's approval. Non-ready results do not authorize external calls; diff --git a/plugin/skills/guildhall-quest/references/hosts.md b/plugin/skills/guildhall-quest/references/hosts.md index 984921f..9a5c8d6 100644 --- a/plugin/skills/guildhall-quest/references/hosts.md +++ b/plugin/skills/guildhall-quest/references/hosts.md @@ -105,3 +105,9 @@ worker and bundled role body without assuming native registration or hooks. Codex uses actual exposed model/effort arguments and a fresh independent context; missing served identity alone is not a reason to demand a study. Never copy effort names between hosts. Reuse the same control approval across role fallback choices. + + +For required reference material, use [bounded input delivery](input-delivery.md) +and record ordinary-work feedback from existing tests/reviews/usage only. Repair +missing chunks within the existing budget; never treat source hashes or worker +assertions as delivery proof or run extra workers to collect feedback. diff --git a/plugin/skills/guildhall-quest/references/input-delivery.md b/plugin/skills/guildhall-quest/references/input-delivery.md new file mode 100644 index 0000000..6e94ef5 --- /dev/null +++ b/plugin/skills/guildhall-quest/references/input-delivery.md @@ -0,0 +1,104 @@ +# Bounded input delivery and ordinary-work feedback + +A correct source file and its hash do not prove that a worker received the whole +file. Treat delivery as a separate fact from model identity and output quality. +Use this procedure for required role/reference material on all three hosts and +for optional study inputs. It does not authorize another worker or paid grader. + +## Freeze and read bounded material + +1. Inventory only the role contract and references authorized by the handoff. + Identify required sections with stable material IDs. Test-author inventory + comes only from its Spec/API/test handoff, never implementation context. + Preserve the source revision and permitted read scope. Do not load unrelated + documents just to create an inventory. +2. `scripts/routing_delivery.py` accepts JSON stdin operations `plan`, `emit` and + `assess`. Plan takes `materials: [{id, text}]`, `max_read_attempts` and + `max_read_ms`; derive these limits from the task's remaining authorized budget, + not a new retry allowance. It returns chunk IDs, byte lengths and SHA256 values. + Store a large manifest in task-owned temporary storage; do not print it through + an undersized tool output. The helper reads no files and launches no models. +3. Emit takes the same authorized `materials` and one `chunk_id`. It returns only + that identified chunk: at most 512 Unicode characters / 2048 UTF-8 bytes. Read + one chunk at a time with enough actual tool output budget, including the JSON + wrapper. A prose request for a larger output limit does not set the limit. +4. On Codex, set the actual shell tool's `max_output_tokens` and the orchestration + tool's own output cap. When functions.exec is present, use a literal first-line + pragma, e.g. `// @exec: {"max_output_tokens": 3000}`, plus the nested command's + `max_output_tokens: 3000` for a single chunk. Do not combine a whole reference + bundle into that one capped call. On Claude native/skill routes, use the actual + Read offset/limit or bounded shell output supported by that worker tool; names + and controls vary, so do not copy Codex parameters into Claude. +5. Inspect the visible result for omissions, truncation markers and the complete + identified content. Source-side hashes alone and a worker saying “read it all” + are insufficient. If trusted host output is unavailable, record `unknown`. + Do not relabel source-file bytes as the observed tool output. + +## Assess, recover and preserve limits + +Assess takes `packet: {schema_version: 1, host, worker_id, manifest, observations}`. +Each observation has `chunk_id`, zero-based `attempt`, the actual visible `text`, +`truncated` (boolean or null), `source` (`host_tool_output`, `worker_assertion`, +`unavailable`), local evidence reference (or null), and `elapsed_ms` (or null). +Use `host_tool_output` only for host-visible evidence you actually inspected. +These are supplied records: the helper cannot authenticate a fabricated source +label or prove worker comprehension. Inspect the underlying evidence honestly. + +The helper checks visible content/length/hash, duplicate attempts, truncation +flags/markers, omissions and cumulative read budgets. It returns complete, +incomplete, unknown or budget_exceeded, missing IDs and the recoverable subset. +Append only missing chunk reads within the remaining budget; retain all earlier +attempts. Unknown timing does not authorize budgeted recovery. A complete later +read can repair a truncated chunk, but does not erase consumed attempts/time. +A duplicate attempt is invalid input. Never automatically restart the worker. + +For ordinary work, repair required missing inputs within scope or report the +input-delivery failure. Unknown completeness must be reported as unknown; it +is not proof of complete delivery, and does not itself reinstate a study gate +for Dynamic routing. Existing tests/review still determine the work's outcome. +Keep these receipts in the permitted quest plan (or final fast-lane report). + +## Optional study integration + +New study state uses version 2; existing manifests/state remain readable and +original outcomes/grades stay unchanged. New live trials cannot be claimed without a frozen delivery manifest, and an +invalid prior live trial stops further claims before more model usage. Before claiming a trial, freeze the +expected manifest with `study_runner.py delivery_plan --directory … --run … +--input manifest.json`. All trials for one fixture must use the same manifest. +After actual reads, append the cumulative assess packet with the `delivery` +operation. It must match that manifest, host and worker; earlier observations +cannot be removed. Record the ordinary outcome and actual consumption normally. + +`blind` excludes incomplete/unknown live delivery before paying for grading. +`grade` refuses those trials. `export` returns `invalid_input_delivery` and no +comparison dataset if any trial is invalid, even if an old grade exists; it does +not cherry-pick a favorable subset or manufacture quality failures. Missing +legacy live delivery is unknown. Historical synthetic fixtures keep explicitly +synthetic arithmetic compatibility; this is not live delivery evidence. +Keep prior exported reports unchanged and annotate their limits separately. +No new study or regrading run is required for ordinary Dynamic activation. + +## Feedback from work already performed + +`scripts/routing_feedback.py` accepts the v5 request, its decision, a worker +outcome, observation and optional existing usage packet. Outcome fields are +worker_id, status (completed/failed/interrupted), tests +(passed/failed/unknown/not_applicable), review +(accepted/rejected/unknown/not_applicable), retries, local evidence references +and delivery status. Observation fields are source +(host_metadata/worker_assertion/unknown), evidence, requested_resolved and +observed model/effort pairs (nullable), and configuration_supported (nullable). + +Reuse trusted host metadata, actual tests/reviews/retries and complete usage from +that work. Compare resolved identities only when known; requested aliases are not +served names. Worker assertions never become observed identity. Known forced +substitution/unsupported configuration suspends future choices without resetting +counters or replaying work. Unknown identity alone does not suspend Dynamic mode. + +Optional usage input uses the existing routing_usage schema, correlated to this +worker and host/role/category. Cache/reasoning semantics and incomplete counts +remain intact. Raw tokens are a subscription proxy; account-wide percentages do +not establish per-task allowance or monetary cost. The result leaves subscription +allowance percent null, requests zero extra runs and proposes catalog review only. +It never trains, mutates a policy/catalog, gives qualification or writes files. +Do not schedule duplicate runs, independent grading or extra retries for feedback. diff --git a/plugin/skills/guildhall-quest/references/qualification-study.md b/plugin/skills/guildhall-quest/references/qualification-study.md index a4337c4..487e21d 100644 --- a/plugin/skills/guildhall-quest/references/qualification-study.md +++ b/plugin/skills/guildhall-quest/references/qualification-study.md @@ -1,5 +1,11 @@ # Run a bounded routing study +This advanced workflow is optional. Dynamic routing does not require a study. +Before a paid trial, follow [bounded delivery](input-delivery.md), freeze its +expected input manifest and verify actual visible chunks. Exclude incomplete or +unknown live delivery before grading or comparing model quality/efficiency. +A failed study does not authorize another run or prevent Dynamic activation. + Guildhall supplies a development headroom analyzer and a study controller. They run no models themselves. The controller prepares isolated Git worktrees and explicit dispatch packets for the actual host, then imports measured outcomes diff --git a/plugin/skills/guildhall-quest/references/quest.md b/plugin/skills/guildhall-quest/references/quest.md index 2284776..2151d5e 100644 --- a/plugin/skills/guildhall-quest/references/quest.md +++ b/plugin/skills/guildhall-quest/references/quest.md @@ -211,3 +211,9 @@ routing solely from a role's own assertion. Use [host preflight and evidence](host-evidence.md) for optional schema-v3 routing; retain worker scopes, gates, state and explicit activation. + + +For required reference material, use [bounded input delivery](input-delivery.md) +and record ordinary-work feedback from existing tests/reviews/usage only. Repair +missing chunks within the existing budget; never treat source hashes or worker +assertions as delivery proof or run extra workers to collect feedback. diff --git a/plugin/skills/guildhall-quest/references/routing.md b/plugin/skills/guildhall-quest/references/routing.md index a9dd9c6..2d3bb3b 100644 --- a/plugin/skills/guildhall-quest/references/routing.md +++ b/plugin/skills/guildhall-quest/references/routing.md @@ -313,3 +313,9 @@ or automatically promotes a profile. Follow the [role matrix and migration guide](role-eligibility.md). Eligibility never changes role contracts, test-author handoffs, review membership or lifecycle gates. Upgrading does not add roles to existing allowlists or qualify profiles. + + +For required reference material, use [bounded input delivery](input-delivery.md) +and record ordinary-work feedback from existing tests/reviews/usage only. Repair +missing chunks within the existing budget; never treat source hashes or worker +assertions as delivery proof or run extra workers to collect feedback. diff --git a/plugin/skills/guildhall-quest/scripts/routing_delivery.py b/plugin/skills/guildhall-quest/scripts/routing_delivery.py new file mode 100644 index 0000000..798135f --- /dev/null +++ b/plugin/skills/guildhall-quest/scripts/routing_delivery.py @@ -0,0 +1,109 @@ +#!/usr/bin/env python3 +"""Bounded handoff chunks and supplied host-visible delivery checks; no model calls.""" +from __future__ import annotations +import hashlib +import json +from pathlib import Path +import re +import runpy +import sys + +_R=runpy.run_path(str(Path(__file__).with_name('route_model.py')),run_name='_delivery_router') +obj,enum,array=_R['obj'],_R['enum'],_R['array'] +LIMIT=2*1024*1024 +MATERIAL=obj(id=_R['CID'],text=dict(type='string',maxLength=262144)) +CHUNK=obj(id=dict(type='string',pattern=r'^[a-z][a-z0-9_-]{0,31}:[0-9]+$',maxLength=40), + bytes=dict(type='integer',minimum=0,maximum=4096),sha256=_R['HASH']) +MANIFEST=obj(schema_version=enum([1]),chunks=array(CHUNK,1024,1), + max_read_attempts=dict(type='integer',minimum=1,maximum=4096), + max_read_ms=dict(type='integer',minimum=1,maximum=3600000)) +OBSERVATION=obj(chunk_id=CHUNK['properties']['id'],attempt=dict(type='integer',minimum=0), + text=dict(type='string',maxLength=8192),truncated=_R['nullable'](_R['BOOL']), + source=enum(['host_tool_output','worker_assertion','unavailable']),evidence=_R['nullable'](_R['STRING']), + elapsed_ms=_R['nullable'](_R['NUMBER'])) +PACKET=obj(schema_version=enum([1]),host=enum(_R['ROUTES']),worker_id=_R['STRING'], + manifest=MANIFEST,observations=array(OBSERVATION,4096)) +TRUNCATED=re.compile(r'(?:output (?:was |is )?truncated|truncated output|\[\.\.\.\s*truncated\s*\.\.\.\]|omitted \d+ lines)',re.I) + + +def need(value): + if not value:raise ValueError('invalid_delivery_input') + + +def chunks(materials): + _R['validate'](materials,array(MATERIAL,64,1)) + need(len({m['id'] for m in materials})==len(materials)) + need(sum(len(m['text'].encode()) for m in materials)<=LIMIT//2) + result=[] + for material in materials: + # At most 512 Unicode characters / 2048 UTF-8 bytes per visible read. + pieces=[material['text'][n:n+512] for n in range(0,len(material['text']),512)] or [''] + result.extend(dict(id=f"{material['id']}:{i}",text=text) for i,text in enumerate(pieces)) + need(len(result)<=1024) + return result + + +def plan(materials,max_read_attempts,max_read_ms): + result=dict(schema_version=1,chunks=[dict(id=c['id'],bytes=len(c['text'].encode()), + sha256=hashlib.sha256(c['text'].encode()).hexdigest()) for c in chunks(materials)], + max_read_attempts=max_read_attempts,max_read_ms=max_read_ms) + _R['validate'](result,MANIFEST) + need(max_read_attempts>=len(result['chunks'])) + return result + + +def emit(materials,chunk_id): + result=next((c for c in chunks(materials) if c['id']==chunk_id),None) + need(result is not None) + return result + + +def assess(packet): + _R['validate'](packet,PACKET) + manifest=packet['manifest'];expected={c['id']:c for c in manifest['chunks']} + need(len(expected)==len(manifest['chunks'])) + observations=packet['observations'];seen=set();complete=set();failed=set();unknown=set() + elapsed=0;timing_known=True + for o in observations: + cid=o['chunk_id'];identity=(cid,o['attempt']) + need(cid in expected and identity not in seen);seen.add(identity) + if o['elapsed_ms'] is None:timing_known=False + else:elapsed+=o['elapsed_ms'] + if o['source']!='host_tool_output' or o['truncated'] is None or o['evidence'] is None: + unknown.add(cid);continue + raw=o['text'].encode() + if (o['truncated'] or TRUNCATED.search(o['text']) or len(raw)!=expected[cid]['bytes'] or + hashlib.sha256(raw).hexdigest()!=expected[cid]['sha256']): + failed.add(cid) + else:complete.add(cid) + missing=sorted(set(expected)-complete) + exhausted=len(observations)>manifest['max_read_attempts'] or elapsed>manifest['max_read_ms'] + status=('budget_exceeded' if exhausted else 'complete' if not missing else + 'incomplete' if failed-set(complete) or set(missing)-unknown else 'unknown') + remaining=max(0,manifest['max_read_attempts']-len(observations)) + return dict(schema_version=1,host=packet['host'],worker_id=packet['worker_id'], + status=status,comparison_eligible=status=='complete',missing_chunks=missing, + manifest_hash=_R['policy_hash'](manifest),evidence_hash=_R['policy_hash'](packet), + attempts_used=len(observations),attempts_remaining=remaining,read_ms=elapsed if timing_known else None, + recoverable_chunks=missing[:remaining] if not exhausted and timing_known and elapsed=len(manifest['chunks'])) + need(len({c['id'] for c in manifest['chunks']})==len(manifest['chunks'])) + need(manifest['max_read_ms']<=state['manifest']['timeout_seconds']*1000) + peers=[r['delivery_manifest'] for r in state['runs'] if r['fixture_id']==run['fixture_id'] and 'delivery_manifest' in r] + need(all(p==manifest for p in peers)) + run['delivery_manifest']=manifest;save(root,state) + return dict(run_id=run_id,manifest_hash=digest(manifest),dispatch=False) + + +def delivery(directory,run_id,packet): + """Append visible-read evidence; never erase earlier attempts or reset budgets.""" + assessment=_D['assess'](packet) + with locked(directory) as (root,state): + run,_=selected(state,run_id) + need(run['status']=='running' and run.get('delivery_manifest')==packet['manifest']) + need(packet['host']==state['manifest']['host']['route']) + hashes=[digest(o) for o in packet['observations']] + prior=run.get('delivery_observations',[]) + need(hashes[:len(prior)]==prior) + need('delivery' not in run or run['delivery']['worker_id']==packet['worker_id']) + run['delivery']=assessment;run['delivery_observations']=hashes;save(root,state) + return assessment + + +def delivery_status(state,run): + assessment=run.get('delivery') + if assessment is not None: + return assessment['status'] + # Historical synthetic fixtures retain arithmetic coverage, not live proof. + return 'synthetic_unchecked' if state['manifest']['synthetic'] else 'unknown' + + +def comparable(state,run): + return delivery_status(state,run) in ('complete','synthetic_unchecked') + + def claim(directory,run_id): with locked(directory) as (root,state): run,fixture=selected(state,run_id) need(run['status']=='pending' and run['candidate'] is not None) need(not any(r['status']=='running' for r in state['runs'])) + if not state['manifest']['synthetic']: + need(state.get('schema_version',1)==1 or 'delivery_manifest' in run) + need(all(comparable(state,r) for r in state['runs'] if r['status']=='recorded')) recorded=[r['outcome'] for r in state['runs'] if r['status']=='recorded'] need(all(r['usage_tokens'] is not None for r in recorded)) need(state['prior_usage_tokens']+sum(r['usage_tokens'] for r in recorded)') known=state['prior_usage_tokens']+sum(r['outcome']['usage_tokens'] or 0 for r in state['runs'] if r['status']=='recorded') if known+(outcome['usage_tokens'] or 0)>state['manifest']['usage_budget_tokens']:violations.append('') + if 'delivery' in run:need(run['delivery']['worker_id']==outcome['worker_id']) run.update(status='recorded',outcome=outcome,tree_hash=tree_hash,violations=violations) - save(root,state);return dict(run_id=run_id,violations=violations,recorded=True) + save(root,state);return dict(run_id=run_id,violations=violations,recorded=True, + input_delivery=delivery_status(state,run),comparison_eligible=comparable(state,run)) def blind(directory): with locked(directory) as (root,state): - packets=[] + packets=[];excluded=[] for run in state['runs']: if run['status']!='recorded':continue + if not comparable(state,run): + excluded.append(dict(run_id=run['id'],reason='input_delivery_'+delivery_status(state,run)));continue _,fixture=selected(state,run['id']) _,snapshot_hash,artifacts=tree_evidence(verify_tree(root,state,run),include_artifacts=True) need(snapshot_hash==run['tree_hash']) @@ -226,7 +277,7 @@ def blind(directory): text=json.dumps(packet).lower() need(all(c['model'].lower() not in text for c in state['manifest']['candidates'])) packets.append(packet) - return dict(schema_version=1,packets=packets,instruction='Grade accepted/critical_misses independently; do not inspect study.json, worktree names or strategy metadata.') + return dict(schema_version=2,packets=packets,excluded=excluded,instruction='Grade accepted/critical_misses independently; do not inspect study.json, worktree names or strategy metadata.') def grade(directory,grades): @@ -237,7 +288,7 @@ def grade(directory,grades): shape(item,'blind_id accepted critical_misses reason') need(type(item['accepted']) is bool);_S['number'](item['critical_misses'],True);_S['string'](item['reason']) need(item['blind_id'] not in seen);seen.add(item['blind_id']) - run,_=selected(state,item['blind_id']);need(run['status']=='recorded' and 'grade' not in run) + run,_=selected(state,item['blind_id']);need(run['status']=='recorded' and 'grade' not in run and comparable(state,run)) need(tree_evidence(verify_tree(root,state,run))[1]==run['tree_hash']) run['grade']=item save(root,state);return dict(frozen_grades=len(grades),qualification=False) @@ -245,7 +296,15 @@ def grade(directory,grades): def export(directory): with locked(directory) as (root,state): - need(all(r['status']=='recorded' and 'grade' in r for r in state['runs'])) + need(all(r['status']=='recorded' for r in state['runs'])) + invalid=[dict(run_id=r['id'],reason='input_delivery_'+delivery_status(state,r)) for r in state['runs'] if not comparable(state,r)] + if invalid: + # Do not export a filtered, deceptively favorable comparison or regrade + # old outputs. Original outcomes/usage/grades stay unchanged on disk. + return dict(schema_version=2,status='invalid_input_delivery',qualification=False, + manifest_hash=state['manifest_hash'],invalid_trials=invalid, + reason='No model comparison exported; preserve original outcomes and consumed budget.') + need(all('grade' in r for r in state['runs'])) records=[] for run in state['runs']: need(tree_evidence(verify_tree(root,state,run))[1]==run['tree_hash']) @@ -266,7 +325,7 @@ def export(directory): def main(): parser=argparse.ArgumentParser(description=__doc__) - parser.add_argument('operation',choices=['prepare','select','claim','record','blind','grade','export']) + parser.add_argument('operation',choices=['prepare','select','claim','delivery_plan','delivery','record','blind','grade','export']) parser.add_argument('--directory',required=True,type=Path) parser.add_argument('--manifest',type=Path);parser.add_argument('--repo',type=Path) parser.add_argument('--phase',default='development',choices=['development','holdout']) diff --git a/plugin/skills/guildhall-routing-setup/references/global-routing.md b/plugin/skills/guildhall-routing-setup/references/global-routing.md index 2ecc50e..9e4f0d1 100644 --- a/plugin/skills/guildhall-routing-setup/references/global-routing.md +++ b/plugin/skills/guildhall-routing-setup/references/global-routing.md @@ -63,8 +63,8 @@ Use an ordinary setup conversation outside quest execution. For example: > Configure Guildhall specialist model routing globally for this host. Read the > installed global and model-routing guides. Discover supported model/effort > settings without paid probes. Prepare an off global policy, keep unknown metrics -> and qualifications unknown, and show the proposed shadow policy, all-project -> scope, objective, candidates, budgets, outbound fields and host evidence. +> and qualifications unknown, and show the proposed Dynamic policy, all-project +> scope, objective, candidates, budgets, outbound fields and worker controls. > Activate only after I explicitly approve that exact proposal. Retain any > existing project policy and explain which source currently wins. @@ -83,16 +83,17 @@ different source do not copy consent. Credentials stay in the host environment, using the policy's `key_env`; never store the secret value in either file. Each approval binds the effective canonical policy hash, source and host route, -current host configuration fingerprint, and reviewed evidence hashes. Changing +current host configuration fingerprint, and either v5 reviewed control facts or +legacy/qualified reviewed evidence hashes. Changing another host's global entry or JSON formatting does not invalidate it. Changing the selected policy or host configuration requires renewed activation. Expiry, -revocation, missing reviewed host evidence or corrupt approval state prevents +revocation, missing required control/evidence review or corrupt approval state prevents reuse. Approval expiry may be omitted; qualification still has its own expiry. When reviewed host evidence has an expiry, do not approve beyond that expiry. The helper fingerprints the supplied host request fields (sorting supported settings), excluding the evidence hash itself; that hash must separately appear -in approved evidence. Supply fresh, truthful host metadata each time, not a stale +in approved evidence. V5 uses the control fingerprint described below instead. Supply fresh, truthful host metadata each time, not a stale snapshot to keep approval working. Host/role qualification checks still happen in the routing engine. `ready` means reusable consent, not adaptive qualification or proof of which model executed. @@ -165,7 +166,8 @@ recognize a valid opt-out or valid off policy, and skip both Python helpers. Check global defaults when the project file is absent. If selected configuration cannot be validated, stop configuration resolution; never guess that it is off. -For enabled routing, call `status` with current host evidence. Use its `policy` +For enabled routing, call `status` with current host controls/evidence shaped to +the selected request version. V5 dynamic requires no execution-evidence hashes. Use its `policy` and `activation` verbatim in the existing routing request, setting request schema version to the selected policy's schema version. Add a summary hash only after the exact summary's approval. Non-ready results do not authorize external calls; diff --git a/plugin/skills/guildhall-routing-setup/references/input-delivery.md b/plugin/skills/guildhall-routing-setup/references/input-delivery.md new file mode 100644 index 0000000..6e94ef5 --- /dev/null +++ b/plugin/skills/guildhall-routing-setup/references/input-delivery.md @@ -0,0 +1,104 @@ +# Bounded input delivery and ordinary-work feedback + +A correct source file and its hash do not prove that a worker received the whole +file. Treat delivery as a separate fact from model identity and output quality. +Use this procedure for required role/reference material on all three hosts and +for optional study inputs. It does not authorize another worker or paid grader. + +## Freeze and read bounded material + +1. Inventory only the role contract and references authorized by the handoff. + Identify required sections with stable material IDs. Test-author inventory + comes only from its Spec/API/test handoff, never implementation context. + Preserve the source revision and permitted read scope. Do not load unrelated + documents just to create an inventory. +2. `scripts/routing_delivery.py` accepts JSON stdin operations `plan`, `emit` and + `assess`. Plan takes `materials: [{id, text}]`, `max_read_attempts` and + `max_read_ms`; derive these limits from the task's remaining authorized budget, + not a new retry allowance. It returns chunk IDs, byte lengths and SHA256 values. + Store a large manifest in task-owned temporary storage; do not print it through + an undersized tool output. The helper reads no files and launches no models. +3. Emit takes the same authorized `materials` and one `chunk_id`. It returns only + that identified chunk: at most 512 Unicode characters / 2048 UTF-8 bytes. Read + one chunk at a time with enough actual tool output budget, including the JSON + wrapper. A prose request for a larger output limit does not set the limit. +4. On Codex, set the actual shell tool's `max_output_tokens` and the orchestration + tool's own output cap. When functions.exec is present, use a literal first-line + pragma, e.g. `// @exec: {"max_output_tokens": 3000}`, plus the nested command's + `max_output_tokens: 3000` for a single chunk. Do not combine a whole reference + bundle into that one capped call. On Claude native/skill routes, use the actual + Read offset/limit or bounded shell output supported by that worker tool; names + and controls vary, so do not copy Codex parameters into Claude. +5. Inspect the visible result for omissions, truncation markers and the complete + identified content. Source-side hashes alone and a worker saying “read it all” + are insufficient. If trusted host output is unavailable, record `unknown`. + Do not relabel source-file bytes as the observed tool output. + +## Assess, recover and preserve limits + +Assess takes `packet: {schema_version: 1, host, worker_id, manifest, observations}`. +Each observation has `chunk_id`, zero-based `attempt`, the actual visible `text`, +`truncated` (boolean or null), `source` (`host_tool_output`, `worker_assertion`, +`unavailable`), local evidence reference (or null), and `elapsed_ms` (or null). +Use `host_tool_output` only for host-visible evidence you actually inspected. +These are supplied records: the helper cannot authenticate a fabricated source +label or prove worker comprehension. Inspect the underlying evidence honestly. + +The helper checks visible content/length/hash, duplicate attempts, truncation +flags/markers, omissions and cumulative read budgets. It returns complete, +incomplete, unknown or budget_exceeded, missing IDs and the recoverable subset. +Append only missing chunk reads within the remaining budget; retain all earlier +attempts. Unknown timing does not authorize budgeted recovery. A complete later +read can repair a truncated chunk, but does not erase consumed attempts/time. +A duplicate attempt is invalid input. Never automatically restart the worker. + +For ordinary work, repair required missing inputs within scope or report the +input-delivery failure. Unknown completeness must be reported as unknown; it +is not proof of complete delivery, and does not itself reinstate a study gate +for Dynamic routing. Existing tests/review still determine the work's outcome. +Keep these receipts in the permitted quest plan (or final fast-lane report). + +## Optional study integration + +New study state uses version 2; existing manifests/state remain readable and +original outcomes/grades stay unchanged. New live trials cannot be claimed without a frozen delivery manifest, and an +invalid prior live trial stops further claims before more model usage. Before claiming a trial, freeze the +expected manifest with `study_runner.py delivery_plan --directory … --run … +--input manifest.json`. All trials for one fixture must use the same manifest. +After actual reads, append the cumulative assess packet with the `delivery` +operation. It must match that manifest, host and worker; earlier observations +cannot be removed. Record the ordinary outcome and actual consumption normally. + +`blind` excludes incomplete/unknown live delivery before paying for grading. +`grade` refuses those trials. `export` returns `invalid_input_delivery` and no +comparison dataset if any trial is invalid, even if an old grade exists; it does +not cherry-pick a favorable subset or manufacture quality failures. Missing +legacy live delivery is unknown. Historical synthetic fixtures keep explicitly +synthetic arithmetic compatibility; this is not live delivery evidence. +Keep prior exported reports unchanged and annotate their limits separately. +No new study or regrading run is required for ordinary Dynamic activation. + +## Feedback from work already performed + +`scripts/routing_feedback.py` accepts the v5 request, its decision, a worker +outcome, observation and optional existing usage packet. Outcome fields are +worker_id, status (completed/failed/interrupted), tests +(passed/failed/unknown/not_applicable), review +(accepted/rejected/unknown/not_applicable), retries, local evidence references +and delivery status. Observation fields are source +(host_metadata/worker_assertion/unknown), evidence, requested_resolved and +observed model/effort pairs (nullable), and configuration_supported (nullable). + +Reuse trusted host metadata, actual tests/reviews/retries and complete usage from +that work. Compare resolved identities only when known; requested aliases are not +served names. Worker assertions never become observed identity. Known forced +substitution/unsupported configuration suspends future choices without resetting +counters or replaying work. Unknown identity alone does not suspend Dynamic mode. + +Optional usage input uses the existing routing_usage schema, correlated to this +worker and host/role/category. Cache/reasoning semantics and incomplete counts +remain intact. Raw tokens are a subscription proxy; account-wide percentages do +not establish per-task allowance or monetary cost. The result leaves subscription +allowance percent null, requests zero extra runs and proposes catalog review only. +It never trains, mutates a policy/catalog, gives qualification or writes files. +Do not schedule duplicate runs, independent grading or extra retries for feedback. diff --git a/plugin/skills/guildhall-routing-setup/references/qualification-study.md b/plugin/skills/guildhall-routing-setup/references/qualification-study.md index a4337c4..487e21d 100644 --- a/plugin/skills/guildhall-routing-setup/references/qualification-study.md +++ b/plugin/skills/guildhall-routing-setup/references/qualification-study.md @@ -1,5 +1,11 @@ # Run a bounded routing study +This advanced workflow is optional. Dynamic routing does not require a study. +Before a paid trial, follow [bounded delivery](input-delivery.md), freeze its +expected input manifest and verify actual visible chunks. Exclude incomplete or +unknown live delivery before grading or comparing model quality/efficiency. +A failed study does not authorize another run or prevent Dynamic activation. + Guildhall supplies a development headroom analyzer and a study controller. They run no models themselves. The controller prepares isolated Git worktrees and explicit dispatch packets for the actual host, then imports measured outcomes diff --git a/plugin/skills/guildhall-routing-setup/references/routing.md b/plugin/skills/guildhall-routing-setup/references/routing.md index a9dd9c6..2d3bb3b 100644 --- a/plugin/skills/guildhall-routing-setup/references/routing.md +++ b/plugin/skills/guildhall-routing-setup/references/routing.md @@ -313,3 +313,9 @@ or automatically promotes a profile. Follow the [role matrix and migration guide](role-eligibility.md). Eligibility never changes role contracts, test-author handoffs, review membership or lifecycle gates. Upgrading does not add roles to existing allowlists or qualify profiles. + + +For required reference material, use [bounded input delivery](input-delivery.md) +and record ordinary-work feedback from existing tests/reviews/usage only. Repair +missing chunks within the existing budget; never treat source hashes or worker +assertions as delivery proof or run extra workers to collect feedback. diff --git a/plugin/skills/guildhall-routing-setup/references/setup-flow.md b/plugin/skills/guildhall-routing-setup/references/setup-flow.md index 308fed1..bd1c2af 100644 --- a/plugin/skills/guildhall-routing-setup/references/setup-flow.md +++ b/plugin/skills/guildhall-routing-setup/references/setup-flow.md @@ -198,3 +198,9 @@ configuration and approval only; it never resets an active quest's counters. qualification, corrupt files, unsafe permissions or an existing override. Do not silently overwrite malformed files, relax permissions or clear locks. Suggest the smallest concrete repair, then apply only the requested repair. + + +For a supplied study failure involving truncated references, read +[input delivery](input-delivery.md) and inspect only the supplied report. Preserve +its original outcomes. Offer Dynamic routing based on supported controls; do not +require regrading, extra trials or fabricated delivery evidence to enable it. diff --git a/plugin/skills/guildhall-routing-setup/scripts/routing_delivery.py b/plugin/skills/guildhall-routing-setup/scripts/routing_delivery.py new file mode 100644 index 0000000..798135f --- /dev/null +++ b/plugin/skills/guildhall-routing-setup/scripts/routing_delivery.py @@ -0,0 +1,109 @@ +#!/usr/bin/env python3 +"""Bounded handoff chunks and supplied host-visible delivery checks; no model calls.""" +from __future__ import annotations +import hashlib +import json +from pathlib import Path +import re +import runpy +import sys + +_R=runpy.run_path(str(Path(__file__).with_name('route_model.py')),run_name='_delivery_router') +obj,enum,array=_R['obj'],_R['enum'],_R['array'] +LIMIT=2*1024*1024 +MATERIAL=obj(id=_R['CID'],text=dict(type='string',maxLength=262144)) +CHUNK=obj(id=dict(type='string',pattern=r'^[a-z][a-z0-9_-]{0,31}:[0-9]+$',maxLength=40), + bytes=dict(type='integer',minimum=0,maximum=4096),sha256=_R['HASH']) +MANIFEST=obj(schema_version=enum([1]),chunks=array(CHUNK,1024,1), + max_read_attempts=dict(type='integer',minimum=1,maximum=4096), + max_read_ms=dict(type='integer',minimum=1,maximum=3600000)) +OBSERVATION=obj(chunk_id=CHUNK['properties']['id'],attempt=dict(type='integer',minimum=0), + text=dict(type='string',maxLength=8192),truncated=_R['nullable'](_R['BOOL']), + source=enum(['host_tool_output','worker_assertion','unavailable']),evidence=_R['nullable'](_R['STRING']), + elapsed_ms=_R['nullable'](_R['NUMBER'])) +PACKET=obj(schema_version=enum([1]),host=enum(_R['ROUTES']),worker_id=_R['STRING'], + manifest=MANIFEST,observations=array(OBSERVATION,4096)) +TRUNCATED=re.compile(r'(?:output (?:was |is )?truncated|truncated output|\[\.\.\.\s*truncated\s*\.\.\.\]|omitted \d+ lines)',re.I) + + +def need(value): + if not value:raise ValueError('invalid_delivery_input') + + +def chunks(materials): + _R['validate'](materials,array(MATERIAL,64,1)) + need(len({m['id'] for m in materials})==len(materials)) + need(sum(len(m['text'].encode()) for m in materials)<=LIMIT//2) + result=[] + for material in materials: + # At most 512 Unicode characters / 2048 UTF-8 bytes per visible read. + pieces=[material['text'][n:n+512] for n in range(0,len(material['text']),512)] or [''] + result.extend(dict(id=f"{material['id']}:{i}",text=text) for i,text in enumerate(pieces)) + need(len(result)<=1024) + return result + + +def plan(materials,max_read_attempts,max_read_ms): + result=dict(schema_version=1,chunks=[dict(id=c['id'],bytes=len(c['text'].encode()), + sha256=hashlib.sha256(c['text'].encode()).hexdigest()) for c in chunks(materials)], + max_read_attempts=max_read_attempts,max_read_ms=max_read_ms) + _R['validate'](result,MANIFEST) + need(max_read_attempts>=len(result['chunks'])) + return result + + +def emit(materials,chunk_id): + result=next((c for c in chunks(materials) if c['id']==chunk_id),None) + need(result is not None) + return result + + +def assess(packet): + _R['validate'](packet,PACKET) + manifest=packet['manifest'];expected={c['id']:c for c in manifest['chunks']} + need(len(expected)==len(manifest['chunks'])) + observations=packet['observations'];seen=set();complete=set();failed=set();unknown=set() + elapsed=0;timing_known=True + for o in observations: + cid=o['chunk_id'];identity=(cid,o['attempt']) + need(cid in expected and identity not in seen);seen.add(identity) + if o['elapsed_ms'] is None:timing_known=False + else:elapsed+=o['elapsed_ms'] + if o['source']!='host_tool_output' or o['truncated'] is None or o['evidence'] is None: + unknown.add(cid);continue + raw=o['text'].encode() + if (o['truncated'] or TRUNCATED.search(o['text']) or len(raw)!=expected[cid]['bytes'] or + hashlib.sha256(raw).hexdigest()!=expected[cid]['sha256']): + failed.add(cid) + else:complete.add(cid) + missing=sorted(set(expected)-complete) + exhausted=len(observations)>manifest['max_read_attempts'] or elapsed>manifest['max_read_ms'] + status=('budget_exceeded' if exhausted else 'complete' if not missing else + 'incomplete' if failed-set(complete) or set(missing)-unknown else 'unknown') + remaining=max(0,manifest['max_read_attempts']-len(observations)) + return dict(schema_version=1,host=packet['host'],worker_id=packet['worker_id'], + status=status,comparison_eligible=status=='complete',missing_chunks=missing, + manifest_hash=_R['policy_hash'](manifest),evidence_hash=_R['policy_hash'](packet), + attempts_used=len(observations),attempts_remaining=remaining,read_ms=elapsed if timing_known else None, + recoverable_chunks=missing[:remaining] if not exhausted and timing_known and elapsed=len(manifest['chunks'])) + need(len({c['id'] for c in manifest['chunks']})==len(manifest['chunks'])) + need(manifest['max_read_ms']<=state['manifest']['timeout_seconds']*1000) + peers=[r['delivery_manifest'] for r in state['runs'] if r['fixture_id']==run['fixture_id'] and 'delivery_manifest' in r] + need(all(p==manifest for p in peers)) + run['delivery_manifest']=manifest;save(root,state) + return dict(run_id=run_id,manifest_hash=digest(manifest),dispatch=False) + + +def delivery(directory,run_id,packet): + """Append visible-read evidence; never erase earlier attempts or reset budgets.""" + assessment=_D['assess'](packet) + with locked(directory) as (root,state): + run,_=selected(state,run_id) + need(run['status']=='running' and run.get('delivery_manifest')==packet['manifest']) + need(packet['host']==state['manifest']['host']['route']) + hashes=[digest(o) for o in packet['observations']] + prior=run.get('delivery_observations',[]) + need(hashes[:len(prior)]==prior) + need('delivery' not in run or run['delivery']['worker_id']==packet['worker_id']) + run['delivery']=assessment;run['delivery_observations']=hashes;save(root,state) + return assessment + + +def delivery_status(state,run): + assessment=run.get('delivery') + if assessment is not None: + return assessment['status'] + # Historical synthetic fixtures retain arithmetic coverage, not live proof. + return 'synthetic_unchecked' if state['manifest']['synthetic'] else 'unknown' + + +def comparable(state,run): + return delivery_status(state,run) in ('complete','synthetic_unchecked') + + def claim(directory,run_id): with locked(directory) as (root,state): run,fixture=selected(state,run_id) need(run['status']=='pending' and run['candidate'] is not None) need(not any(r['status']=='running' for r in state['runs'])) + if not state['manifest']['synthetic']: + need(state.get('schema_version',1)==1 or 'delivery_manifest' in run) + need(all(comparable(state,r) for r in state['runs'] if r['status']=='recorded')) recorded=[r['outcome'] for r in state['runs'] if r['status']=='recorded'] need(all(r['usage_tokens'] is not None for r in recorded)) need(state['prior_usage_tokens']+sum(r['usage_tokens'] for r in recorded)') known=state['prior_usage_tokens']+sum(r['outcome']['usage_tokens'] or 0 for r in state['runs'] if r['status']=='recorded') if known+(outcome['usage_tokens'] or 0)>state['manifest']['usage_budget_tokens']:violations.append('') + if 'delivery' in run:need(run['delivery']['worker_id']==outcome['worker_id']) run.update(status='recorded',outcome=outcome,tree_hash=tree_hash,violations=violations) - save(root,state);return dict(run_id=run_id,violations=violations,recorded=True) + save(root,state);return dict(run_id=run_id,violations=violations,recorded=True, + input_delivery=delivery_status(state,run),comparison_eligible=comparable(state,run)) def blind(directory): with locked(directory) as (root,state): - packets=[] + packets=[];excluded=[] for run in state['runs']: if run['status']!='recorded':continue + if not comparable(state,run): + excluded.append(dict(run_id=run['id'],reason='input_delivery_'+delivery_status(state,run)));continue _,fixture=selected(state,run['id']) _,snapshot_hash,artifacts=tree_evidence(verify_tree(root,state,run),include_artifacts=True) need(snapshot_hash==run['tree_hash']) @@ -226,7 +277,7 @@ def blind(directory): text=json.dumps(packet).lower() need(all(c['model'].lower() not in text for c in state['manifest']['candidates'])) packets.append(packet) - return dict(schema_version=1,packets=packets,instruction='Grade accepted/critical_misses independently; do not inspect study.json, worktree names or strategy metadata.') + return dict(schema_version=2,packets=packets,excluded=excluded,instruction='Grade accepted/critical_misses independently; do not inspect study.json, worktree names or strategy metadata.') def grade(directory,grades): @@ -237,7 +288,7 @@ def grade(directory,grades): shape(item,'blind_id accepted critical_misses reason') need(type(item['accepted']) is bool);_S['number'](item['critical_misses'],True);_S['string'](item['reason']) need(item['blind_id'] not in seen);seen.add(item['blind_id']) - run,_=selected(state,item['blind_id']);need(run['status']=='recorded' and 'grade' not in run) + run,_=selected(state,item['blind_id']);need(run['status']=='recorded' and 'grade' not in run and comparable(state,run)) need(tree_evidence(verify_tree(root,state,run))[1]==run['tree_hash']) run['grade']=item save(root,state);return dict(frozen_grades=len(grades),qualification=False) @@ -245,7 +296,15 @@ def grade(directory,grades): def export(directory): with locked(directory) as (root,state): - need(all(r['status']=='recorded' and 'grade' in r for r in state['runs'])) + need(all(r['status']=='recorded' for r in state['runs'])) + invalid=[dict(run_id=r['id'],reason='input_delivery_'+delivery_status(state,r)) for r in state['runs'] if not comparable(state,r)] + if invalid: + # Do not export a filtered, deceptively favorable comparison or regrade + # old outputs. Original outcomes/usage/grades stay unchanged on disk. + return dict(schema_version=2,status='invalid_input_delivery',qualification=False, + manifest_hash=state['manifest_hash'],invalid_trials=invalid, + reason='No model comparison exported; preserve original outcomes and consumed budget.') + need(all('grade' in r for r in state['runs'])) records=[] for run in state['runs']: need(tree_evidence(verify_tree(root,state,run))[1]==run['tree_hash']) @@ -266,7 +325,7 @@ def export(directory): def main(): parser=argparse.ArgumentParser(description=__doc__) - parser.add_argument('operation',choices=['prepare','select','claim','record','blind','grade','export']) + parser.add_argument('operation',choices=['prepare','select','claim','delivery_plan','delivery','record','blind','grade','export']) parser.add_argument('--directory',required=True,type=Path) parser.add_argument('--manifest',type=Path);parser.add_argument('--repo',type=Path) parser.add_argument('--phase',default='development',choices=['development','holdout']) diff --git a/scripts/build_portable.py b/scripts/build_portable.py index e03cff3..4692a14 100644 --- a/scripts/build_portable.py +++ b/scripts/build_portable.py @@ -10,7 +10,7 @@ SETUP_DEST = Path('plugin/skills/guildhall-routing-setup') DESTINATIONS = (DEST, SETUP_DEST) SETUP_REFERENCES = ('global-routing', 'model-routing', 'routing', 'host-evidence', - 'role-eligibility', 'qualification-study', 'task-routing') + 'role-eligibility', 'qualification-study', 'task-routing', 'input-delivery') def outputs(root: Path) -> dict[Path, bytes]: diff --git a/tests/test_routing_delivery.py b/tests/test_routing_delivery.py new file mode 100644 index 0000000..1f27207 --- /dev/null +++ b/tests/test_routing_delivery.py @@ -0,0 +1,154 @@ +"""Offline truncation/recovery and normal-work feedback, without paid workers.""" +import copy +import runpy +import json +import subprocess +import sys +import tempfile +import shutil +from pathlib import Path +import unittest +from test_routing import ROOT, load_script, provider, NOW +from test_routing_dynamic import dynamic_request +import test_routing_study as study_tests + +D=runpy.run_path(str(ROOT/'plugin/portable/scripts/routing_delivery.py')) +F=runpy.run_path(str(ROOT/'plugin/portable/scripts/routing_feedback.py')) + + +def packet(host='codex-skill'): + materials=[dict(id='role',text='Allowed role contract. '+('é漢字 '*180)),dict(id='reference',text='Required credential details and GUI inheritance.')] + chunk_list=D['chunks'](materials) + manifest=D['plan'](materials,len(chunk_list)+2,1000) + return dict(schema_version=1,host=host,worker_id='synthetic-worker',manifest=manifest, + observations=[dict(chunk_id=c['id'],attempt=0,text=c['text'],truncated=False,source='host_tool_output',evidence='synthetic-visible-tool-result',elapsed_ms=1) for c in chunk_list]) + + +class DeliveryTests(unittest.TestCase): + def test_complete_truncated_missing_duplicated_and_unknown_on_every_host(self): + for host in ['codex-skill','claude-native','claude-skill']: + p=packet(host);self.assertEqual(D['assess'](p)['status'],'complete') + for change in ['truncated','marker','omitted','modified','assertion','unknown']: + q=copy.deepcopy(p) + if change=='truncated':q['observations'][0]['truncated']=True + if change=='marker':q['observations'][0]['text']='Warning: truncated output' + if change=='omitted':q['observations'].pop() + if change=='modified':q['observations'][0]['text']+='X' + if change=='assertion': + for o in q['observations']:o['source']='worker_assertion' + if change=='unknown': + for o in q['observations']:o['truncated']=None + out=D['assess'](q);self.assertFalse(out['comparison_eligible'],change) + p['observations'].append(copy.deepcopy(p['observations'][0])) + with self.assertRaises(ValueError):D['assess'](p) + + def test_recover_only_missing_chunk_without_restarting_budget(self): + p=packet();original=copy.deepcopy(p['observations'][0]) + p['observations'][0]['text']='short';p['observations'][0]['truncated']=True + out=D['assess'](p);self.assertEqual(out['recoverable_chunks'],[original['chunk_id']]) + original['attempt']=1;p['observations'].append(original) + out=D['assess'](p);self.assertTrue(out['comparison_eligible']);self.assertFalse(out['replay_worker']) + p['observations'][-1]['elapsed_ms']=2000 + self.assertEqual(D['assess'](p)['status'],'budget_exceeded') + + def test_emit_is_bounded_and_hashing_source_alone_is_not_delivery(self): + materials=[dict(id='ref',text='漢字'*3000)] + manifest=D['plan'](materials,20,500) + for c in manifest['chunks']: + emitted=D['emit'](materials,c['id']) + self.assertLessEqual(len(emitted['text'].encode()),2048) + p=packet();p['observations']=[] + self.assertFalse(D['assess'](p)['comparison_eligible']) + + + def test_delivery_cli_works_from_each_independently_copied_skill(self): + for name in ['guildhall-quest','guildhall-routing-setup']: + with tempfile.TemporaryDirectory() as tmp: + installed=Path(tmp)/'skill';shutil.copytree(ROOT/'plugin/skills'/name,installed) + result=subprocess.run([sys.executable,'-I','-B',str(installed/'scripts/routing_delivery.py')], + input=json.dumps(dict(operation='assess',packet=packet())),text=True,capture_output=True,timeout=10) + self.assertEqual(result.returncode,0,result.stderr) + self.assertTrue(json.loads(result.stdout)['comparison_eligible']) + + +class StudyDeliveryTests(unittest.TestCase): + setUp=study_tests.RunnerTests.setUp + prepare=study_tests.RunnerTests.prepare + outcome=study_tests.RunnerTests.outcome + + def test_invalid_trial_is_not_graded_or_exported_and_consumption_remains(self): + state=self.prepare();p=packet() + for run in state['runs']: + self.r['delivery_plan'](self.directory,run['id'],p['manifest']) + self.r['claim'](self.directory,run['id']) + bad=copy.deepcopy(p);bad['observations'][0]['truncated']=True + self.r['delivery'](self.directory,run['id'],bad) + result=self.r['record'](self.directory,run['id'],self.outcome()) + self.assertFalse(result['comparison_eligible']) + self.assertEqual(self.r['blind'](self.directory)['packets'],[]) + with self.assertRaises(ValueError):self.r['grade'](self.directory,[dict(blind_id=state['runs'][0]['id'],accepted=True,critical_misses=0,reason='would hide bad delivery')]) + result=self.r['export'](self.directory) + self.assertEqual(result['status'],'invalid_input_delivery') + saved=self.r['read'](self.directory/'study.json') + self.assertEqual(sum(r['outcome']['usage_tokens'] for r in saved['runs']),400) + + def test_delivery_manifest_freezes_before_claim_and_recovery_appends(self): + state=self.prepare();run=state['runs'][0];p=packet() + self.r['delivery_plan'](self.directory,run['id'],p['manifest']);self.r['claim'](self.directory,run['id']) + with self.assertRaises(ValueError):self.r['delivery_plan'](self.directory,run['id'],p['manifest']) + missing=copy.deepcopy(p);missing['observations'].pop() + self.r['delivery'](self.directory,run['id'],missing) + self.assertEqual(self.r['delivery'](self.directory,run['id'],p)['status'],'complete') + with self.assertRaises(ValueError):self.r['delivery'](self.directory,run['id'],missing) + self.assertTrue(self.r['record'](self.directory,run['id'],self.outcome())['comparison_eligible']) + + def test_legacy_live_trial_without_delivery_is_unknown_not_model_failure(self): + self.m['synthetic']=False;state=self.prepare();run=state['runs'][0] + state['schema_version']=1;self.r['save'](self.directory,state) + self.r['claim'](self.directory,run['id']) + result=self.r['record'](self.directory,run['id'],self.outcome()) + self.assertEqual(result['input_delivery'],'unknown') + self.assertEqual(self.r['blind'](self.directory)['packets'],[]) + + + def test_new_live_study_requires_manifest_and_stops_after_bad_delivery(self): + self.m['synthetic']=False;state=self.prepare();run=state['runs'][0];p=packet() + with self.assertRaises(ValueError):self.r['claim'](self.directory,run['id']) + for item in state['runs']:self.r['delivery_plan'](self.directory,item['id'],p['manifest']) + self.r['claim'](self.directory,run['id']) + p['observations'][0]['truncated']=True + self.r['delivery'](self.directory,run['id'],p) + self.r['record'](self.directory,run['id'],self.outcome()) + with self.assertRaises(ValueError):self.r['claim'](self.directory,state['runs'][1]['id']) + + +class FeedbackTests(unittest.TestCase): + def test_unknown_identity_and_usage_need_no_runs_but_known_substitution_suspends(self): + r=dynamic_request();decision=load_script().route(r,transport=lambda *a:provider(),now=NOW) + outcome=dict(worker_id='worker',status='completed',tests='passed',review='unknown',retries=0,evidence=['test-result'],delivery='complete') + observation=dict(source='unknown',evidence=None,requested_resolved=dict(model=None,effort=None),observed=dict(model=None,effort=None),configuration_supported=None) + result=F['feedback'](r,decision,outcome,observation) + self.assertFalse(result['state']['adaptive_suspended']);self.assertIsNone(result['usage']) + self.assertIsNone(result['subscription_allowance_percent']);self.assertEqual(result['additional_runs'],0) + observation.update(source='host_metadata',evidence='host-tool-result',requested_resolved=dict(model='resolved-fast',effort='high'),observed=dict(model='forced-model',effort=None)) + result=F['feedback'](r,decision,outcome,observation) + self.assertTrue(result['state']['adaptive_suspended']);self.assertFalse(result['replay_worker']) + self.assertEqual(result['state']['calls_used'],decision['state']['calls_used']) + observation['source']='worker_assertion' + self.assertIsNone(F['feedback'](r,decision,outcome,observation)['observed']['model']) + + + def test_feedback_correlates_task_and_usage_and_preserves_unknown_totals(self): + from test_routing_usage import packet as usage_packet + r=dynamic_request();decision=load_script().route(r,transport=lambda *a:provider(),now=NOW) + outcome=dict(worker_id='w',status='completed',tests='passed',review='unknown',retries=0,evidence=['tests'],delivery='unknown') + observation=dict(source='unknown',evidence=None,requested_resolved=dict(model=None,effort=None),observed=dict(model=None,effort=None),configuration_supported=None) + usage=usage_packet();usage['inventory_complete']=False + result=F['feedback'](r,decision,outcome,observation,usage) + self.assertIsNone(result['usage']['meters']['host']['usage_tokens']) + usage['scope']['role']='pr-author' + with self.assertRaises(ValueError):F['feedback'](r,decision,outcome,observation,usage) + r['task']['risk']='high' + with self.assertRaises(ValueError):F['feedback'](r,decision,outcome,observation) + +if __name__=='__main__':unittest.main() diff --git a/tests/test_routing_study.py b/tests/test_routing_study.py index a87ca23..589d75b 100644 --- a/tests/test_routing_study.py +++ b/tests/test_routing_study.py @@ -142,8 +142,14 @@ def test_live_outcome_usage_and_effort_must_match_capture(self): self.m['synthetic']=False;self.m['host']['evidence_requirement']='execution_observed' for c in self.m['candidates']:c['effort']='high' state=self.prepare() + material=[dict(id='task',text=self.m['fixtures'][0]['prompt'])] + delivery_manifest=self.r['_D']['plan'](material,2,1000) for run,change in zip(state['runs'],['usage','effort','valid']): + self.r['delivery_plan'](self.directory,run['id'],delivery_manifest) packet=self.r['claim'](self.directory,run['id']) + self.r['delivery'](self.directory,run['id'],dict(schema_version=1,host='codex-skill',worker_id='synthetic-worker', + manifest=delivery_manifest,observations=[dict(chunk_id='task:0',attempt=0,text=material[0]['text'], + truncated=False,source='host_tool_output',evidence='synthetic-visible-input',elapsed_ms=1)])) out=self.outcome() out['host_report']=dict(synthetic=False,complete=True,worker_id=out['worker_id'], scope=dict(host='codex-skill',role='docs-writer',category='docs'),requested=packet['settings'],