Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
231 changes: 166 additions & 65 deletions backend/.env.example
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
# Server runtime: development | test | production.
# ── Server Runtime ──────────────────────────────────────────────────────────────
# Server runtime environment: development | test | production.
NODE_ENV=development

# HTTP port for the backend API.
Expand All @@ -13,8 +14,8 @@ NPM_PACKAGE_VERSION=0.1.0
# Seconds to wait during graceful shutdown before forcing process exit.
GRACEFUL_SHUTDOWN_TIMEOUT=30

# SQLite/PostgreSQL connection string used by backend persistence.
DATABASE_URL=./data/ai-net.db
# Timeout in milliseconds for health probe HTTP checks.
HEALTH_PROBE_TIMEOUT_MS=5000

# Comma-separated browser origins allowed by CORS.
ALLOWED_ORIGINS=http://localhost:3000
Expand All @@ -38,34 +39,60 @@ STELLAR_HORIZON_URL=https://horizon-testnet.stellar.org
# Optional legacy alias for STELLAR_HORIZON_URL.
STELLAR_HORIZON=

# Optional backend coordinator secret key for payment release signing.
STELLAR_COORDINATOR_SECRET=

# Optional testnet secret key used by live Stellar E2E tests.
STELLAR_TEST_SECRET=

# Optional public key used when local agents self-register.
STELLAR_PUBLIC_KEY=

# Set true only in local/test runs to bypass Horizon account validation.
SKIP_STELLAR_ACCOUNT_VERIFY=false

# Soroban RPC URL used by registry event sync.
SOROBAN_RPC_URL=https://soroban-testnet.stellar.org

# Optional deployed AgentRegistry contract ID for event sync.
REGISTRY_CONTRACT_ID=

# Set true only in local/test runs to bypass Horizon account validation.
SKIP_STELLAR_ACCOUNT_VERIFY=false
# Optional backend coordinator secret key for payment release signing.
STELLAR_COORDINATOR_SECRET=

# Optional testnet secret key used by live Stellar E2E tests.
STELLAR_TEST_SECRET=

# ── API & Authentication ───────────────────────────────────────────────────────
# Optional comma-separated bearer API keys for protected API access.
API_KEYS=

# Optional shared secret for admin endpoints such as /health/dashboard.
ADMIN_API_KEY=

# JWT secret key used to sign and verify access tokens.
AUTH_JWT_SECRET=ai-net-default-auth-secret-change-in-production

# Access token validity in seconds. Default: 900 (15 min).
AUTH_ACCESS_TOKEN_TTL_SECONDS=900

# Refresh token sliding expiry validity in seconds. Default: 604800 (7 days).
AUTH_REFRESH_TOKEN_TTL_SECONDS=604800

# Max absolute session lifetime in seconds. Default: 2592000 (30 days).
AUTH_SESSION_MAX_TTL_SECONDS=2592000

# ── API Versioning ────────────────────────────────────────────────────────────
# Latest API version advertised by versioning middleware.
API_LATEST_VERSION=2.0

# Comma-separated API versions accepted by the server.
API_SUPPORTED_VERSIONS=1.0,1.1,2.0

# API version used when the client omits API-Version.
API_DEFAULT_VERSION=1.0

# Optional HTTP Sunset header value for v1 clients.
API_V1_SUNSET_DATE=

# ── Venice AI & LLM Budget ────────────────────────────────────────────────────
# Venice AI API key. Required outside NODE_ENV=test.
VENICE_API_KEY=your_venice_api_key_here

# ── Database ──────────────────────────────────────────────────────────────────
# Filesystem path to the SQLite database holding the ai-net schema.
# Applied by `npm run db:migrate` (see src/db/migrations/).
DATABASE_URL=./data/ai-net.db

# Optional: keep the versioned migration files outside src/db/migrations.
# DB_MIGRATIONS_DIR=./db/migrations
# Venice API base URL.
VENICE_BASE_URL=https://api.venice.ai/api/v1

Expand All @@ -78,9 +105,75 @@ VENICE_CACHE_TTL_MS=86400000
# Shorter Venice cache TTL for coding-agent responses in milliseconds.
VENICE_CACHE_CODING_TTL_MS=3600000

# Similarity threshold for semantic Venice cache reuse.
# Similarity threshold for semantic Venice cache reuse (0.0 to 1.0).
VENICE_CACHE_SIMILARITY_THRESHOLD=0.8

# Venice HTTP request timeout in milliseconds.
VENICE_REQUEST_TIMEOUT_MS=10000

# Maximum retries per provider before failing over.
VENICE_PROVIDER_MAX_RETRIES=3

# Optional comma-separated fallback API keys for Venice failover.
VENICE_FALLBACK_API_KEYS=

# Optional comma-separated fallback base URLs for Venice failover.
VENICE_FALLBACK_BASE_URLS=

# Total tokens (input + output) a single task may consume before it halts.
TASK_TOKEN_BUDGET=200000

# Ceiling on one LLM call's max_tokens.
LLM_MAX_TOKENS_PER_CALL=8192

# Ceiling on one LLM call's input prompt; longer prompts are trimmed.
LLM_MAX_PROMPT_TOKENS=16000

# Format: model_name=inputUsd:outputUsd,... overrides for the pricing table.
VENICE_PRICING=

# How often in-flight task costs are flushed to the database in ms.
COST_FLUSH_INTERVAL_MS=60000

# ── Database & SQLite WAL Read Pool ───────────────────────────────────────────
# Filesystem path to the primary SQLite database file holding the ai-net schema.
# Note: SQLite with WAL mode is used for single-writer thread safety.
DATABASE_URL=./data/ai-net.db

# Optional: keep the versioned migration files outside src/db/migrations.
DB_MIGRATIONS_DIR=

# SQLite connection pool size (minimum eager read-only connections).
DB_POOL_MIN=2

# SQLite connection pool size (maximum concurrent WAL read-only connections).
DB_POOL_MAX=10

# Timeout in ms to acquire a read-only SQLite connection handle.
DB_POOL_ACQUIRE_TIMEOUT_MS=5000

# Whether to run SELECT 1 on pooled SQLite reader handles before reuse.
DB_POOL_HEALTH_CHECK=true

# Database backup target directory.
DB_BACKUP_DIR=./data/backups

# Number of database backups to retain.
DB_BACKUP_RETENTION_COUNT=5

# Interval in ms for database maintenance tasks (e.g. WAL checkpoint, VACUUM).
DB_MAINTENANCE_INTERVAL_MS=3600000

# Auto-vacuum page count threshold before maintenance executes VACUUM.
DB_MAINTENANCE_VACUUM_THRESHOLD=100

# Path to the admin audit log SQLite database file.
ADMIN_AUDIT_DB_PATH=

# Target directory for admin-triggered system backups.
ADMIN_BACKUP_DIR=

# ── Caching ───────────────────────────────────────────────────────────────────
# Cache backend: lru | redis.
CACHE_DRIVER=lru

Expand All @@ -99,40 +192,39 @@ CACHE_TTL_STATS=30
# Health cache TTL in seconds.
CACHE_TTL_HEALTH=10

# Deployment-scoped prefix for registry cache keys (Issue #427).
# Change this per environment (e.g. "staging", "prod") when multiple
# deployments share the same Redis instance.
# Deployment-scoped prefix for registry cache keys.
REGISTRY_CACHE_KEY_PREFIX=registry

# ── Rate Limiting & Quotas ────────────────────────────────────────────────────
# Maximum accepted task prompt length in characters.
MAX_PROMPT_LENGTH=10000

# Maximum tasks per wallet per rolling 24h window. Set 0 to disable.
DAILY_TASK_LIMIT_PER_WALLET=100

# Per-IP rate-limit window in milliseconds.
# Global fallback per-IP rate-limit window in milliseconds.
RATE_LIMIT_WINDOW_MS=60000

# Maximum task-create requests per IP per window.
# Global fallback maximum task-create requests per IP per window.
RATE_LIMIT_MAX_REQUESTS=20

# Per-route-group limits (token-bucket, per IP, rolling window)
# Public routes: /health, /api/stats, GET /api/agents
# Rate limit for agent registration endpoint (/api/agents/register).
REGISTER_RATE_LIMIT_MAX_REQUESTS=10

# Rate limit for public unauthenticated endpoints (/health, /api/stats, GET /api/agents).
RATE_LIMIT_PUBLIC_WINDOW_MS=60000
RATE_LIMIT_PUBLIC_MAX_REQUESTS=120
# Authenticated routes: /api/tasks

# Rate limit for authenticated endpoints (/api/tasks).
RATE_LIMIT_AUTHED_WINDOW_MS=60000
RATE_LIMIT_AUTHED_MAX_REQUESTS=30
# Admin routes: /api/admin/*

# Rate limit for admin endpoints (/api/admin/*).
RATE_LIMIT_ADMIN_WINDOW_MS=60000
RATE_LIMIT_ADMIN_MAX_REQUESTS=20

# ── Per-Wallet Daily Quota ────────────────────────────────────────────────────
# Maximum tasks a single wallet address may create within a rolling 24-hour
# window. Set to 0 to disable the quota entirely.
# Per-wallet rolling 24-hour daily task creation limit (0 to disable).
DAILY_TASK_LIMIT_PER_WALLET=100

# Agent cleanup loop interval in milliseconds.
# ── Heartbeat & Agent Watchdog ────────────────────────────────────────────────
# Agent heartbeat loop interval in milliseconds.
HEARTBEAT_INTERVAL_MS=300000

# Minutes before an unseen agent is considered stale.
Expand All @@ -141,33 +233,42 @@ HEARTBEAT_STALE_THRESHOLD_MINUTES=5
# Hours before stale offline agents are deleted.
AGENT_OFFLINE_DELETE_HOURS=24

# Optional webhook for reconciliation discrepancy alerts.
RECONCILIATION_WEBHOOK_URL=
# Agent watchdog check interval in milliseconds.
AGENT_WATCHDOG_INTERVAL_MS=60000

# Automated reconciliation interval in milliseconds.
RECONCILIATION_INTERVAL_MS=86400000
# Grace period in minutes before watchdog marks unresponsive agents offline.
AGENT_WATCHDOG_GRACE_MINUTES=10

# Minimum response size in bytes before compression is attempted.
COMPRESSION_THRESHOLD=1024
# Error registry cleanup interval in milliseconds.
ERROR_REGISTRY_MAINTENANCE_INTERVAL_MS=3600000

# gzip/Brotli compression level. gzip uses 1-9.
COMPRESSION_LEVEL=6
# Maximum error records kept per agent in the error registry.
ERROR_REGISTRY_CAP_PER_AGENT=100

# Enable Brotli when clients advertise br support.
COMPRESSION_ENABLE_BROTLI=true
# ── Event Store Retention & Compaction ────────────────────────────────────────
# On-disk path for the append-only event store SQLite database.
EVENT_STORE_PATH=./data/events.db

# Latest API version advertised by versioning middleware.
API_LATEST_VERSION=2.0
# Retention window in days for finished task events.
EVENT_RETENTION_DAYS=30

# Comma-separated API versions accepted by the server.
API_SUPPORTED_VERSIONS=1.0,1.1,2.0
# How often the compaction pass runs in milliseconds.
EVENT_COMPACTION_INTERVAL_MS=3600000

# API version used when the client omits API-Version.
API_DEFAULT_VERSION=1.0
# Maximum tasks compacted per pass.
EVENT_COMPACTION_BATCH_TASKS=50

# Optional HTTP Sunset header value for v1 clients.
API_V1_SUNSET_DATE=
# Master switch for the event store retention compaction job.
EVENT_COMPACTION_ENABLED=true

# ── Idempotency Store ─────────────────────────────────────────────────────────
# Retention TTL in milliseconds for idempotency keys.
IDEMPOTENCY_TTL_MS=86400000

# Background cleanup interval in milliseconds for expired idempotency keys.
IDEMPOTENCY_CLEANUP_MS=300000

# ── WebSockets ────────────────────────────────────────────────────────────────
# Maximum concurrent WebSocket stream connections per client IP.
WS_MAX_CONNECTIONS_PER_CLIENT=5

Expand All @@ -183,6 +284,7 @@ WS_HEARTBEAT_INTERVAL_MS=30000
# WebSocket pong timeout in milliseconds.
WS_PONG_TIMEOUT_MS=10000

# ── Metrics ───────────────────────────────────────────────────────────────────
# Health dashboard cache TTL in milliseconds.
METRICS_CACHE_TTL_MS=5000

Expand All @@ -192,22 +294,21 @@ METRICS_WINDOW_MS=60000
# Maximum HTTP request samples retained in memory.
METRICS_MAX_SAMPLES=1000

# ── Event store retention & compaction ───────────────────────────────────────
# ── Quality Scorer ────────────────────────────────────────────────────────────
# Completeness dimension score weight (0.0 to 1.0).
QUALITY_WEIGHT_COMPLETENESS=0.4

# On-disk path for the append-only event store. Must be a file path for the
# retention job to be meaningful; the event log is discarded on restart when
# this is :memory:.
EVENT_STORE_PATH=./data/events.db
# Relevance dimension score weight (0.0 to 1.0).
QUALITY_WEIGHT_RELEVANCE=0.3

# Retention window in days. Events for finished tasks older than this are
# archived (full-fidelity) and purged from the live task_events table.
EVENT_RETENTION_DAYS=30
# Format dimension score weight (0.0 to 1.0).
QUALITY_WEIGHT_FORMAT=0.3

# How often the compaction pass runs in milliseconds.
EVENT_COMPACTION_INTERVAL_MS=3600000
# Score threshold below which output is flagged for review.
QUALITY_REVIEW_THRESHOLD=60

# Maximum tasks compacted per pass, bounding writer-lock hold time.
EVENT_COMPACTION_BATCH_TASKS=50
# Enable percentile rank normalization across historical agent output scores.
QUALITY_PERCENTILE_ENABLED=false

# Master switch for the retention job.
EVENT_COMPACTION_ENABLED=true
Expand Down
5 changes: 0 additions & 5 deletions backend/jest.config.js
Original file line number Diff line number Diff line change
Expand Up @@ -29,12 +29,7 @@ module.exports = {
'!src/**/*.d.ts',
'!src/**/*.test.ts',
'!src/**/*.spec.ts',
'!src/**/index.ts',
'!src/**/.gitkeep',
'!src/registry/sync.ts',
'!src/api/routes/stream.ts',
'!src/index.ts',
'!src/checkSpec.ts',
],
coverageThreshold: {
global: {
Expand Down
5 changes: 2 additions & 3 deletions backend/src/api/middleware/auth.ts
Original file line number Diff line number Diff line change
Expand Up @@ -153,10 +153,9 @@ export function resolveAdminApiKey(): string | undefined {
fromConfig = (require("../../config") as typeof import("../../config")).getConfig()
.ADMIN_API_KEY;
} catch {
// Config not loaded — fall through to the environment.
// Config not loaded — ignore.
}
const key = fromConfig ?? process.env.ADMIN_API_KEY;
return key && key.length > 0 ? key : undefined;
return fromConfig && fromConfig.length > 0 ? fromConfig : undefined;
}

/** Constant-time string comparison; length differences short-circuit safely. */
Expand Down
8 changes: 5 additions & 3 deletions backend/src/api/middleware/errorHandler.ts
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,6 @@ import { AppError } from "../../errors";
import { getConfig } from "../../config";
import { HTTP_STATUS_FOR_CODE } from "../../errors/ErrorCode";

const isProduction = process.env.NODE_ENV === "production";

/**
* Build the canonical error envelope for every API response.
Expand Down Expand Up @@ -164,7 +163,10 @@ export function errorHandler(
"unhandled error",
);

const message = isProduction
const isProd = getConfig().NODE_ENV === "production";
const isDev = getConfig().NODE_ENV === "development";

const message = isProd
? "Internal server error"
: err instanceof Error
? err.message || "Internal server error"
Expand All @@ -176,7 +178,7 @@ export function errorHandler(
statusCode,
path,
correlationId,
details: isDevelopment && err instanceof Error ? { stack: err.stack } : undefined,
details: isDev && err instanceof Error ? { stack: err.stack } : undefined,
});

res.status(statusCode).json(body);
Expand Down
Loading