mirror of
https://github.com/luckyyzh/pi-agent-integrated.git
synced 2026-10-03 02:59:35 +00:00
feat: add codex fast mode and harden subagent launch
This commit is contained in:
@@ -80,6 +80,12 @@ npm run dev
|
||||
|
||||
所有设备相关模型配置都位于被 Git 忽略的 `data/` 或 `.env` 中。
|
||||
|
||||
#### GPT Codex 快速模式
|
||||
|
||||
选择 `openai-codex` 的 GPT-5.5 或 GPT-5.6 模型后,输入框底部会显示闪电按钮。开启后按钮显示 `Fast mode ON`,并将该模型的 `serviceTier` 设为 `priority`,请求会使用 Codex 的快速档位;关闭后恢复普通档位。该设置写入 `data/agent/models.json`,下次请求立即生效,无需重启;非 GPT Codex 模型不会显示此按钮。
|
||||
|
||||
快速档位会增加费用:GPT-5.6 通常约为 2 倍,GPT-5.5 的 priority 费率约为 2.5 倍。Web UI 会在按钮提示中明确显示 `service_tier: priority` 和费用倍率。
|
||||
|
||||
### 可选环境变量
|
||||
|
||||
需要可选服务时,复制模板:
|
||||
@@ -147,6 +153,8 @@ data/workspaces/default/ 默认工作目录
|
||||
|
||||
Windows 的 Playwright 不下载独立 Chromium;首次 `setup` 只缓存 MCP 的 Node.js 包,浏览器执行使用系统 Edge。macOS 的 setup 不安装或启用 Playwright;如需浏览器自动化,可在 Web UI 的 MCP 面板中手动添加并配置。
|
||||
|
||||
Windows 下的 `pi-subagents` 子进程由受管启动器自动处理:启动时将本地 `pi-web/node_modules/.bin` 加入子进程 PATH,并把 Pi Web 使用的 `pi-coding-agent` 包链接到受管 Profile,使子代理直接解析本地 `dist/cli.js`。这不会修改系统级 PATH,每次项目启动时会自动恢复。
|
||||
|
||||
#### 视觉子代理(vision)
|
||||
|
||||
DeepSeek 等纯文本模型不能接收图片。仓库内置 `vision` 子代理(`.agents/vision.md`):它通过 `vision` 工具调用视觉模型读取图片,把完整 OCR、版式结构与语义描述返回给主模型,主模型基于文本继续推理。视觉后端可插拔(本地 Ollama 或任意 OpenAI 兼容视觉 API)——仓库**不预设默认后端**,首次使用前需自行选择并配置。
|
||||
@@ -326,6 +334,12 @@ The repository contains no author model endpoint, API key, OAuth token, or defau
|
||||
|
||||
Device-specific model configuration remains in ignored `data/` or `.env` files.
|
||||
|
||||
#### GPT Codex fast mode
|
||||
|
||||
When an `openai-codex` GPT-5.5 or GPT-5.6 model is selected, the lightning button appears beside the model selector. Enabling it shows `Fast mode ON` and sets that model's `serviceTier` to `priority`, using Codex's faster service tier; disabling it restores the default tier. The setting is written to `data/agent/models.json` and takes effect on the next request without a restart. Non-Codex models do not show the button.
|
||||
|
||||
Fast mode costs more: GPT-5.6 is typically about 2x, while GPT-5.5 priority pricing is about 2.5x. The Web UI tooltip displays `service_tier: priority` and the applicable credit multiplier.
|
||||
|
||||
### Optional environment configuration
|
||||
|
||||
macOS/Linux:
|
||||
@@ -391,6 +405,8 @@ Versions are pinned in the platform defaults under `config/`: Windows uses `mcp.
|
||||
|
||||
On Windows, Playwright never downloads a standalone Chromium: setup caches only its Node package and browser execution uses system Edge. On macOS, setup does not install or enable Playwright; add it manually through the MCP panel if browser automation is needed.
|
||||
|
||||
On Windows, the managed launcher prepares `pi-subagents` child processes automatically: it prepends the local `pi-web/node_modules/.bin` directory to the child PATH and links the Pi Web `pi-coding-agent` package into the managed profile, allowing subagents to resolve the local `dist/cli.js` directly. This does not modify the system-wide PATH and is recreated on each project launch.
|
||||
|
||||
#### Vision subagent
|
||||
|
||||
Text-only models such as DeepSeek cannot receive image attachments. The repository ships a `vision` subagent (`.agents/vision.md`) that calls a vision model through the `vision` tool and returns a full OCR, layout, and semantic description the main model can reason over. The vision backend is pluggable (local Ollama or any OpenAI-compatible vision API) — the repository does **not** ship a default backend; pick and configure one before first use.
|
||||
|
||||
@@ -10,15 +10,16 @@ $supervisorLog = Join-Path $logsDirectory 'pi-web-supervisor.log'
|
||||
$restartRequestPath = Join-Path $projectRoot 'data\agent\restart-request.json'
|
||||
$webUrl = 'http://127.0.0.1:30141/'
|
||||
$restartRequestVersion = 1
|
||||
$maxResumeAttempts = 3
|
||||
|
||||
New-Item -ItemType Directory -Path $logsDirectory -Force | Out-Null
|
||||
New-Item -ItemType Directory -Path (Split-Path -Parent $restartRequestPath) -Force | Out-Null
|
||||
Set-Content -LiteralPath $supervisorLog -Value "$(Get-Date -Format o) supervisor started"
|
||||
Set-Content -LiteralPath $supervisorLog -Value "$(Get-Date -Format o) supervisor started" -Encoding UTF8
|
||||
|
||||
function Write-SupervisorLog {
|
||||
param([string]$Message)
|
||||
|
||||
Add-Content -LiteralPath $supervisorLog -Value "$(Get-Date -Format o) $Message"
|
||||
Add-Content -LiteralPath $supervisorLog -Value "$(Get-Date -Format o) $Message" -Encoding UTF8
|
||||
}
|
||||
|
||||
function Get-RestartRequest {
|
||||
@@ -72,6 +73,20 @@ function Remove-RestartRequest {
|
||||
Remove-Item -LiteralPath $restartRequestPath -Force -ErrorAction SilentlyContinue
|
||||
}
|
||||
|
||||
function Archive-RestartRequest {
|
||||
param([object]$Request)
|
||||
|
||||
if (-not (Test-Path -LiteralPath $restartRequestPath -PathType Leaf)) {
|
||||
return
|
||||
}
|
||||
|
||||
$requestId = (Get-RequestField $Request 'requestId') -replace '[^A-Za-z0-9-]', '_'
|
||||
$timestamp = Get-Date -Format 'yyyyMMdd-HHmmss'
|
||||
$archivePath = Join-Path $logsDirectory "failed-restart-request-$timestamp-$requestId.json"
|
||||
Move-Item -LiteralPath $restartRequestPath -Destination $archivePath -Force
|
||||
Write-SupervisorLog "archived failed restart request at $archivePath"
|
||||
}
|
||||
|
||||
function Stop-ProcessTree {
|
||||
param([System.Diagnostics.Process]$Process)
|
||||
|
||||
@@ -149,6 +164,7 @@ $testInstructions
|
||||
}
|
||||
|
||||
$pendingRequest = Get-RestartRequest
|
||||
$resumeAttemptCount = 0
|
||||
|
||||
while ($true) {
|
||||
$process = $null
|
||||
@@ -176,10 +192,20 @@ while ($true) {
|
||||
try {
|
||||
Resume-AgentSession $pendingRequest
|
||||
$pendingRequest = $null
|
||||
$resumeAttemptCount = 0
|
||||
}
|
||||
catch {
|
||||
$resumeAttemptCount += 1
|
||||
Write-SupervisorLog "Agent resume failed: $($_.Exception.Message)"
|
||||
$nextResumeAttempt = (Get-Date).AddSeconds(3)
|
||||
if ($resumeAttemptCount -ge $maxResumeAttempts) {
|
||||
Write-SupervisorLog "Agent resume abandoned after $resumeAttemptCount attempts"
|
||||
Archive-RestartRequest $pendingRequest
|
||||
$pendingRequest = $null
|
||||
$resumeAttemptCount = 0
|
||||
}
|
||||
else {
|
||||
$nextResumeAttempt = (Get-Date).AddSeconds(3)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -188,6 +214,7 @@ while ($true) {
|
||||
$candidate = Get-RestartRequest
|
||||
if ($candidate) {
|
||||
$pendingRequest = $candidate
|
||||
$resumeAttemptCount = 0
|
||||
Write-SupervisorLog "restart request $($candidate.requestId) detected"
|
||||
Stop-ProcessTree $process
|
||||
break
|
||||
|
||||
+146
-89
@@ -1,123 +1,180 @@
|
||||
import { stat } from "fs/promises";
|
||||
import { resolve } from "path";
|
||||
import { createAgentSessionServices, getAgentDir, type SettingsManager } from "@earendil-works/pi-coding-agent";
|
||||
import {
|
||||
createAgentSessionServices,
|
||||
getAgentDir,
|
||||
type SettingsManager,
|
||||
} from "@earendil-works/pi-coding-agent";
|
||||
import { getSupportedThinkingLevels } from "@earendil-works/pi-ai";
|
||||
import { loadModelsWithCache, withModelRuntimeError, type ModelsData } from "@/lib/models-cache";
|
||||
import { getAllowedFileRoots, isExistingFilePathAllowed } from "@/lib/file-access";
|
||||
import {
|
||||
loadModelsWithCache,
|
||||
withModelRuntimeError,
|
||||
type ModelsData,
|
||||
} from "@/lib/models-cache";
|
||||
import {
|
||||
getAllowedFileRoots,
|
||||
isExistingFilePathAllowed,
|
||||
} from "@/lib/file-access";
|
||||
import { projectTrustReloadOptions } from "@/lib/project-trust";
|
||||
import { createAppSettingsManager, getAppResourceLoaderOptions } from "@/lib/app-runtime";
|
||||
import {
|
||||
createAppSettingsManager,
|
||||
getAppResourceLoaderOptions,
|
||||
} from "@/lib/app-runtime";
|
||||
|
||||
export const dynamic = "force-dynamic";
|
||||
|
||||
const modelNameCollator = new Intl.Collator(undefined, { numeric: true, sensitivity: "base" });
|
||||
const modelNameCollator = new Intl.Collator(undefined, {
|
||||
numeric: true,
|
||||
sensitivity: "base",
|
||||
});
|
||||
|
||||
function compareModelEntries(
|
||||
a: { id: string; name: string; provider: string },
|
||||
b: { id: string; name: string; provider: string }
|
||||
a: { id: string; name: string; provider: string },
|
||||
b: { id: string; name: string; provider: string },
|
||||
): number {
|
||||
return modelNameCollator.compare(a.name || a.id, b.name || b.id)
|
||||
|| modelNameCollator.compare(a.provider, b.provider)
|
||||
|| modelNameCollator.compare(a.id, b.id);
|
||||
return (
|
||||
modelNameCollator.compare(a.name || a.id, b.name || b.id) ||
|
||||
modelNameCollator.compare(a.provider, b.provider) ||
|
||||
modelNameCollator.compare(a.id, b.id)
|
||||
);
|
||||
}
|
||||
|
||||
const THINKING_SUFFIXES = new Set(["off", "minimal", "low", "medium", "high", "xhigh", "max"]);
|
||||
const THINKING_SUFFIXES = new Set([
|
||||
"off",
|
||||
"minimal",
|
||||
"low",
|
||||
"medium",
|
||||
"high",
|
||||
"xhigh",
|
||||
"max",
|
||||
]);
|
||||
|
||||
function stripThinkingSuffix(modelRef: string): string {
|
||||
const trimmed = modelRef.trim();
|
||||
const colonIndex = trimmed.lastIndexOf(":");
|
||||
if (colonIndex === -1) return trimmed;
|
||||
const suffix = trimmed.substring(colonIndex + 1);
|
||||
return THINKING_SUFFIXES.has(suffix) ? trimmed.substring(0, colonIndex) : trimmed;
|
||||
const trimmed = modelRef.trim();
|
||||
const colonIndex = trimmed.lastIndexOf(":");
|
||||
if (colonIndex === -1) return trimmed;
|
||||
const suffix = trimmed.substring(colonIndex + 1);
|
||||
return THINKING_SUFFIXES.has(suffix)
|
||||
? trimmed.substring(0, colonIndex)
|
||||
: trimmed;
|
||||
}
|
||||
|
||||
function filterByExactEnabledModels<T extends { id: string; provider: string }>(
|
||||
available: readonly T[],
|
||||
enabledModels: string[] | undefined,
|
||||
available: readonly T[],
|
||||
enabledModels: string[] | undefined,
|
||||
): readonly T[] {
|
||||
if (!enabledModels || enabledModels.length === 0) return available;
|
||||
if (!enabledModels || enabledModels.length === 0) return available;
|
||||
|
||||
const refs = new Set(enabledModels.map(stripThinkingSuffix).filter(Boolean));
|
||||
const visible = available.filter((m) => refs.has(`${m.provider}/${m.id}`) || refs.has(m.id));
|
||||
return visible.length > 0 ? visible : available;
|
||||
const refs = new Set(enabledModels.map(stripThinkingSuffix).filter(Boolean));
|
||||
const visible = available.filter(
|
||||
(m) => refs.has(`${m.provider}/${m.id}`) || refs.has(m.id),
|
||||
);
|
||||
return visible.length > 0 ? visible : available;
|
||||
}
|
||||
|
||||
async function loadModels(cwd: string): Promise<ModelsData> {
|
||||
const nameMap = new Map<string, string>();
|
||||
let modelList: { id: string; name: string; provider: string }[] = [];
|
||||
let defaultModel: { provider: string; modelId: string } | null = null;
|
||||
const thinkingLevels: Record<string, string[]> = {};
|
||||
const thinkingLevelMaps: Record<string, Record<string, string | null>> = {};
|
||||
const nameMap = new Map<string, string>();
|
||||
let modelList: { id: string; name: string; provider: string }[] = [];
|
||||
let defaultModel: { provider: string; modelId: string } | null = null;
|
||||
const thinkingLevels: Record<string, string[]> = {};
|
||||
const thinkingLevelMaps: Record<string, Record<string, string | null>> = {};
|
||||
|
||||
const agentDir = getAgentDir();
|
||||
// Gate untrusted project extensions: enumerating models still imports and
|
||||
// runs a repository's .pi/extensions factories, so honor project trust here
|
||||
// too (see lib/project-trust.ts, #236).
|
||||
const trustReloadOptions = projectTrustReloadOptions(cwd, agentDir);
|
||||
const services = await createAgentSessionServices({
|
||||
cwd,
|
||||
agentDir,
|
||||
settingsManager: createAppSettingsManager(cwd, agentDir),
|
||||
resourceLoaderOptions: getAppResourceLoaderOptions(),
|
||||
...(trustReloadOptions ? { resourceLoaderReloadOptions: trustReloadOptions } : {}),
|
||||
});
|
||||
const available = await services.modelRuntime.getAvailable();
|
||||
const modelError = services.modelRuntime.getError();
|
||||
const settings: SettingsManager = services.settingsManager;
|
||||
const enabledModels = settings.getEnabledModels();
|
||||
const visible = filterByExactEnabledModels(available, enabledModels);
|
||||
modelList = visible.map((m: { id: string; name: string; provider: string }) => ({
|
||||
id: m.id,
|
||||
name: m.name,
|
||||
provider: m.provider,
|
||||
})).sort(compareModelEntries);
|
||||
for (const m of visible) {
|
||||
const key = `${m.provider}:${m.id}`;
|
||||
nameMap.set(key, m.name);
|
||||
thinkingLevels[key] = getSupportedThinkingLevels(m);
|
||||
if (m.thinkingLevelMap) thinkingLevelMaps[key] = m.thinkingLevelMap;
|
||||
}
|
||||
const agentDir = getAgentDir();
|
||||
// Gate untrusted project extensions: enumerating models still imports and
|
||||
// runs a repository's .pi/extensions factories, so honor project trust here
|
||||
// too (see lib/project-trust.ts, #236).
|
||||
const trustReloadOptions = projectTrustReloadOptions(cwd, agentDir);
|
||||
const services = await createAgentSessionServices({
|
||||
cwd,
|
||||
agentDir,
|
||||
settingsManager: createAppSettingsManager(cwd, agentDir),
|
||||
resourceLoaderOptions: getAppResourceLoaderOptions(),
|
||||
...(trustReloadOptions
|
||||
? { resourceLoaderReloadOptions: trustReloadOptions }
|
||||
: {}),
|
||||
});
|
||||
const available = await services.modelRuntime.getAvailable();
|
||||
const modelError = services.modelRuntime.getError();
|
||||
const settings: SettingsManager = services.settingsManager;
|
||||
const enabledModels = settings.getEnabledModels();
|
||||
const visible = filterByExactEnabledModels(available, enabledModels);
|
||||
modelList = visible
|
||||
.map(
|
||||
(m: {
|
||||
id: string;
|
||||
name: string;
|
||||
provider: string;
|
||||
serviceTier?: string;
|
||||
}) => ({
|
||||
id: m.id,
|
||||
name: m.name,
|
||||
provider: m.provider,
|
||||
serviceTier: m.serviceTier,
|
||||
}),
|
||||
)
|
||||
.sort(compareModelEntries);
|
||||
for (const m of visible) {
|
||||
const key = `${m.provider}:${m.id}`;
|
||||
nameMap.set(key, m.name);
|
||||
thinkingLevels[key] = getSupportedThinkingLevels(m);
|
||||
if (m.thinkingLevelMap) thinkingLevelMaps[key] = m.thinkingLevelMap;
|
||||
}
|
||||
|
||||
const provider = settings.getDefaultProvider();
|
||||
const modelId = settings.getDefaultModel();
|
||||
if (provider && modelId && visible.some((m) => m.provider === provider && m.id === modelId)) {
|
||||
defaultModel = { provider, modelId };
|
||||
}
|
||||
const provider = settings.getDefaultProvider();
|
||||
const modelId = settings.getDefaultModel();
|
||||
if (
|
||||
provider &&
|
||||
modelId &&
|
||||
visible.some((m) => m.provider === provider && m.id === modelId)
|
||||
) {
|
||||
defaultModel = { provider, modelId };
|
||||
}
|
||||
|
||||
return withModelRuntimeError(
|
||||
{ models: Object.fromEntries(nameMap), modelList, defaultModel, thinkingLevels, thinkingLevelMaps },
|
||||
modelError,
|
||||
);
|
||||
return withModelRuntimeError(
|
||||
{
|
||||
models: Object.fromEntries(nameMap),
|
||||
modelList,
|
||||
defaultModel,
|
||||
thinkingLevels,
|
||||
thinkingLevelMaps,
|
||||
},
|
||||
modelError,
|
||||
);
|
||||
}
|
||||
|
||||
const EMPTY_MODELS: ModelsData = {
|
||||
models: {},
|
||||
modelList: [],
|
||||
defaultModel: null,
|
||||
thinkingLevels: {},
|
||||
thinkingLevelMaps: {},
|
||||
models: {},
|
||||
modelList: [],
|
||||
defaultModel: null,
|
||||
thinkingLevels: {},
|
||||
thinkingLevelMaps: {},
|
||||
};
|
||||
|
||||
export async function GET(req: Request) {
|
||||
const requestedCwd = new URL(req.url).searchParams.get("cwd") || process.cwd();
|
||||
const cwd = resolve(requestedCwd);
|
||||
const requestedCwd =
|
||||
new URL(req.url).searchParams.get("cwd") || process.cwd();
|
||||
const cwd = resolve(requestedCwd);
|
||||
|
||||
let cwdStat;
|
||||
try {
|
||||
cwdStat = await stat(cwd);
|
||||
} catch {
|
||||
return Response.json({ error: `Directory does not exist: ${cwd}` }, { status: 400 });
|
||||
}
|
||||
if (!cwdStat.isDirectory()) {
|
||||
return Response.json({ error: `Not a directory: ${cwd}` }, { status: 400 });
|
||||
}
|
||||
const allowedRoots = await getAllowedFileRoots();
|
||||
if (!isExistingFilePathAllowed(cwd, allowedRoots)) {
|
||||
return Response.json({ error: "Access denied" }, { status: 403 });
|
||||
}
|
||||
let cwdStat;
|
||||
try {
|
||||
cwdStat = await stat(cwd);
|
||||
} catch {
|
||||
return Response.json(
|
||||
{ error: `Directory does not exist: ${cwd}` },
|
||||
{ status: 400 },
|
||||
);
|
||||
}
|
||||
if (!cwdStat.isDirectory()) {
|
||||
return Response.json({ error: `Not a directory: ${cwd}` }, { status: 400 });
|
||||
}
|
||||
const allowedRoots = await getAllowedFileRoots();
|
||||
if (!isExistingFilePathAllowed(cwd, allowedRoots)) {
|
||||
return Response.json({ error: "Access denied" }, { status: 403 });
|
||||
}
|
||||
|
||||
try {
|
||||
return Response.json(await loadModelsWithCache(cwd, () => loadModels(cwd)));
|
||||
} catch {
|
||||
return Response.json(EMPTY_MODELS);
|
||||
}
|
||||
try {
|
||||
return Response.json(await loadModelsWithCache(cwd, () => loadModels(cwd)));
|
||||
} catch {
|
||||
return Response.json(EMPTY_MODELS);
|
||||
}
|
||||
}
|
||||
|
||||
+3440
-2204
File diff suppressed because it is too large
Load Diff
+2158
-1539
File diff suppressed because it is too large
Load Diff
+70
-56
@@ -1,80 +1,94 @@
|
||||
export interface ModelsData {
|
||||
models: Record<string, string>;
|
||||
modelList: { id: string; name: string; provider: string }[];
|
||||
defaultModel: { provider: string; modelId: string } | null;
|
||||
thinkingLevels: Record<string, string[]>;
|
||||
thinkingLevelMaps: Record<string, Record<string, string | null>>;
|
||||
modelError?: string;
|
||||
models: Record<string, string>;
|
||||
modelList: {
|
||||
id: string;
|
||||
name: string;
|
||||
provider: string;
|
||||
serviceTier?: string;
|
||||
}[];
|
||||
defaultModel: { provider: string; modelId: string } | null;
|
||||
thinkingLevels: Record<string, string[]>;
|
||||
thinkingLevelMaps: Record<string, Record<string, string | null>>;
|
||||
modelError?: string;
|
||||
}
|
||||
|
||||
interface ModelsCacheState {
|
||||
entries: Map<string, { data: ModelsData; expiresAt: number }>;
|
||||
inFlight: Map<string, Promise<ModelsData>>;
|
||||
generation: number;
|
||||
entries: Map<string, { data: ModelsData; expiresAt: number }>;
|
||||
inFlight: Map<string, Promise<ModelsData>>;
|
||||
generation: number;
|
||||
}
|
||||
|
||||
declare global {
|
||||
var __piModelsCacheState: ModelsCacheState | undefined;
|
||||
var __piModelsCacheState: ModelsCacheState | undefined;
|
||||
}
|
||||
|
||||
const MODELS_CACHE_TTL_MS = 60_000;
|
||||
const MAX_MODELS_CACHE_ENTRIES = 32;
|
||||
|
||||
function getModelsCacheState(): ModelsCacheState {
|
||||
if (!globalThis.__piModelsCacheState) {
|
||||
globalThis.__piModelsCacheState = {
|
||||
entries: new Map(),
|
||||
inFlight: new Map(),
|
||||
generation: 0,
|
||||
};
|
||||
}
|
||||
return globalThis.__piModelsCacheState;
|
||||
if (!globalThis.__piModelsCacheState) {
|
||||
globalThis.__piModelsCacheState = {
|
||||
entries: new Map(),
|
||||
inFlight: new Map(),
|
||||
generation: 0,
|
||||
};
|
||||
}
|
||||
return globalThis.__piModelsCacheState;
|
||||
}
|
||||
|
||||
export function invalidateModelsCache(): void {
|
||||
const state = getModelsCacheState();
|
||||
state.generation += 1;
|
||||
state.entries.clear();
|
||||
state.inFlight.clear();
|
||||
const state = getModelsCacheState();
|
||||
state.generation += 1;
|
||||
state.entries.clear();
|
||||
state.inFlight.clear();
|
||||
}
|
||||
|
||||
export function withModelRuntimeError(data: ModelsData, modelError: string | undefined): ModelsData {
|
||||
return modelError ? { ...data, modelError } : data;
|
||||
export function withModelRuntimeError(
|
||||
data: ModelsData,
|
||||
modelError: string | undefined,
|
||||
): ModelsData {
|
||||
return modelError ? { ...data, modelError } : data;
|
||||
}
|
||||
|
||||
export function loadModelsWithCache(cwd: string, loader: () => Promise<ModelsData>): Promise<ModelsData> {
|
||||
const state = getModelsCacheState();
|
||||
const cached = state.entries.get(cwd);
|
||||
if (cached) {
|
||||
if (cached.expiresAt > Date.now()) return Promise.resolve(cached.data);
|
||||
state.entries.delete(cwd);
|
||||
}
|
||||
export function loadModelsWithCache(
|
||||
cwd: string,
|
||||
loader: () => Promise<ModelsData>,
|
||||
): Promise<ModelsData> {
|
||||
const state = getModelsCacheState();
|
||||
const cached = state.entries.get(cwd);
|
||||
if (cached) {
|
||||
if (cached.expiresAt > Date.now()) return Promise.resolve(cached.data);
|
||||
state.entries.delete(cwd);
|
||||
}
|
||||
|
||||
const existingLoad = state.inFlight.get(cwd);
|
||||
if (existingLoad) return existingLoad;
|
||||
const existingLoad = state.inFlight.get(cwd);
|
||||
if (existingLoad) return existingLoad;
|
||||
|
||||
const generation = state.generation;
|
||||
const loadPromise: Promise<ModelsData> = Promise.resolve()
|
||||
.then(loader)
|
||||
.then((data) => {
|
||||
if (state.generation === generation && state.inFlight.get(cwd) === loadPromise) {
|
||||
const now = Date.now();
|
||||
for (const [key, entry] of state.entries) {
|
||||
if (entry.expiresAt <= now) state.entries.delete(key);
|
||||
}
|
||||
while (state.entries.size >= MAX_MODELS_CACHE_ENTRIES) {
|
||||
const oldestKey = state.entries.keys().next().value;
|
||||
if (oldestKey === undefined) break;
|
||||
state.entries.delete(oldestKey);
|
||||
}
|
||||
state.entries.set(cwd, { data, expiresAt: now + MODELS_CACHE_TTL_MS });
|
||||
}
|
||||
return data;
|
||||
})
|
||||
.finally(() => {
|
||||
if (state.inFlight.get(cwd) === loadPromise) state.inFlight.delete(cwd);
|
||||
});
|
||||
const generation = state.generation;
|
||||
const loadPromise: Promise<ModelsData> = Promise.resolve()
|
||||
.then(loader)
|
||||
.then((data) => {
|
||||
if (
|
||||
state.generation === generation &&
|
||||
state.inFlight.get(cwd) === loadPromise
|
||||
) {
|
||||
const now = Date.now();
|
||||
for (const [key, entry] of state.entries) {
|
||||
if (entry.expiresAt <= now) state.entries.delete(key);
|
||||
}
|
||||
while (state.entries.size >= MAX_MODELS_CACHE_ENTRIES) {
|
||||
const oldestKey = state.entries.keys().next().value;
|
||||
if (oldestKey === undefined) break;
|
||||
state.entries.delete(oldestKey);
|
||||
}
|
||||
state.entries.set(cwd, { data, expiresAt: now + MODELS_CACHE_TTL_MS });
|
||||
}
|
||||
return data;
|
||||
})
|
||||
.finally(() => {
|
||||
if (state.inFlight.get(cwd) === loadPromise) state.inFlight.delete(cwd);
|
||||
});
|
||||
|
||||
state.inFlight.set(cwd, loadPromise);
|
||||
return loadPromise;
|
||||
state.inFlight.set(cwd, loadPromise);
|
||||
return loadPromise;
|
||||
}
|
||||
|
||||
@@ -41,6 +41,7 @@ export function buildBaseOptions(
|
||||
maxRetries: options?.maxRetries,
|
||||
maxRetryDelayMs: options?.maxRetryDelayMs,
|
||||
metadata: options?.metadata,
|
||||
serviceTier: options?.serviceTier,
|
||||
env: options?.env,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -113,6 +113,13 @@ export interface ProviderResponse {
|
||||
headers: Record<string, string>;
|
||||
}
|
||||
|
||||
/**
|
||||
* Request-level service tier for providers that support it.
|
||||
* OpenAI Codex / Responses: "priority" is fast mode (higher cost, lower latency), "flex" is the cheaper tier.
|
||||
* See https://developers.openai.com/codex/speed for Codex fast mode semantics.
|
||||
*/
|
||||
export type ServiceTier = "auto" | "default" | "flex" | "priority" | "scale";
|
||||
|
||||
export interface StreamOptions {
|
||||
temperature?: number;
|
||||
maxTokens?: number;
|
||||
@@ -189,6 +196,12 @@ export interface StreamOptions {
|
||||
* For example, Anthropic uses `user_id` for abuse tracking and rate limiting.
|
||||
*/
|
||||
metadata?: Record<string, unknown>;
|
||||
/**
|
||||
* Request-level service tier. Providers that understand it (OpenAI Codex,
|
||||
* OpenAI Responses) send it as `service_tier`; other providers ignore it.
|
||||
* "priority" is Codex fast mode (1.5x speed, ~2x cost).
|
||||
*/
|
||||
serviceTier?: ServiceTier | null;
|
||||
/**
|
||||
* Provider-scoped environment values. These take precedence over process.env for
|
||||
* provider configuration such as regional settings, endpoint placeholders, and
|
||||
@@ -773,6 +786,12 @@ export interface Model<TApi extends Api> {
|
||||
contextWindow: number;
|
||||
maxTokens: number;
|
||||
headers?: Record<string, string>;
|
||||
/**
|
||||
* Optional default request-level service tier for this model (e.g. "priority"
|
||||
* for Codex fast mode). Configured via models.json; forwarded to providers
|
||||
* that support `service_tier`.
|
||||
*/
|
||||
serviceTier?: ServiceTier;
|
||||
/** Compatibility overrides for OpenAI-compatible APIs. If not set, auto-detected from baseUrl. */
|
||||
compat?: TApi extends "openai-completions"
|
||||
? OpenAICompletionsCompat
|
||||
|
||||
@@ -151,6 +151,13 @@ const ModelCostSchema = Type.Object({
|
||||
tiers: Type.Optional(Type.Array(ModelCostTierSchema)),
|
||||
});
|
||||
|
||||
const ServiceTierSchema = Type.Union([
|
||||
Type.Literal("auto"),
|
||||
Type.Literal("default"),
|
||||
Type.Literal("priority"),
|
||||
Type.Literal("flex"),
|
||||
]);
|
||||
|
||||
const ModelDefinitionSchema = Type.Object({
|
||||
id: Type.String({ minLength: 1 }),
|
||||
name: Type.Optional(Type.String({ minLength: 1 })),
|
||||
@@ -163,6 +170,7 @@ const ModelDefinitionSchema = Type.Object({
|
||||
contextWindow: Type.Optional(Type.Number()),
|
||||
maxTokens: Type.Optional(Type.Number()),
|
||||
headers: Type.Optional(Type.Record(Type.String(), Type.String())),
|
||||
serviceTier: Type.Optional(ServiceTierSchema),
|
||||
compat: Type.Optional(ProviderCompatSchema),
|
||||
});
|
||||
|
||||
@@ -183,6 +191,7 @@ const ModelOverrideSchema = Type.Object({
|
||||
contextWindow: Type.Optional(Type.Number()),
|
||||
maxTokens: Type.Optional(Type.Number()),
|
||||
headers: Type.Optional(Type.Record(Type.String(), Type.String())),
|
||||
serviceTier: Type.Optional(ServiceTierSchema),
|
||||
compat: Type.Optional(ProviderCompatSchema),
|
||||
});
|
||||
|
||||
|
||||
@@ -117,6 +117,7 @@ function applyModelOverride(model: Model<Api>, override: ModelsJsonModelOverride
|
||||
: model.cost,
|
||||
contextWindow: override.contextWindow ?? model.contextWindow,
|
||||
maxTokens: override.maxTokens ?? model.maxTokens,
|
||||
serviceTier: override.serviceTier ?? model.serviceTier,
|
||||
compat: mergeCompat(model.compat, override.compat),
|
||||
};
|
||||
}
|
||||
@@ -154,6 +155,7 @@ function modelFromJson(
|
||||
contextWindow: definition.contextWindow ?? 128000,
|
||||
maxTokens: definition.maxTokens ?? 16384,
|
||||
headers: undefined,
|
||||
serviceTier: definition.serviceTier,
|
||||
compat: mergeCompat(providerConfig.compat, definition.compat),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -309,8 +309,13 @@ export async function createAgentSession(options: CreateAgentSessionOptions = {}
|
||||
const websocketConnectTimeoutMs =
|
||||
options?.websocketConnectTimeoutMs ?? settingsManager.getWebSocketConnectTimeoutMs();
|
||||
const headerRunner = extensionRunnerRef.current;
|
||||
// Model-level service tier (models.json `serviceTier`, e.g. "priority" for
|
||||
// Codex fast mode) is forwarded as request `service_tier`. Providers that
|
||||
// don't understand it ignore the option.
|
||||
const serviceTier = model.serviceTier;
|
||||
return modelRuntime.streamSimple(model, context, {
|
||||
...options,
|
||||
...(serviceTier ? { serviceTier } : {}),
|
||||
timeoutMs,
|
||||
websocketConnectTimeoutMs,
|
||||
maxRetries: options?.maxRetries ?? providerRetrySettings.maxRetries,
|
||||
|
||||
+54
-2
@@ -1,7 +1,7 @@
|
||||
import { spawnSync } from "node:child_process";
|
||||
import { copyFileSync, existsSync, mkdirSync } from "node:fs";
|
||||
import { copyFileSync, existsSync, lstatSync, mkdirSync, symlinkSync } from "node:fs";
|
||||
import { platform } from "node:os";
|
||||
import { dirname, join, resolve } from "node:path";
|
||||
import { delimiter, dirname, join, resolve } from "node:path";
|
||||
import { loadEnvFile } from "node:process";
|
||||
import { fileURLToPath } from "node:url";
|
||||
|
||||
@@ -55,6 +55,56 @@ const seedFiles = [
|
||||
];
|
||||
|
||||
const persistedWindowsEnvironmentKeys = ["SEARXNG_TOKEN", "SEARXNG_URL"];
|
||||
const piWebBinDir = join(rootDir, "pi-web", "node_modules", ".bin");
|
||||
const piCodingAgentPackageDir = join(
|
||||
rootDir,
|
||||
"pi-web",
|
||||
"node_modules",
|
||||
"@earendil-works",
|
||||
"pi-coding-agent",
|
||||
);
|
||||
const managedPeerPackageDir = join(
|
||||
agentDir,
|
||||
"npm",
|
||||
"node_modules",
|
||||
"@earendil-works",
|
||||
"pi-coding-agent",
|
||||
);
|
||||
|
||||
function ensurePiSubagentRuntime() {
|
||||
// pi-subagents resolves the host CLI through its optional coding-agent peer.
|
||||
// The managed profile is installed separately from pi-web, so expose the
|
||||
// existing local package instead of installing a second coding-agent copy.
|
||||
if (!existsSync(piCodingAgentPackageDir)) return;
|
||||
|
||||
mkdirSync(dirname(managedPeerPackageDir), { recursive: true });
|
||||
if (existsSync(managedPeerPackageDir)) return;
|
||||
try {
|
||||
if (lstatSync(managedPeerPackageDir)) return;
|
||||
} catch {
|
||||
// The peer link is absent; create it below.
|
||||
}
|
||||
|
||||
try {
|
||||
symlinkSync(
|
||||
piCodingAgentPackageDir,
|
||||
managedPeerPackageDir,
|
||||
platform() === "win32" ? "junction" : "dir",
|
||||
);
|
||||
} catch (error) {
|
||||
const message = error instanceof Error ? error.message : String(error);
|
||||
console.error(`[subagents] could not link the local Pi CLI package: ${message}`);
|
||||
}
|
||||
}
|
||||
|
||||
function prependPiWebBinToPath(baseEnv) {
|
||||
if (!existsSync(piWebBinDir)) return baseEnv.PATH ?? baseEnv.Path;
|
||||
const pathKey = baseEnv.PATH !== undefined || baseEnv.Path === undefined ? "PATH" : "Path";
|
||||
const currentPath = baseEnv[pathKey] ?? "";
|
||||
const entries = currentPath.split(delimiter).filter(Boolean);
|
||||
if (!entries.includes(piWebBinDir)) entries.unshift(piWebBinDir);
|
||||
return entries.join(delimiter);
|
||||
}
|
||||
|
||||
function readPersistedWindowsEnvironment(baseEnv) {
|
||||
if (platform() !== "win32") return {};
|
||||
@@ -112,10 +162,12 @@ export function ensureProfile({ quiet = false } = {}) {
|
||||
|
||||
export function managedEnvironment(baseEnv = process.env) {
|
||||
ensureProfile({ quiet: true });
|
||||
ensurePiSubagentRuntime();
|
||||
const persistedEnvironment = readPersistedWindowsEnvironment(baseEnv);
|
||||
return {
|
||||
...baseEnv,
|
||||
...persistedEnvironment,
|
||||
PATH: prependPiWebBinToPath({ ...baseEnv, ...persistedEnvironment }),
|
||||
PI_AGENT_MANAGED_RUNTIME: "1",
|
||||
PI_AGENT_APP_ROOT: rootDir,
|
||||
PI_AGENT_DATA_DIR: dataDir,
|
||||
|
||||
Reference in New Issue
Block a user