mirror of
https://github.com/luckyyzh/pi-agent-integrated.git
synced 2026-10-03 02:59:35 +00:00
feat(vision): cache key include prompt, parallel multi-image, compressed hook prompt, PNG compression, retry logic
This commit is contained in:
@@ -12,11 +12,13 @@ export interface Base64ImageAttachment {
|
|||||||
}
|
}
|
||||||
|
|
||||||
function isBase64DataChar(code: number): boolean {
|
function isBase64DataChar(code: number): boolean {
|
||||||
return (code >= 0x41 && code <= 0x5a)
|
return (
|
||||||
|| (code >= 0x61 && code <= 0x7a)
|
(code >= 0x41 && code <= 0x5a) ||
|
||||||
|| (code >= 0x30 && code <= 0x39)
|
(code >= 0x61 && code <= 0x7a) ||
|
||||||
|| code === 0x2b
|
(code >= 0x30 && code <= 0x39) ||
|
||||||
|| code === 0x2f;
|
code === 0x2b ||
|
||||||
|
code === 0x2f
|
||||||
|
);
|
||||||
}
|
}
|
||||||
|
|
||||||
export function getBase64DecodedByteLength(data: string): number | null {
|
export function getBase64DecodedByteLength(data: string): number | null {
|
||||||
@@ -32,10 +34,16 @@ export function getBase64DecodedByteLength(data: string): number | null {
|
|||||||
return (data.length / 4) * 3 - padding;
|
return (data.length / 4) * 3 - padding;
|
||||||
}
|
}
|
||||||
|
|
||||||
export function isBase64ImageWithinLimits(value: unknown): value is Base64ImageAttachment {
|
export function isBase64ImageWithinLimits(
|
||||||
|
value: unknown,
|
||||||
|
): value is Base64ImageAttachment {
|
||||||
if (!value || typeof value !== "object") return false;
|
if (!value || typeof value !== "object") return false;
|
||||||
const image = value as Partial<Base64ImageAttachment>;
|
const image = value as Partial<Base64ImageAttachment>;
|
||||||
if (typeof image.data !== "string" || typeof image.mimeType !== "string" || !image.mimeType.startsWith("image/")) {
|
if (
|
||||||
|
typeof image.data !== "string" ||
|
||||||
|
typeof image.mimeType !== "string" ||
|
||||||
|
!image.mimeType.startsWith("image/")
|
||||||
|
) {
|
||||||
return false;
|
return false;
|
||||||
}
|
}
|
||||||
const bytes = getBase64DecodedByteLength(image.data);
|
const bytes = getBase64DecodedByteLength(image.data);
|
||||||
@@ -43,12 +51,19 @@ export function isBase64ImageWithinLimits(value: unknown): value is Base64ImageA
|
|||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 上传前压缩:仅 JPEG 且长边超过 MAX_IMAGE_EDGE_PX 时缩放重编码(EXIF 方向由
|
* 上传前压缩:JPEG 和 PNG 且长边超过 MAX_IMAGE_EDGE_PX 时缩放并统一转 JPEG。
|
||||||
* createImageBitmap 默认 from-image 纠正)。PNG/WebP/GIF 原样返回(无损/动画场景)。
|
* PNG 截图(代码编辑器/UI)通常无透明通道,转 JPEG 体积大幅减小。
|
||||||
|
* WebP/GIF 原样返回(WebP 已经高效;GIF 可能是动画)。
|
||||||
* 重编码无收益(更小或失败)时退回原文件。
|
* 重编码无收益(更小或失败)时退回原文件。
|
||||||
*/
|
*/
|
||||||
export async function compressImageFile(file: File): Promise<File> {
|
export async function compressImageFile(file: File): Promise<File> {
|
||||||
if (file.type !== "image/jpeg") return file;
|
// 仅压缩 JPEG 和 PNG(最常见的截图/照片格式)
|
||||||
|
if (
|
||||||
|
!file.type ||
|
||||||
|
!(file.type.startsWith("image/jpeg") || file.type.startsWith("image/png"))
|
||||||
|
) {
|
||||||
|
return file;
|
||||||
|
}
|
||||||
let bitmap: ImageBitmap;
|
let bitmap: ImageBitmap;
|
||||||
try {
|
try {
|
||||||
bitmap = await createImageBitmap(file);
|
bitmap = await createImageBitmap(file);
|
||||||
@@ -68,10 +83,12 @@ export async function compressImageFile(file: File): Promise<File> {
|
|||||||
if (!ctx) return file;
|
if (!ctx) return file;
|
||||||
ctx.drawImage(bitmap, 0, 0, width, height);
|
ctx.drawImage(bitmap, 0, 0, width, height);
|
||||||
const blob = await new Promise<Blob | null>((resolve) =>
|
const blob = await new Promise<Blob | null>((resolve) =>
|
||||||
canvas.toBlob(resolve, "image/jpeg", IMAGE_JPEG_QUALITY)
|
canvas.toBlob(resolve, "image/jpeg", IMAGE_JPEG_QUALITY),
|
||||||
);
|
);
|
||||||
if (!blob || blob.size >= file.size) return file; // 重编码无收益则用原文件
|
if (!blob || blob.size >= file.size) return file; // 重编码无收益则用原文件
|
||||||
return new File([blob], file.name.replace(/\.\w+$/, ".jpg"), { type: "image/jpeg" });
|
return new File([blob], file.name.replace(/\.\w+$/, ".jpg"), {
|
||||||
|
type: "image/jpeg",
|
||||||
|
});
|
||||||
} finally {
|
} finally {
|
||||||
bitmap.close();
|
bitmap.close();
|
||||||
}
|
}
|
||||||
@@ -85,7 +102,11 @@ export function validateAgentImages(value: unknown): string | null {
|
|||||||
return `A message can include at most ${MAX_ATTACHED_IMAGES} images`;
|
return `A message can include at most ${MAX_ATTACHED_IMAGES} images`;
|
||||||
}
|
}
|
||||||
for (const image of value) {
|
for (const image of value) {
|
||||||
if (!image || typeof image !== "object" || (image as { type?: unknown }).type !== "image") {
|
if (
|
||||||
|
!image ||
|
||||||
|
typeof image !== "object" ||
|
||||||
|
(image as { type?: unknown }).type !== "image"
|
||||||
|
) {
|
||||||
return "Each attachment must be an image";
|
return "Each attachment must be an image";
|
||||||
}
|
}
|
||||||
if (!isBase64ImageWithinLimits(image)) {
|
if (!isBase64ImageWithinLimits(image)) {
|
||||||
|
|||||||
@@ -43,10 +43,10 @@ const DEFAULT_PROMPT = [
|
|||||||
* the main model's context every turn, so verbosity costs both latency and
|
* the main model's context every turn, so verbosity costs both latency and
|
||||||
* tokens. The `vision` tool keeps the detailed DEFAULT_PROMPT above. */
|
* tokens. The `vision` tool keeps the detailed DEFAULT_PROMPT above. */
|
||||||
const HOOK_PROMPT = [
|
const HOOK_PROMPT = [
|
||||||
"用中文简要描述这张图片(主模型依赖此转录理解图片,需准确但精简):",
|
`用中文简要描述此图(主模型完全依赖此转录):`,
|
||||||
"1. 所有可见文字:按阅读顺序转录(含标签、数字、按钮、代码),无文字则写“无”。",
|
`1. 可见文字逐字转录(标签/代码/数字),无则写无文字。`,
|
||||||
"2. 图片内容:主体、场景、布局,2-3 句。",
|
`2. 画面内容 1-2 句。`,
|
||||||
"总长约 100 字,不要分节模板,直接输出。",
|
`控制在 50 字内。`,
|
||||||
].join("\n");
|
].join("\n");
|
||||||
|
|
||||||
const visionParams = Type.Object({
|
const visionParams = Type.Object({
|
||||||
@@ -420,10 +420,10 @@ function dataUrlToLoadedImage(url: string): LoadedImage | null {
|
|||||||
return { base64: match[2], mime: match[1] };
|
return { base64: match[2], mime: match[1] };
|
||||||
}
|
}
|
||||||
|
|
||||||
function imageCacheKey(image: LoadedImage): string {
|
function imageCacheKey(image: LoadedImage, prompt: string): string {
|
||||||
// Whole-image hash: PNG/JPG headers repeat for same dimensions, so a short
|
// Hash image + prompt so different prompts (hook vs tool) don't share cache.
|
||||||
// prefix would collide across different images of the same size.
|
// PNG/JPG headers repeat for same dimensions, so full base64 is needed.
|
||||||
return createHash("md5").update(image.base64).digest("hex");
|
return createHash("md5").update(image.base64).update(prompt).digest("hex");
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
/**
|
||||||
@@ -438,7 +438,7 @@ async function getImageDescription(
|
|||||||
prompt: string,
|
prompt: string,
|
||||||
signal: AbortSignal | undefined,
|
signal: AbortSignal | undefined,
|
||||||
): Promise<string> {
|
): Promise<string> {
|
||||||
const key = imageCacheKey(image);
|
const key = imageCacheKey(image, prompt);
|
||||||
await loadDescriptionCache();
|
await loadDescriptionCache();
|
||||||
const cached = IMAGE_DESCRIPTION_CACHE.get(key);
|
const cached = IMAGE_DESCRIPTION_CACHE.get(key);
|
||||||
if (cached) return cached;
|
if (cached) return cached;
|
||||||
@@ -517,16 +517,26 @@ export default function visionExtension(pi: ExtensionAPI) {
|
|||||||
if (!Array.isArray(messages) || messages.length === 0) return;
|
if (!Array.isArray(messages) || messages.length === 0) return;
|
||||||
if (!isTextOnlyModel(ctx.model)) return; // vision-capable models pass through untouched
|
if (!isTextOnlyModel(ctx.model)) return; // vision-capable models pass through untouched
|
||||||
|
|
||||||
|
// Phase 1: collect image parts, replace with placeholders
|
||||||
// Replace each image in place with its stable transcription. Do not move
|
// Replace each image in place with its stable transcription. Do not move
|
||||||
// historical descriptions to a later message: DeepSeek caches exact prompt
|
// historical descriptions to a later message: DeepSeek caches exact prompt
|
||||||
// prefixes, so rewriting an old image message invalidates everything after it.
|
// prefixes, so rewriting an old image message invalidates everything after it.
|
||||||
let dirty = false;
|
interface PendingImage {
|
||||||
|
img: LoadedImage;
|
||||||
|
num: number;
|
||||||
|
msgIndex: number;
|
||||||
|
partIndex: number;
|
||||||
|
}
|
||||||
|
const pendingImages: PendingImage[] = [];
|
||||||
let imageNumber = 0;
|
let imageNumber = 0;
|
||||||
for (const msg of messages) {
|
let dirty = false;
|
||||||
|
|
||||||
|
for (let msgIdx = 0; msgIdx < messages.length; msgIdx++) {
|
||||||
|
const msg = messages[msgIdx];
|
||||||
if (msg?.role !== "user" || !Array.isArray(msg.content)) continue;
|
if (msg?.role !== "user" || !Array.isArray(msg.content)) continue;
|
||||||
const newContent: unknown[] = [];
|
const newContent: unknown[] = [];
|
||||||
let messageDirty = false;
|
for (let partIdx = 0; partIdx < msg.content.length; partIdx++) {
|
||||||
for (const part of msg.content) {
|
const part = msg.content[partIdx];
|
||||||
const p = part as { type?: string; image_url?: { url?: string } };
|
const p = part as { type?: string; image_url?: { url?: string } };
|
||||||
if (p?.type !== "image_url" || typeof p.image_url?.url !== "string") {
|
if (p?.type !== "image_url" || typeof p.image_url?.url !== "string") {
|
||||||
newContent.push(part);
|
newContent.push(part);
|
||||||
@@ -540,29 +550,60 @@ export default function visionExtension(pi: ExtensionAPI) {
|
|||||||
}
|
}
|
||||||
|
|
||||||
imageNumber += 1;
|
imageNumber += 1;
|
||||||
let description: string;
|
pendingImages.push({
|
||||||
|
img,
|
||||||
|
num: imageNumber,
|
||||||
|
msgIndex: msgIdx,
|
||||||
|
partIndex: partIdx,
|
||||||
|
});
|
||||||
|
newContent.push(part); // placeholder, replaced after transcription
|
||||||
|
dirty = true;
|
||||||
|
}
|
||||||
|
if (newContent !== msg.content) {
|
||||||
|
msg.content = newContent;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (!dirty) return;
|
||||||
|
|
||||||
|
// Phase 2: transcribe all images in parallel
|
||||||
|
const results = await Promise.all(
|
||||||
|
pendingImages.map(async ({ img, num, msgIndex, partIndex }) => {
|
||||||
|
for (let attempt = 1; attempt <= 2; attempt++) {
|
||||||
try {
|
try {
|
||||||
description = await getImageDescription(
|
const desc = await getImageDescription(
|
||||||
img,
|
img,
|
||||||
HOOK_PROMPT,
|
HOOK_PROMPT,
|
||||||
ctx.signal,
|
ctx.signal,
|
||||||
);
|
);
|
||||||
} catch {
|
return { msgIndex, partIndex, num, desc };
|
||||||
// Keep failures deterministic. Variable error details in the prompt
|
} catch (err) {
|
||||||
// would themselves invalidate the cache on every retry.
|
console.warn(
|
||||||
description = "[图片转录失败]";
|
`vision: transcription attempt ${attempt} failed:`,
|
||||||
|
err instanceof Error ? err.message : String(err),
|
||||||
|
);
|
||||||
|
if (attempt === 2) {
|
||||||
|
// Deterministic fallback — variable errors would invalidate cache
|
||||||
|
return { msgIndex, partIndex, num, desc: "[图片转录失败]" };
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}),
|
||||||
|
);
|
||||||
|
|
||||||
|
// Phase 3: replace placeholders with descriptions
|
||||||
|
for (const r of results) {
|
||||||
|
if (!r) continue;
|
||||||
|
const { msgIndex, partIndex, num, desc } = r;
|
||||||
|
const msg = messages[msgIndex];
|
||||||
|
if (msg && Array.isArray(msg.content)) {
|
||||||
|
msg.content[partIndex] = {
|
||||||
|
type: "text",
|
||||||
|
text: `\n\n【图片${num}】\n${desc}\n\n`,
|
||||||
|
};
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
newContent.push({
|
|
||||||
type: "text",
|
|
||||||
text: `\n\n【图片${imageNumber}】\n${description}\n\n`,
|
|
||||||
});
|
|
||||||
messageDirty = true;
|
|
||||||
dirty = true;
|
|
||||||
}
|
|
||||||
if (messageDirty) msg.content = newContent;
|
|
||||||
}
|
|
||||||
if (!dirty) return;
|
|
||||||
return payload;
|
return payload;
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in New Issue
Block a user