From 8f84ad9c395dbb0f2ed3ed7c7754a7b8d5d8a01e Mon Sep 17 00:00:00 2001
From: =?UTF-8?q?AI=E4=B8=8D=E6=AD=A2=E8=AF=AD?=
<12096460+jnMetaCode@users.noreply.github.com>
Date: Tue, 22 Sep 2026 19:24:13 +0800
Subject: [PATCH] =?UTF-8?q?fix(rank):=20=E7=9C=8B=E5=9B=BE=E6=8A=8A?=
=?UTF-8?q?=E5=85=B3=E5=80=99=E9=80=89=E4=BB=8E=201=20=E8=B5=B7=E7=BC=96?=
=?UTF-8?q?=E5=8F=B7=EF=BC=88=E4=B8=8E=E5=9B=BE=E7=89=87=E5=8D=A0=E4=BD=8D?=
=?UTF-8?q?=E4=B8=80=E8=87=B4=EF=BC=89=EF=BC=8C=E8=A7=A3=E6=9E=90=E5=85=BC?=
=?UTF-8?q?=E5=AE=B9=200=20=E8=B5=B7=EF=BC=9B=E7=9C=8B=E5=9B=BE=E6=8A=8A?=
=?UTF-8?q?=E5=85=B3=E5=8F=AF=E9=80=89=E6=9C=AC=E6=9C=BA=20Ollama=20?=
=?UTF-8?q?=E8=A7=86=E8=A7=89=E6=A8=A1=E5=9E=8B?=
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
提示词里"候选 0:[图片1]"两套编号打架,本机 3B 视觉模型 3 选 1 只对 1/5(第 1 条永远算 0 分);统一后 4/5。
ollama 进看图把关的供应商列表,型号只列本机能看图的(需引擎 ≥ 0.19.3 把图真发出去;旧引擎验证会如实报看不了图)。
---
CHANGELOG.md | 8 ++++++++
server/kaipian.mjs | 4 +++-
src/kaipian/Kaipian.tsx | 3 ++-
src/kaipian/i18n.ts | 1 +
src/local/ollama.mjs | 9 ++++++---
src/sources/rank.mjs | 15 ++++++++++++---
tests/ollama.test.mjs | 5 ++++-
tests/rank.test.mjs | 10 ++++++++++
8 files changed, 46 insertions(+), 9 deletions(-)
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 2497717..28a07b9 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -2,6 +2,14 @@
## [Unreleased]
+- **看图把关的候选编号打架**:提示词里写"候选 0:[图片1]"——候选从 0 起、引擎留的图片占位从 1 起,模型回的 `i` 按哪个数没法确定。
+ 本机 3B 视觉模型真机 3 选 1 只对 1/5:第 1 条候选永远被算成 0 分。现在候选也从 1 起编号,解析兼容仍按 0 起数的模型(出现 i=0 就按 0 起)。
+ 同一模型同一批图:1/5 → 4/5(剩下那条是把猫雕像当真猫,模型的极限)。
+- **看图把关可以走本机 Ollama 的视觉模型**(qwen2.5vl / llava / minicpm-v…;型号只列名字带 vl / llava / vision 的),至此写脚本 + 看图 + 出图
+ 三步都能 0 key 全本机。**需要引擎 ≥ 0.19.3**(AO 的 ollama 连接器以前剥掉图片,已改为走它的 `images` 字段并把图片算进 num_ctx——
+ 真机 12 张缩略图只按文本估 4096,Ollama 直接 400);旧引擎下"验证并开启"真发一张红图,会如实报看不了图。
+ 真机 3B:一镜从"本机出图 56 秒"变成"本机看图选中素材 44 秒",但它给了一张满是文字的示意图 7 分(提示词要求 ≤ 2)——3B 判断力有限,界面提示建议 7B。
+
- **MCP server(`openshorts mcp`)**:让 Claude Code / Cursor 等 agent 直接"话题 → 成片"。五个工具:`create_video`、`render_project`、`job_status`、
`list_projects`、`doctor`。出片 1–25 分钟远超工具调用超时,所以任务化:立刻回任务号,状态落盘在 `~/.openshorts/mcp-jobs/`;
server 被客户端重启后旧任务如实报 interrupted,不让 agent 永远等;子进程不 detach,server 退出一并收掉。任务跑的是 CLI 的
diff --git a/server/kaipian.mjs b/server/kaipian.mjs
index f489e86..9798649 100644
--- a/server/kaipian.mjs
+++ b/server/kaipian.mjs
@@ -278,7 +278,9 @@ kaipian.get('/providers/text', async (_req, res, next) => {
}));
// 本机 Ollama 不在 AO 的 API 供应商表里(它不要 key),单独探测后排在最前:不花钱的放前面
const ol = await ollamaStatus();
- list.unshift({ id: 'ollama', local: true, running: ol.running, baseUrl: ol.baseUrl, hasKey: ol.running && ol.models.length > 0, fromEnv: false, envKey: null, models: ol.models, visionModels: [], vision: false });
+ // 看图把关也能走本机:只列名字带 vl / llava / vision 的型号(需要引擎 ≥ 0.19.3 才真的把图发给 Ollama;旧引擎会剥图,
+ // "验证并开启"那一步真发一张红图,答不出来就会如实报"这个模型看不了图")
+ list.unshift({ id: 'ollama', local: true, running: ol.running, baseUrl: ol.baseUrl, hasKey: ol.running && ol.models.length > 0, fromEnv: false, envKey: null, models: ol.models, visionModels: ol.visionModels ?? [], vision: ol.running && (ol.visionModels ?? []).length > 0 });
const c = readConfig();
res.json({ providers: list, vision: c.vision ?? { provider: '', model: '' }, text: c.text ?? { provider: '', model: '' } });
} catch (e) { next(e); }
diff --git a/src/kaipian/Kaipian.tsx b/src/kaipian/Kaipian.tsx
index a4ece04..df420b2 100644
--- a/src/kaipian/Kaipian.tsx
+++ b/src/kaipian/Kaipian.tsx
@@ -342,8 +342,9 @@ export const Kaipian = () => {
+ {vis.provider === 'ollama' &&
{t('用本机视觉模型看图:不花钱、不联网。3B 的小模型 3 选 1 能对 4/5,偶尔把雕像当真猫;想更稳装 7B:')}ollama pull qwen2.5vl:7b
}
{vis.provider && }
diff --git a/src/kaipian/i18n.ts b/src/kaipian/i18n.ts
index ffb4f49..bc6dbfd 100644
--- a/src/kaipian/i18n.ts
+++ b/src/kaipian/i18n.ts
@@ -47,6 +47,7 @@ const EN: Record = {
'没连上本机的 Ollama(': 'Cannot reach Ollama on this machine (', ')。装好并启动后回来刷新:': '). Install and start it, then refresh: ',
'Ollama 在跑,但还没装能写稿的模型。终端里跑:': 'Ollama is running but has no chat model yet. In a terminal run: ',
'用这台机器上的模型写脚本:不花钱、不联网。7B 级的小模型偶尔写得偏短,开片会自动要求重写一次;想更稳就换 14B 以上。': 'Write scripts with a model on this machine: free and offline. 7B-class models sometimes write too short — OpenShorts asks for one rewrite automatically; use 14B+ for steadier results.',
+ '用本机视觉模型看图:不花钱、不联网。3B 的小模型 3 选 1 能对 4/5,偶尔把雕像当真猫;想更稳装 7B:': 'Judge footage with a vision model on this machine: free and offline. A 3B model picks the right clip 4 times out of 5 and occasionally mistakes a statue for a cat; for steadier results install 7B: ',
'复制诊断信息': 'Copy diagnostics', '已复制 ✓': 'Copied ✓', '正在体检…': 'Checking…', '去反馈 ↗': 'Report an issue ↗',
'报问题时贴上它:版本、系统、体检结果,key 已打码。': 'Paste this when reporting a problem: version, system, health check — keys are masked.',
'还没有写脚本用的文本模型 key。': 'No text-model key for script writing yet. Open ',
diff --git a/src/local/ollama.mjs b/src/local/ollama.mjs
index 0fba7a0..56ba97f 100644
--- a/src/local/ollama.mjs
+++ b/src/local/ollama.mjs
@@ -13,14 +13,17 @@ export const ollamaBaseUrl = (env = process.env) => {
/** 嵌入模型聊不了天:列进下拉等于埋雷(选了它写脚本直接报错) */
const isEmbedding = (m) => /embed|bge-|minilm|nomic|e5-|gte-/i.test(m.name ?? '') || /bert/i.test(m.details?.family ?? '');
+/** 能看图的本机模型(按名字判:qwen2.5vl / llava / minicpm-v / moondream / llama3.2-vision…)——看图把关只能选这些 */
+export const isVisionModel = (name) => /vl|llava|vision|minicpm-v|moondream|bakllava|gemma3/i.test(String(name));
+
export async function ollamaStatus({ fetchImpl = fetch, env = process.env, timeoutMs = 1500 } = {}) {
const baseUrl = ollamaBaseUrl(env);
try {
const r = await fetchImpl(`${baseUrl}/api/tags`, { signal: AbortSignal.timeout(timeoutMs) });
- if (!r.ok) return { running: false, baseUrl, models: [], reason: `HTTP ${r.status}` };
+ if (!r.ok) return { running: false, baseUrl, models: [], visionModels: [], reason: `HTTP ${r.status}` };
const all = (await r.json()).models ?? [];
// 大的在前:同一台机器上,参数多的那个写口播稿明显更稳
const models = all.filter((m) => !isEmbedding(m)).sort((a, b) => (b.size ?? 0) - (a.size ?? 0)).map((m) => m.name);
- return { running: true, baseUrl, models };
- } catch (e) { return { running: false, baseUrl, models: [], reason: String(e?.cause?.code ?? e?.name ?? e).slice(0, 60) }; }
+ return { running: true, baseUrl, models, visionModels: models.filter(isVisionModel) };
+ } catch (e) { return { running: false, baseUrl, models: [], visionModels: [], reason: String(e?.cause?.code ?? e?.name ?? e).slice(0, 60) }; }
}
diff --git a/src/sources/rank.mjs b/src/sources/rank.mjs
index 80b5164..1000d86 100644
--- a/src/sources/rank.mjs
+++ b/src/sources/rank.mjs
@@ -66,7 +66,16 @@ export async function evidenceFrame(candidate, { fetchImpl = fetch, timeoutMs =
/** 解析回复里的 JSON 数组 [{i, score, why}] */
export function parseScores(text, n) {
const m = String(text).match(/\[[\s\S]*\]/); if (!m) return null;
- try { const arr = JSON.parse(m[0]); const out = Array.from({ length: n }, (_, i) => ({ i, score: 0, why: '' })); for (const x of arr) { const i = Number(x.i ?? x.index); if (i >= 0 && i < n) out[i] = { i, score: Math.max(0, Math.min(10, Number(x.score) || 0)), why: String(x.why ?? '') }; } return out; } catch { return null; }
+ try {
+ const arr = JSON.parse(m[0]); if (!Array.isArray(arr)) return null;
+ // 候选按 1..n 编号(与连接器留的 [图片N] 占位一致——以前"候选 0:[图片1]"两套编号打架,模型回的 i 按哪个数没法确定)。
+ // 兼容仍按 0 起数的模型:出现了 i=0 或 i=n 越界的情况就按 0 起解释
+ const idx = arr.map((x) => Number(x.i ?? x.index)).filter(Number.isFinite);
+ const oneBased = !idx.includes(0) && idx.every((i) => i >= 1 && i <= n);
+ const out = Array.from({ length: n }, (_, i) => ({ i, score: 0, why: '' }));
+ for (const x of arr) { const i = Number(x.i ?? x.index) - (oneBased ? 1 : 0); if (i >= 0 && i < n) out[i] = { i, score: Math.max(0, Math.min(10, Number(x.score) || 0)), why: String(x.why ?? '') }; }
+ return out;
+ } catch { return null; }
}
/**
@@ -89,8 +98,8 @@ export async function rankCandidates(candidates, intent, { connector, cfg, thres
if (!usable.length) return candidates.map((c) => ({ ...c, score: null }));
const zh = /[一-鿿]/.test(intent);
const prompt = (zh
- ? [`你是短视频剪辑师。下面是同一段口播要配的画面意图,以及 ${usable.length} 条候选素材各一帧。给每条打分 0–10:画面主体、场景与意图是否匹配(主体对得上给 6 分起,完全无关 0–2 分,图表/文字/标题卡一律 ≤ 2)。`, `画面意图:${intent}`, ...usable.map((x, k) => `候选 ${k}:${x.f}`), '只输出 JSON 数组:[{"i":0,"score":7,"why":"一句话"}, …]']
- : [`You are a video editor. Below is the visual intent for one narration segment and one frame from each of ${usable.length} candidate clips. Score each 0–10 for how well subject/scene match the intent (subject matches → ≥6; unrelated → 0–2; charts/text/title cards ≤ 2).`, `Intent: ${intent}`, ...usable.map((x, k) => `Candidate ${k}: ${x.f}`), 'Output only a JSON array: [{"i":0,"score":7,"why":"…"}, …]']).join('\n');
+ ? [`你是短视频剪辑师。下面是同一段口播要配的画面意图,以及 ${usable.length} 条候选素材各一帧。给每条打分 0–10:画面主体、场景与意图是否匹配(主体对得上给 6 分起,完全无关 0–2 分,图表/文字/标题卡一律 ≤ 2)。`, `画面意图:${intent}`, ...usable.map((x, k) => `候选 ${k + 1}:${x.f}`), `只输出 JSON 数组,i 是候选编号(1 到 ${usable.length}):[{"i":1,"score":7,"why":"一句话"}, …]`]
+ : [`You are a video editor. Below is the visual intent for one narration segment and one frame from each of ${usable.length} candidate clips. Score each 0–10 for how well subject/scene match the intent (subject matches → ≥6; unrelated → 0–2; charts/text/title cards ≤ 2).`, `Intent: ${intent}`, ...usable.map((x, k) => `Candidate ${k + 1}: ${x.f}`), `Output only a JSON array where i is the candidate number (1 to ${usable.length}): [{"i":1,"score":7,"why":"one sentence"}, …]`]).join('\n');
let scores = null;
// 推理模型(Agnes 2.0-flash)会先吐几百字思考再给 JSON:预算给足。
// 接口偶尔会抽(真机上六镜里抽了一次),所以多试两次并退避——一次失败就等于这一镜没人把关。
diff --git a/tests/ollama.test.mjs b/tests/ollama.test.mjs
index 531035d..3671e87 100644
--- a/tests/ollama.test.mjs
+++ b/tests/ollama.test.mjs
@@ -9,13 +9,16 @@ const TAGS = { models: [
{ name: 'mxbai-embed-large:latest', size: 669_000_000, details: { family: 'bert' } },
{ name: 'qwen2.5:14b', size: 8_988_000_000, details: { family: 'qwen2' } },
{ name: 'qwen2.5-coder:7b', size: 4_683_000_000, details: { family: 'qwen2' } },
+ { name: 'qwen2.5vl:3b', size: 3_200_000_000, details: { family: 'qwen25vl' } },
+ { name: 'llava:7b', size: 4_700_000_000, details: { family: 'llama' } },
] };
const ok = (body) => async () => ({ ok: true, json: async () => body });
test('嵌入模型不进下拉(选了它写脚本会直接报错);大模型排前面', async () => {
const s = await ollamaStatus({ fetchImpl: ok(TAGS), env: {} });
assert.equal(s.running, true);
- assert.deepEqual(s.models, ['qwen2.5:14b', 'qwen2.5-coder:7b', 'llama3:latest']);
+ assert.deepEqual(s.models, ['qwen2.5:14b', 'llava:7b', 'qwen2.5-coder:7b', 'llama3:latest', 'qwen2.5vl:3b']);
+ assert.deepEqual(s.visionModels, ['llava:7b', 'qwen2.5vl:3b'], '看图把关只列能看图的;纯文本模型选了等于没开');
});
test('没在跑 / 回了错误码:running=false 并带原因,不抛', async () => {
diff --git a/tests/rank.test.mjs b/tests/rank.test.mjs
index 5a78507..df62c58 100644
--- a/tests/rank.test.mjs
+++ b/tests/rank.test.mjs
@@ -92,3 +92,13 @@ test('低于补位线的照样判退,绝不混进成片', { skip: !hasFfmpeg &
assert.ok(r.every((c) => !c.fillOnly));
fs.rmSync(d, { recursive: true, force: true });
});
+
+test('parseScores:候选按 1..n 编号时按 1 起解释;模型仍按 0 起(出现 i=0)时按 0 起——真机 3B 视觉模型回 i=1,2 把第 1 条永远算成 0 分', () => {
+ const one = parseScores('[{"i":1,"score":8},{"i":2,"score":1},{"i":3,"score":2}]', 3);
+ assert.deepEqual(one.map((x) => x.score), [8, 1, 2]);
+ const zero = parseScores('[{"i":0,"score":8},{"i":1,"score":1},{"i":2,"score":2}]', 3);
+ assert.deepEqual(zero.map((x) => x.score), [8, 1, 2]);
+ // 只回了一部分且从 1 起:缺的那条 0 分,不能把第 1 条的分挪到第 0 条
+ const partial = parseScores('[{"i":2,"score":9}]', 3);
+ assert.deepEqual(partial.map((x) => x.score), [0, 9, 0]);
+});