diff --git a/README.md b/README.md index 65ded3f..c6ba6b4 100644 --- a/README.md +++ b/README.md @@ -138,6 +138,12 @@ OrcaRouter exposes one endpoint for all frontier and open-weight models, so you Set `ORCAROUTER_API_KEY` (and optionally `ORCAROUTER_API_BASE`, which defaults to `https://api.orcarouter.ai/v1`) instead of pasting the key into the UI if you prefer environment-based configuration. A key is available at https://www.orcarouter.ai. +#### Using Cheaper Inference as your LLM gateway (optional) + +FunClip can also send LLM-assisted clipping through [Cheaper Inference](https://cheaperinference.com), an OpenAI-compatible LLM gateway. One API key gives access to models from several labs. Select any `cheaperinference/` model in the **LLM Model Name** dropdown, paste a Cheaper Inference API key in the **APIKEY** box, and click 'LLM Inference'. FunClip removes the `cheaperinference/` prefix and sends the transcript and prompts to `https://api.cheaperinference.com/v1/chat/completions`. The returned segments work with the existing 'AI Clip' button. + +Set `CHEAPER_INFERENCE_API_KEY` (and optionally `CHEAPER_INFERENCE_API_BASE`, which defaults to `https://api.cheaperinference.com/v1`) instead of pasting the key into the UI if you prefer environment-based configuration. A key is available at https://cheaperinference.com/signup. + #### Content-aware clipping with TwelveLabs Pegasus (optional) Besides the transcript-based LLMs above, FunClip can optionally use [TwelveLabs](https://twelvelabs.io) Pegasus, a video understanding model that reasons over the actual video (visuals + audio) rather than only the ASR transcript. This helps pick highlight segments even when the transcript alone is ambiguous (e.g. action, scene changes, on-screen events). To use it, select the `pegasus1.5` model name, paste your TwelveLabs API key, upload a video, and click 'LLM Inference' — Pegasus returns segments in the same `N. [start-end] text` format, so the existing 'AI Clip' button works unchanged. It needs `pip install twelvelabs`, and a free API key is available at https://twelvelabs.io. diff --git a/README_zh.md b/README_zh.md index 49a534a..7836eac 100644 --- a/README_zh.md +++ b/README_zh.md @@ -163,6 +163,12 @@ OrcaRouter 用单一端点接入所有前沿与开源模型,无需修改 FunCl 也可以不填 UI,而是设置 `ORCAROUTER_API_KEY` 环境变量(可选 `ORCAROUTER_API_BASE`,默认为 `https://api.orcarouter.ai/v1`)。Key 可在 https://www.orcarouter.ai 获取。 +#### 使用 Cheaper Inference 作为 LLM 网关(可选) + +FunClip 也可以将 LLM 智能裁剪发送到 [Cheaper Inference](https://cheaperinference.com)——一个 OpenAI 兼容的 LLM 网关。一个 API key 即可使用多家厂商的模型。在 **LLM Model Name** 下拉框选择任意 `cheaperinference/` 模型,在 **APIKEY** 输入框粘贴 Cheaper Inference API key,点击“LLM推理”。FunClip 会去掉 `cheaperinference/` 前缀,把字幕与 prompt 发送到 `https://api.cheaperinference.com/v1/chat/completions`,返回的分段与现有“AI Clip”按钮兼容。 + +也可以不填 UI,而是设置 `CHEAPER_INFERENCE_API_KEY` 环境变量(可选 `CHEAPER_INFERENCE_API_BASE`,默认为 `https://api.cheaperinference.com/v1`)。Key 可在 https://cheaperinference.com/signup 获取。 + ### B.通过命令行调用使用FunClip的相关功能 ```shell # 下载下面命令用到的示例视频 diff --git a/funclip/launch.py b/funclip/launch.py index cfca30d..08ae700 100644 --- a/funclip/launch.py +++ b/funclip/launch.py @@ -150,7 +150,7 @@ def video_clip_addsub(dest_text, video_spk_input, start_ost, end_ost, state, out ) def llm_inference(system_content, user_content, srt_text, model, apikey, video_input=None): - SUPPORT_LLM_PREFIX = ['litellm', 'qwen', 'gpt', 'g4f', 'moonshot', 'deepseek', 'atlascloud', 'minimax', 'orcarouter', 'pegasus'] + SUPPORT_LLM_PREFIX = ['litellm', 'qwen', 'gpt', 'g4f', 'moonshot', 'deepseek', 'atlascloud', 'minimax', 'orcarouter', 'cheaperinference', 'pegasus'] if model.startswith('litellm/'): return litellm_call(apikey, model, user_content+'\n'+srt_text, system_content) if model.startswith('pegasus'): @@ -162,7 +162,7 @@ def llm_inference(system_content, user_content, srt_text, model, apikey, video_i return call_twelvelabs_pegasus(apikey, video_input, model=model, prompt=system_content) if model.startswith('qwen'): return call_qwen_model(apikey, model, user_content+'\n'+srt_text, system_content) - if model.startswith('gpt') or model.startswith('moonshot') or model.startswith('deepseek') or model.startswith('atlascloud/') or model.startswith('minimax/') or model.startswith('orcarouter/'): + if model.startswith('gpt') or model.startswith('moonshot') or model.startswith('deepseek') or model.startswith('atlascloud/') or model.startswith('minimax/') or model.startswith('orcarouter/') or model.startswith('cheaperinference/'): return openai_call(apikey, model, user_content+'\n'+srt_text, system_content) elif model.startswith('g4f'): model = "-".join(model.split('-')[1:]) @@ -273,6 +273,10 @@ def AI_clip_subti(LLM_res, dest_text, video_spk_input, start_ost, end_ost, video "orcarouter/fusion", "orcarouter/fusion-flash", "orcarouter/fusion-mini", + "cheaperinference/gpt-5.4-mini", + "cheaperinference/gpt-5.4", + "cheaperinference/claude-sonnet-5", + "cheaperinference/deepseek-v4-flash", "pegasus1.5"], value="deepseek-chat", label="LLM Model Name", diff --git a/funclip/llm/openai_api.py b/funclip/llm/openai_api.py index 4882558..66615df 100644 --- a/funclip/llm/openai_api.py +++ b/funclip/llm/openai_api.py @@ -19,6 +19,11 @@ ORCAROUTER_API_BASE = "https://api.orcarouter.ai/v1" ORCAROUTER_MODEL_PREFIX = "orcarouter/" +# Cheaper Inference is an OpenAI-compatible LLM gateway. Its model IDs are +# bare (e.g. `gpt-5.4-mini`), so the `cheaperinference/` prefix is stripped. +CHEAPER_INFERENCE_API_BASE = "https://api.cheaperinference.com/v1" +CHEAPER_INFERENCE_MODEL_PREFIX = "cheaperinference/" + def _resolve_model_config(model): base_url = None @@ -53,6 +58,16 @@ def _resolve_model_config(model): if not base_url: base_url = ORCAROUTER_API_BASE api_key_env = "ORCAROUTER_API_KEY" + elif model.startswith(CHEAPER_INFERENCE_MODEL_PREFIX): + model = model[len(CHEAPER_INFERENCE_MODEL_PREFIX):] + if not model: + raise ValueError( + "Model name is empty after stripping cheaperinference/ prefix" + ) + base_url = os.environ.get("CHEAPER_INFERENCE_API_BASE", CHEAPER_INFERENCE_API_BASE).strip() + if not base_url: + base_url = CHEAPER_INFERENCE_API_BASE + api_key_env = "CHEAPER_INFERENCE_API_KEY" elif model.startswith("deepseek"): base_url = "https://api.deepseek.com" elif model.startswith("gpt-3.5-turbo"): diff --git a/tests/test_cheaperinference_api.py b/tests/test_cheaperinference_api.py new file mode 100644 index 0000000..f142f9c --- /dev/null +++ b/tests/test_cheaperinference_api.py @@ -0,0 +1,92 @@ +"""Tests for Cheaper Inference routing through the OpenAI-compatible client.""" + +import os +import unittest +from unittest.mock import MagicMock, patch + +from funclip.llm.openai_api import ( + CHEAPER_INFERENCE_API_BASE, + openai_call, +) + + +def _mock_completion(content="ok"): + completion = MagicMock() + completion.choices = [MagicMock()] + completion.choices[0].message.content = content + return completion + + +class TestCheaperInferenceRouting(unittest.TestCase): + def test_cheaperinference_prefix_uses_gateway_base_url(self): + client = MagicMock() + client.chat.completions.create.return_value = _mock_completion("clip plan") + + with patch("funclip.llm.openai_api.OpenAI", return_value=client) as openai_cls: + result = openai_call( + "ci-key", + "cheaperinference/gpt-5.4-mini", + "subtitle text", + "find highlights", + ) + + self.assertEqual(result, "clip plan") + openai_cls.assert_called_once_with( + api_key="ci-key", + base_url=CHEAPER_INFERENCE_API_BASE, + ) + # Cheaper Inference model IDs are bare: the prefix is stripped. + call_kwargs = client.chat.completions.create.call_args[1] + self.assertEqual(call_kwargs["model"], "gpt-5.4-mini") + + def test_cheaperinference_api_key_falls_back_to_env(self): + client = MagicMock() + client.chat.completions.create.return_value = _mock_completion() + + with patch.dict(os.environ, {"CHEAPER_INFERENCE_API_KEY": "env-ci-key"}, clear=False): + with patch("funclip.llm.openai_api.OpenAI", return_value=client) as openai_cls: + openai_call("", "cheaperinference/gpt-5.4-mini", "text") + + openai_cls.assert_called_once_with( + api_key="env-ci-key", + base_url=CHEAPER_INFERENCE_API_BASE, + ) + call_kwargs = client.chat.completions.create.call_args[1] + self.assertEqual(call_kwargs["model"], "gpt-5.4-mini") + + def test_cheaperinference_api_base_env_overrides(self): + client = MagicMock() + client.chat.completions.create.return_value = _mock_completion() + + with patch.dict( + os.environ, + {"CHEAPER_INFERENCE_API_BASE": "https://gateway.example.com/v1"}, + clear=False, + ): + with patch("funclip.llm.openai_api.OpenAI", return_value=client) as openai_cls: + openai_call("ci-key", "cheaperinference/claude-sonnet-5", "text") + + openai_cls.assert_called_once_with( + api_key="ci-key", + base_url="https://gateway.example.com/v1", + ) + + def test_empty_cheaperinference_model_raises(self): + with self.assertRaises(ValueError): + openai_call("key", "cheaperinference/", "text") + + def test_missing_cheaperinference_key_does_not_fall_back_to_openai_key(self): + with patch.dict( + os.environ, + {"OPENAI_API_KEY": "openai-only-key"}, + clear=True, + ): + with patch("funclip.llm.openai_api.OpenAI") as openai_cls: + with self.assertRaisesRegex(ValueError, "CHEAPER_INFERENCE_API_KEY"): + openai_call("", "cheaperinference/gpt-5.4-mini", "text") + + openai_cls.assert_not_called() + + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_cheaperinference_launch_integration.py b/tests/test_cheaperinference_launch_integration.py new file mode 100644 index 0000000..03d1f52 --- /dev/null +++ b/tests/test_cheaperinference_launch_integration.py @@ -0,0 +1,74 @@ +"""Regression tests for Cheaper Inference choices and prompt routing in the launcher.""" + +import ast +import unittest +from pathlib import Path + + +LAUNCH_PATH = Path(__file__).resolve().parents[1] / "funclip" / "launch.py" + + +class TestCheaperInferenceLaunchIntegration(unittest.TestCase): + @classmethod + def setUpClass(cls): + cls.tree = ast.parse(LAUNCH_PATH.read_text(encoding="utf-8")) + + def test_openai_compatible_route_handles_cheaperinference_prefix(self): + llm_inference = next( + node + for node in ast.walk(self.tree) + if isinstance(node, ast.FunctionDef) and node.name == "llm_inference" + ) + openai_call = next( + node + for node in ast.walk(llm_inference) + if isinstance(node, ast.Call) + and isinstance(node.func, ast.Name) + and node.func.id == "openai_call" + ) + # The openai_call dispatch branch must also cover cheaperinference/ models. + dispatch_condition = next( + node + for node in ast.walk(llm_inference) + if isinstance(node, ast.If) and isinstance(node.test, ast.BoolOp) + ) + self.assertIn("cheaperinference/", ast.unparse(dispatch_condition.test)) + self.assertEqual(ast.unparse(openai_call.args[2]), "user_content + '\\n' + srt_text") + self.assertEqual(ast.unparse(openai_call.args[3]), "system_content") + + def test_support_prefix_list_includes_cheaperinference(self): + llm_inference = next( + node + for node in ast.walk(self.tree) + if isinstance(node, ast.FunctionDef) and node.name == "llm_inference" + ) + assigned = [ + node + for node in ast.walk(llm_inference) + if isinstance(node, ast.Assign) and node.targets[0].id == "SUPPORT_LLM_PREFIX" + ] + self.assertEqual(len(assigned), 1) + prefix_list = assigned[0].value + self.assertIsInstance(prefix_list, ast.List) + prefixes = { + node.value + for node in ast.walk(prefix_list) + if isinstance(node, ast.Constant) and isinstance(node.value, str) + } + self.assertIn("cheaperinference", prefixes) + + def test_dropdown_lists_cheaperinference_gateway_models(self): + string_literals = { + node.value + for node in ast.walk(self.tree) + if isinstance(node, ast.Constant) and isinstance(node.value, str) + } + + self.assertIn("cheaperinference/gpt-5.4-mini", string_literals) + self.assertIn("cheaperinference/gpt-5.4", string_literals) + self.assertIn("cheaperinference/claude-sonnet-5", string_literals) + self.assertIn("cheaperinference/deepseek-v4-flash", string_literals) + + +if __name__ == "__main__": + unittest.main()