From 63916d831e36581bacf79cbe1e3b1d824c39fb8f Mon Sep 17 00:00:00 2001 From: Hannah Smith Date: Thu, 17 Sep 2026 13:38:40 -0400 Subject: [PATCH 1/3] Add RedHatAI/Qwen3.8-27B-INT4 model config Register the INT4-quantized Qwen3.8-27B so the job queue can serve it for Michael's intake request. Mirrors the BF16/FP8 siblings (fp8 KV cache, 262144 context) so the leaderboard comparison isolates quantization impact. --- src/coding_agent_bench/models/__init__.py | 2 ++ src/coding_agent_bench/models/configs.py | 14 ++++++++++++++ tests/test_models.py | 22 ++++++++++++++++++++++ 3 files changed, 38 insertions(+) create mode 100644 tests/test_models.py diff --git a/src/coding_agent_bench/models/__init__.py b/src/coding_agent_bench/models/__init__.py index e729db2..1f8fdee 100644 --- a/src/coding_agent_bench/models/__init__.py +++ b/src/coding_agent_bench/models/__init__.py @@ -2,6 +2,7 @@ from coding_agent_bench.models.configs import ( Qwen_Qwen3_8_27B, Qwen_Qwen3_8_27B_FP8, + RedHatAI_Qwen3_8_27B_INT4, RedHatAI_gemma_4_31B_it_FP8_block, RedHatAI_gpt_oss_120b, RedHatAI_Mistral_Small_4_119B_2603_NVFP4, @@ -12,6 +13,7 @@ MODEL_CONFIGS: list[type[ModelConfig]] = [ Qwen_Qwen3_8_27B, Qwen_Qwen3_8_27B_FP8, + RedHatAI_Qwen3_8_27B_INT4, RedHatAI_gemma_4_31B_it_FP8_block, RedHatAI_gpt_oss_120b, RedHatAI_Mistral_Small_4_119B_2603_NVFP4, diff --git a/src/coding_agent_bench/models/configs.py b/src/coding_agent_bench/models/configs.py index f84c964..114e9d2 100644 --- a/src/coding_agent_bench/models/configs.py +++ b/src/coding_agent_bench/models/configs.py @@ -28,6 +28,20 @@ class Qwen_Qwen3_8_27B_FP8(ModelConfig): "--mm-encoder-tp-mode", "data", ] +class RedHatAI_Qwen3_8_27B_INT4(ModelConfig): + + name = "RedHatAI/Qwen3.8-27B-INT4" + model_max_len = 262144 + args = [ + "--model", "RedHatAI/Qwen3.8-27B-INT4", + "--max-model-len", "262144", + "--kv-cache-dtype", "fp8", + "--enable-auto-tool-choice", + "--tool-call-parser", "qwen3_coder", + "--reasoning-parser", "qwen3", + "--mm-encoder-tp-mode", "data", + ] + class RedHatAI_gemma_4_31B_it_FP8_block(ModelConfig): name = "RedHatAI/gemma-4-31B-it-FP8-block" diff --git a/tests/test_models.py b/tests/test_models.py new file mode 100644 index 0000000..a78d372 --- /dev/null +++ b/tests/test_models.py @@ -0,0 +1,22 @@ +from coding_agent_bench.models import get_model_config + + +def test_qwen38_int4_model_config(): + config = get_model_config("RedHatAI/Qwen3.8-27B-INT4") + + assert config.model_max_len == 262144 + assert config.args == [ + "--model", + "RedHatAI/Qwen3.8-27B-INT4", + "--max-model-len", + "262144", + "--kv-cache-dtype", + "fp8", + "--enable-auto-tool-choice", + "--tool-call-parser", + "qwen3_coder", + "--reasoning-parser", + "qwen3", + "--mm-encoder-tp-mode", + "data", + ] From 1a600aae690dec766dfda102f27731e1379e7a11 Mon Sep 17 00:00:00 2001 From: Hannah Smith Date: Thu, 17 Sep 2026 13:39:03 -0400 Subject: [PATCH 2/3] Document OpenRouter agent API keys in .env.example Add ANTHROPIC/OPENAI/OPENROUTER key placeholders needed when submitting jobs with server_url=openrouter. --- .env.example | 8 ++++++++ 1 file changed, 8 insertions(+) diff --git a/.env.example b/.env.example index c7e609e..800e746 100644 --- a/.env.example +++ b/.env.example @@ -32,3 +32,11 @@ NEBIUS_ENABLED=0 # NEBIUS_INSTANCE_NAME_PREFIX=cab-worker # NEBIUS_IDLE_TIMEOUT_SECONDS=600 # HF_TOKEN= + +# Intake Poller (optional) +# GOOGLE_APPLICATION_CREDENTIALS=service-account.json +# GOOGLE_SHEET_ID= +# JOB_QUEUE_URL=http://localhost:8000 +# + SENDER_EMAIL= +# AUTO_APPROVE=false From ec5d0634ab3f3e77df36580831c4fb7cc9a08ad9 Mon Sep 17 00:00:00 2001 From: Hannah Smith Date: Thu, 17 Sep 2026 14:05:43 -0400 Subject: [PATCH 3/3] =?UTF-8?q?Bump=20version:=200.2.4=20=E2=86=92=200.2.5?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- .bumpversion.toml | 2 +- deploy/intake-cronjob.yml | 2 +- deploy/job-queue-service.yml | 2 +- pyproject.toml | 2 +- uv.lock | 2 +- 5 files changed, 5 insertions(+), 5 deletions(-) diff --git a/.bumpversion.toml b/.bumpversion.toml index 2d8ed7e..54912a0 100644 --- a/.bumpversion.toml +++ b/.bumpversion.toml @@ -1,5 +1,5 @@ [tool.bumpversion] -current_version = "0.2.4" +current_version = "0.2.5" commit = true tag = true tag_name = "v{new_version}" diff --git a/deploy/intake-cronjob.yml b/deploy/intake-cronjob.yml index 74db114..90261b1 100644 --- a/deploy/intake-cronjob.yml +++ b/deploy/intake-cronjob.yml @@ -22,7 +22,7 @@ spec: restartPolicy: Never containers: - name: poller - image: ghcr.io/redhat-et/coding_agent_bench:v0.2.4 + image: ghcr.io/redhat-et/coding_agent_bench:v0.2.5 imagePullPolicy: Always command: ["uv", "run", "python", "-m", "coding_agent_bench.intake.poller"] env: diff --git a/deploy/job-queue-service.yml b/deploy/job-queue-service.yml index d51b95e..422f115 100644 --- a/deploy/job-queue-service.yml +++ b/deploy/job-queue-service.yml @@ -41,7 +41,7 @@ spec: securityContext: fsGroup: 1001 containers: - - image: ghcr.io/redhat-et/coding_agent_bench:v0.2.4 + - image: ghcr.io/redhat-et/coding_agent_bench:v0.2.5 name: job-queue command: ["/bin/sh", "-c"] args: diff --git a/pyproject.toml b/pyproject.toml index b7694a7..1bcf4f8 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -1,6 +1,6 @@ [project] name = "coding-agent-bench" -version = "0.2.4" +version = "0.2.5" description = "Add your description here" readme = "README.md" requires-python = ">=3.12" diff --git a/uv.lock b/uv.lock index 721d2b4..c2fb79c 100644 --- a/uv.lock +++ b/uv.lock @@ -393,7 +393,7 @@ wheels = [ [[package]] name = "coding-agent-bench" -version = "0.2.4" +version = "0.2.5" source = { editable = "." } dependencies = [ { name = "awscli" },