mirror of
https://github.com/windmill-labs/windmill.git
synced 2026-08-18 16:02:10 +00:00
b0ddcf31e4
* ci: add path-gated AI agent integration tests workflow Runs integration_tests/ai_agent_tests against real LLM providers (Anthropic/OpenAI/Google) only when AI-agent backend code or the tests change, since runs make paid LLM calls. Adds a conftest fixture that skips provider-parametrized cases whose API keys are absent, so CI exercises only the providers it has secrets for. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: add path-gated ai_evals global-mode smoke workflow Runs the global AI chat eval (global-test1) across one cheap model per provider (Anthropic/OpenAI/Google/DeepSeek) only when the eval harness or copilot chat code change, since runs make paid LLM calls. Builds Windmill CE from source as the AI proxy; global tools/drafts run in the Vitest bridge. Gates on the deterministic draft pipeline (run succeeded + produced a draft + used write_script), not the variable LLM judge score. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: run AI smokes on PR ready-for-review instead of every push Switch the pull_request trigger from `synchronize` (every commit) to `ready_for_review`, with a job guard skipping draft PRs, so the paid LLM runs only fire when a PR is marked ready to merge (plus push-to-main and manual dispatch). Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(ai_evals): lazily load cli mode so non-cli evals skip the cli toolchain The entrypoint eagerly imported modes/cli, which pulls the wmill CLI guidance modules and their JSR deps (@cliffy/*). Global/flow/script/app runs then crashed with "Cannot find module '@cliffy/ansi/colors'" when the cli workspace deps were not installed. Import createCliModeRunner dynamically inside runCliBenchmark instead. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * test(ai_agent): raise low max_completion_tokens to OpenAI's 16 minimum OpenAI's /v1/responses rejects max_output_tokens < 16 with a 400, failing test_low_max_tokens for openai. 16 still exercises a truncated response. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci: run ai_evals workflow on Node 22 for the frontend undici 8.x dep The Vitest bridge loads frontend/node_modules/undici@8.x, which requires Node >=22.19; Node 20 failed with "webidl.util.markAsUncloneable is not a function" when loading vitest.config.ts. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix(ai_evals): run frontend evals autonomously + give global-test1 more turns Frontend evals (flow/script/app/global) ran the production chat prompt, which assumes an interactive human — so cheaper models burned their turn budget asking for confirmation, waiting for approval, or presenting a plan, sometimes hitting maxTurns without producing a draft. Append a shared autonomy note in baseEvalRunner (the path all frontend modes share, mirroring cli mode): act directly on clear requests; only ask on genuinely ambiguous ones (preserving the askUserQuestion cases). Also raise global-test1's maxTurns 8 -> 10 so a model that over-explores still converges. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * ci(ai_evals): watch draft/prompt deps outside copilot/ The global eval runs production frontend code in-process, so the smoke's behavior depends on files outside frontend/src/lib/components/copilot/**: the draft model (userDraft.svelte.ts, userDraftDbSyncer.svelte.ts), script inference (infer.ts), and the chat system prompts ($system_prompts -> system_prompts/auto-generated). Add them to both push and PR path filters so a change there actually triggers the smoke that gates on draft production. Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com> * fix: skip direct provider tests without credentials * feat: add ai evals skip judge flag * fix: simplify ai evals ci gate * fix: simplify ai evals smoke gate * fix: handle ai eval workflow triggers --------- Co-authored-by: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
229 lines
7.5 KiB
Python
229 lines
7.5 KiB
Python
"""
|
|
Completion parameter tests for AI agents.
|
|
|
|
Tests that AI agents correctly handle temperature and max_completion_tokens:
|
|
- Default parameters (undefined)
|
|
- Low temperature (0.0 - deterministic)
|
|
- High temperature (0.9 - more random)
|
|
- Low max_completion_tokens (16 - short response; OpenAI's minimum)
|
|
- High max_completion_tokens (4096 - longer response allowed)
|
|
- Combined parameters
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from .conftest import AIAgentTestClient, create_ai_agent_flow
|
|
from .providers import ALL_PROVIDERS, get_provider_ids
|
|
|
|
|
|
class TestCompletionParams:
|
|
"""Test AI agent temperature and max_completion_tokens parameters."""
|
|
|
|
@pytest.mark.parametrize(
|
|
"provider_config",
|
|
ALL_PROVIDERS,
|
|
ids=get_provider_ids(ALL_PROVIDERS),
|
|
)
|
|
def test_default_params(
|
|
self,
|
|
client: AIAgentTestClient,
|
|
setup_providers,
|
|
provider_config,
|
|
):
|
|
"""
|
|
Test with default parameters (no temperature or max_completion_tokens).
|
|
This serves as a baseline to ensure the agent works without these params.
|
|
"""
|
|
flow_value = create_ai_agent_flow(
|
|
provider_input_transform=provider_config["input_transform"],
|
|
system_prompt="You are a helpful assistant. Be concise.",
|
|
output_type="text",
|
|
)
|
|
|
|
result = client.run_preview_flow(
|
|
flow_value=flow_value,
|
|
args={"user_message": "What is 2 + 2? Answer with just the number."},
|
|
)
|
|
|
|
assert result is not None
|
|
# Result should contain the answer
|
|
result_str = str(result)
|
|
assert "4" in result_str, f"Expected '4' in result: {result}"
|
|
|
|
print(f"Default params result from {provider_config['name']}: {result}")
|
|
|
|
@pytest.mark.parametrize(
|
|
"provider_config",
|
|
ALL_PROVIDERS,
|
|
ids=get_provider_ids(ALL_PROVIDERS),
|
|
)
|
|
def test_low_temperature(
|
|
self,
|
|
client: AIAgentTestClient,
|
|
setup_providers,
|
|
provider_config,
|
|
):
|
|
"""
|
|
Test with temperature=0.0 (deterministic output).
|
|
Low temperature should produce more focused, consistent responses.
|
|
"""
|
|
flow_value = create_ai_agent_flow(
|
|
provider_input_transform=provider_config["input_transform"],
|
|
system_prompt="You are a helpful assistant. Be concise.",
|
|
output_type="text",
|
|
temperature=0.0,
|
|
)
|
|
|
|
result = client.run_preview_flow(
|
|
flow_value=flow_value,
|
|
args={"user_message": "What is 2 + 2? Answer with just the number."},
|
|
)
|
|
|
|
assert result is not None
|
|
result_str = str(result)
|
|
assert "4" in result_str, f"Expected '4' in result: {result}"
|
|
|
|
print(f"Low temperature (0.0) result from {provider_config['name']}: {result}")
|
|
|
|
@pytest.mark.parametrize(
|
|
"provider_config",
|
|
ALL_PROVIDERS,
|
|
ids=get_provider_ids(ALL_PROVIDERS),
|
|
)
|
|
def test_high_temperature(
|
|
self,
|
|
client: AIAgentTestClient,
|
|
setup_providers,
|
|
provider_config,
|
|
):
|
|
"""
|
|
Test with temperature=0.9 (more random output).
|
|
High temperature should still produce valid responses.
|
|
"""
|
|
flow_value = create_ai_agent_flow(
|
|
provider_input_transform=provider_config["input_transform"],
|
|
system_prompt="You are a helpful assistant. Be concise.",
|
|
output_type="text",
|
|
temperature=0.9,
|
|
)
|
|
|
|
result = client.run_preview_flow(
|
|
flow_value=flow_value,
|
|
args={"user_message": "What is 2 + 2? Answer with just the number."},
|
|
)
|
|
|
|
assert result is not None
|
|
# With high temperature, the model might be more creative but should still respond
|
|
result_str = str(result)
|
|
# We just verify we got a non-empty response
|
|
assert len(result_str) > 0, f"Expected non-empty result: {result}"
|
|
|
|
print(f"High temperature (0.9) result from {provider_config['name']}: {result}")
|
|
|
|
@pytest.mark.parametrize(
|
|
"provider_config",
|
|
ALL_PROVIDERS,
|
|
ids=get_provider_ids(ALL_PROVIDERS),
|
|
)
|
|
def test_low_max_tokens(
|
|
self,
|
|
client: AIAgentTestClient,
|
|
setup_providers,
|
|
provider_config,
|
|
):
|
|
"""
|
|
Test with a low max_completion_tokens (16 — short response).
|
|
The response should be truncated or very short. 16 is OpenAI's minimum
|
|
for max_output_tokens; lower values (e.g. 10) are rejected with a 400.
|
|
"""
|
|
flow_value = create_ai_agent_flow(
|
|
provider_input_transform=provider_config["input_transform"],
|
|
system_prompt="You are a helpful assistant.",
|
|
output_type="text",
|
|
max_completion_tokens=16,
|
|
)
|
|
|
|
result = client.run_preview_flow(
|
|
flow_value=flow_value,
|
|
args={"user_message": "Explain the theory of relativity in detail."},
|
|
)
|
|
|
|
assert result is not None
|
|
# The response should be truncated due to low max_tokens
|
|
# We verify we got some response (even if truncated)
|
|
result_str = str(result)
|
|
assert len(result_str) > 0, f"Expected non-empty result: {result}"
|
|
|
|
print(f"Low max_tokens (16) result from {provider_config['name']}: {result}")
|
|
|
|
@pytest.mark.parametrize(
|
|
"provider_config",
|
|
ALL_PROVIDERS,
|
|
ids=get_provider_ids(ALL_PROVIDERS),
|
|
)
|
|
def test_high_max_tokens(
|
|
self,
|
|
client: AIAgentTestClient,
|
|
setup_providers,
|
|
provider_config,
|
|
):
|
|
"""
|
|
Test with max_completion_tokens=4096 (longer response allowed).
|
|
The model should be able to produce longer responses if needed.
|
|
"""
|
|
flow_value = create_ai_agent_flow(
|
|
provider_input_transform=provider_config["input_transform"],
|
|
system_prompt="You are a helpful assistant. Be concise.",
|
|
output_type="text",
|
|
max_completion_tokens=4096,
|
|
)
|
|
|
|
result = client.run_preview_flow(
|
|
flow_value=flow_value,
|
|
args={"user_message": "What is 2 + 2? Answer with just the number."},
|
|
)
|
|
|
|
assert result is not None
|
|
result_str = str(result)
|
|
assert "4" in result_str, f"Expected '4' in result: {result}"
|
|
|
|
print(f"High max_tokens (4096) result from {provider_config['name']}: {result}")
|
|
|
|
@pytest.mark.parametrize(
|
|
"provider_config",
|
|
ALL_PROVIDERS,
|
|
ids=get_provider_ids(ALL_PROVIDERS),
|
|
)
|
|
def test_combined_params(
|
|
self,
|
|
client: AIAgentTestClient,
|
|
setup_providers,
|
|
provider_config,
|
|
):
|
|
"""
|
|
Test with both temperature and max_completion_tokens set.
|
|
Verifies that both parameters work together correctly.
|
|
"""
|
|
flow_value = create_ai_agent_flow(
|
|
provider_input_transform=provider_config["input_transform"],
|
|
system_prompt="You are a helpful assistant. Be concise.",
|
|
output_type="text",
|
|
temperature=0.5,
|
|
max_completion_tokens=100,
|
|
)
|
|
|
|
result = client.run_preview_flow(
|
|
flow_value=flow_value,
|
|
args={"user_message": "What is 2 + 2? Answer with just the number."},
|
|
)
|
|
|
|
assert result is not None
|
|
result_str = str(result)
|
|
assert "4" in result_str, f"Expected '4' in result: {result}"
|
|
|
|
print(f"Combined params (temp=0.5, max_tokens=100) result from {provider_config['name']}: {result}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
pytest.main([__file__, "-v", "-s"])
|