Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
45 commits
Select commit Hold shift + click to select a range
1fd1388
add memory prototyper and dependencies
wenta0g Jan 13, 2026
4ef8e34
add Cloud SQL Proxy sidecar with IAM auth
wenta0g Jan 13, 2026
1bee8f1
db proxy less resource
wenta0g Jan 13, 2026
5a502a8
added new db field with filters
iany0 Jan 20, 2026
d547216
Merge branch 'google:main' into main
wenta0g Jan 25, 2026
dfbe285
cloud sql connect add backup
wenta0g Jan 25, 2026
38f161d
presubmit format fix
wenta0g Jan 27, 2026
40ee45d
lint fix
wenta0g Jan 27, 2026
206f764
add memory_based enhancer
wenta0g Jan 27, 2026
730f3d0
Improved prompt and truncate stderr
iany0 Feb 5, 2026
7c65e0f
add a longer timeout when gemini 429 error (resource exhausted)
wenta0g Feb 6, 2026
f5f77a9
smart truncate for long prompt to fetch true error message
wenta0g Feb 10, 2026
22849b6
create a new test set and enable project specific query
iany0 Feb 23, 2026
d508653
updated benchmark and logic for query cloudsql
iany0 Feb 23, 2026
72ef0cc
safer sql connection err handling, avoid crashing the memory_prototyp…
wenta0g Feb 24, 2026
adab858
bug fix
wenta0g Feb 24, 2026
57957c8
add a comparison benchmark with new target function
wenta0g Feb 24, 2026
16dbabb
updated hard benchmark
iany0 Mar 3, 2026
271dc16
Merge branch 'upstream/main' into main
iany0 Mar 3, 2026
eaec977
text-embedding-004 breaks if set to global
iany0 Mar 4, 2026
84f60d2
test subset for testing endpoint
iany0 Mar 5, 2026
bb4066a
resolved the issue when closing connection
iany0 Mar 5, 2026
1f48c97
updated benchmark
iany0 Mar 5, 2026
8829933
update google adk
iany0 Mar 6, 2026
43b6e18
update google adk dependancy
iany0 Mar 7, 2026
11f5176
Revert "update google adk dependancy"
iany0 Mar 7, 2026
5f14277
Revert "update google adk"
iany0 Mar 7, 2026
46bdba0
attempted to fix incompatibility issue
iany0 Mar 7, 2026
57e9fc8
attempted to fix incompatibility issue and updated benchmark
iany0 Mar 7, 2026
43ff993
finalized benchmark
iany0 Mar 7, 2026
ef16609
retry logic for query
iany0 Mar 10, 2026
5a2c254
indentation
iany0 Mar 10, 2026
e7dc433
hard-45 benchmark
iany0 Mar 11, 2026
5acaf1e
disable function analyzer by default
iany0 Mar 13, 2026
713fb7c
Update benchmarks and helper scripts, removing obsolete targets
iany0 Mar 13, 2026
5d217e5
finalized hard benchmarks
iany0 Mar 14, 2026
3d08137
filtered by model
iany0 Mar 16, 2026
ea03532
fixed argument issue
iany0 Mar 17, 2026
72de930
fixed truncation
iany0 Mar 18, 2026
3cbc2d3
refined truncation logic
iany0 Mar 20, 2026
2b23dc9
added sleep and refined truncation logic
iany0 Mar 21, 2026
9368938
fixed logger
iany0 Mar 22, 2026
206bfc8
removed truncation log
iany0 Mar 22, 2026
f92b84f
arg for filter by date
iany0 Apr 22, 2026
7499f72
added debug print
iany0 Apr 22, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 60 additions & 0 deletions agent/base_agent.py
Original file line number Diff line number Diff line change
Expand Up @@ -31,6 +31,51 @@

import logger
import utils

def patch_gemini_thought_signature():
try:
from google.adk.models.google_llm import Gemini
from google.genai import types

orig_preprocess = Gemini._preprocess_request
async def patched_preprocess(self, llm_request):
await orig_preprocess(self, llm_request)
if getattr(llm_request, 'contents', None) is None:
return

needs_flattening = False
for content in llm_request.contents:
if getattr(content, 'parts', None) is None:
continue
if any(getattr(p, 'function_call', None) and not getattr(p, 'thought_signature', None) for p in content.parts):
needs_flattening = True
break

if needs_flattening:
for content in llm_request.contents:
if getattr(content, 'parts', None) is None:
continue
new_parts = []
for p in content.parts:
if getattr(p, 'function_call', None):
new_parts.append(types.Part(
text=f"Model called tool `{p.function_call.name}` with parameters: {p.function_call.args}"
))
elif getattr(p, 'function_response', None):
new_parts.append(types.Part(
text=f"Tool `{p.function_response.name}` returned result: {p.function_response.response}"
))
else:
new_parts.append(p)
content.parts = new_parts

Gemini._preprocess_request = patched_preprocess
except Exception as e:
import logging
logging.getLogger(__name__).warning("Failed to patch Gemini: %s", e)

patch_gemini_thought_signature()

from data_prep import introspector
from experiment import benchmark as benchmarklib
from llm_toolkit.models import LLM, VertexAIModel
Expand Down Expand Up @@ -69,6 +114,7 @@ def get_tool(self, tool_name: str) -> Optional[BaseTool]:
def chat_llm_with_tools(self, client: Any, prompt: Optional[Prompt], tools,
trial) -> Any:
"""Chat with LLM with tools."""
self._jitter_for_rate_limit(trial=trial)
logger.info(
'<CHAT WITH TOOLS PROMPT:ROUND %02d>%s</CHAT PROMPT:ROUND %02d>',
trial,
Expand All @@ -89,6 +135,7 @@ def chat_llm_with_tools(self, client: Any, prompt: Optional[Prompt], tools,
def chat_llm(self, cur_round: int, client: Any, prompt: Prompt,
trial: int) -> str:
"""Chat with LLM."""
self._jitter_for_rate_limit(trial=trial)
logger.info('<CHAT PROMPT:ROUND %02d>%s</CHAT PROMPT:ROUND %02d>',
cur_round,
prompt.gettext(),
Expand All @@ -104,6 +151,7 @@ def chat_llm(self, cur_round: int, client: Any, prompt: Prompt,

def ask_llm(self, cur_round: int, prompt: Prompt, trial: int) -> str:
"""Ask LLM."""
self._jitter_for_rate_limit(trial=trial)
logger.info('<ASK PROMPT:ROUND %02d>%s</ASK PROMPT:ROUND %02d>',
cur_round,
prompt.gettext(),
Expand Down Expand Up @@ -201,6 +249,17 @@ def _container_handle_bash_commands(self, response: str, tool: BaseTool,
prompt.append(prompt_text)
return prompt

def _jitter_for_rate_limit(
self,
trial: int,
min_sec: float = 1.0,
max_sec: float = 10.0,
) -> None:
"""Sleeps for a very short random duration to avoid 429 collisions."""
duration = random.uniform(min_sec, max_sec)
logger.debug('Jittering for %f before query', duration, trial=trial)
time.sleep(duration)

def _sleep_random_duration(
self,
trial: int,
Expand Down Expand Up @@ -388,6 +447,7 @@ def get_xml_representation(self, response: Optional[dict]) -> str:
def chat_llm(self, cur_round: int, client: Any, prompt: Prompt,
trial: int) -> Any:
"""Call the agent with the given prompt, running async code in sync."""
self._jitter_for_rate_limit(trial=trial)

self.round = cur_round

Expand Down
94 changes: 94 additions & 0 deletions agent/memory_enhancer.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,94 @@
# Copyright 2025 Google LLC
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
"""An LLM agent to improve a fuzz target's runtime performance.
Use it as a usual module locally, or as script in cloud builds.
"""
import logger
from agent.memory_prototyper import MemoryPrototyper
from llm_toolkit.prompt_builder import (CoverageEnhancerTemplateBuilder,
CrashEnhancerTemplateBuilder,
EnhancerTemplateBuilder,
JvmFixingBuilder)
from llm_toolkit.prompts import Prompt, TextPrompt
from results import AnalysisResult, BuildResult, Result


class MemoryEnhancer(MemoryPrototyper):
"""The Agent to refine a compilable fuzz target for higher coverage."""

def _initial_prompt(self, results: list[Result]) -> Prompt:
"""Constructs initial prompt of the agent."""
last_result = results[-1]
benchmark = last_result.benchmark

if not isinstance(last_result, AnalysisResult):
logger.error('The last result in Enhancer is not AnalysisResult: %s',
results,
trial=self.trial)
return Prompt()

last_build_result = None
for result in results[::-1]:
if isinstance(result, BuildResult):
last_build_result = result
break
if not last_build_result:
logger.error('Unable to find the last build result in Enhancer : %s',
results,
trial=self.trial)
return Prompt()

function_requirements = self.get_function_requirements()

if benchmark.language == 'jvm':
# TODO: Do this in a separate agent for JVM coverage.
builder = JvmFixingBuilder(self.llm, benchmark,
last_result.run_result.fuzz_target_source, [])
prompt = builder.build([], None, None)
else:
# TODO(dongge): Refine this logic.
if last_result.semantic_result:
error_desc, errors = last_result.semantic_result.get_error_info()
builder = EnhancerTemplateBuilder(self.llm, benchmark,
last_build_result, error_desc, errors)
elif last_result.crash_result:
crash_result = last_result.crash_result
context_result = last_result.crash_context_result
builder = CrashEnhancerTemplateBuilder(self.llm, benchmark,
last_build_result, crash_result,
context_result)
elif last_result.coverage_result:
builder = CoverageEnhancerTemplateBuilder(
self.llm,
benchmark,
last_build_result,
coverage_result=last_result.coverage_result)
else:
logger.error(
'Last result does not contain either semantic result or '
'coverage result',
trial=self.trial)
# TODO(dongge): Give some default initial prompt.
prompt = TextPrompt(
'Last result does not contain either semantic result or '
'coverage result')
return prompt
prompt = builder.build(example_pair=[],
tool_guides=self.inspect_tool.tutorial(),
project_dir=self.inspect_tool.project_dir,
function_requirements=function_requirements)
# TODO: A different file name/dir.
prompt.save(self.args.work_dirs.prompt)

return prompt
Loading
Loading