hermes-agent/gemini_thinking.sh

python batch_runner.py \
  --dataset_file="source-data/agent_tasks_eval_10.jsonl" \
  --batch_size=1 \
  --run_name="agenttasks_eval_gemini-3-3-10-thinking-2025-11-22" \
  --distribution="science" \
  --prokletor_client="HermesToolClient" \
  --model="gemini-3-pro-preview" \
  --base_url="https://generativelanguage.googleapis.com/v1beta/openai/" \
  --api_key="${GEMINI_API_KEY}" \
  --num_workers=10 \
  --max_turns=60 \
  --verbose \
  --ephemeral_system_prompt="You have access to a variety of tools to help you solve scientific, math, and technology problems presented to you. You can use them in sequence and build off of the results of prior tools you've used results. Always use the terminal or search tool if it can provide additional context, verify formulas, double check concepts and recent studies and understanding, doing all calculations, etc. You should only be confident in your own reasoning, knowledge, or calculations if you've exhaustively used all tools available to you to that can help you verify or validate your work. Always pip install any packages you need to use the python scripts you want to run. If you need to use a tool that isn't available, you can use the terminal tool to install or create it in many cases as well. Do not use the terminal tool to communicate with the user, as they cannot see your commands, only your final response after completing the task. The web search tool only gets you urls and brief descriptions you need to run web extract to actually visit those urls. If you need to check if you have a certain API key available please do so in a way that does not expose the key. For verbose tools like installs please use the quietest version. Also please make sure you include -y in your install commands or the terminal will get stuck at the y/n stage."