Repository navigation
Expand file tree
/
Copy pathagent_config.yaml
More file actions
227 lines (202 loc) · 10.2 KB
/
Copy pathagent_config.yaml
File metadata and controls
227 lines (202 loc) · 10.2 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
# =============================================================================
# AGENT CONFIGURATION
# =============================================================================
# This file contains agent capabilities and tool settings.
# Credentials should be stored in .env file.
# For non agent related config, see config.yaml
llm:
# Model name from src/llms/manifest/models.json
name: "claude-opus-5-5"
flash: "claude-sonnet-5-5"
compaction: "" # blank → same model as flash
fetch: "" # blank → same model as flash
fallback:
- "deepseek-flash"
- "glm-5.2"
sandbox:
# Provider selection: "daytona" (cloud) or "docker" (local).
# - daytona: Cloud-hosted sandboxes via Daytona API. Each workspace gets an isolated
# VM with auto-stop/archive/delete lifecycle. Snapshots are versioned by a hash of
# this config (working_dir, Python version, dependencies, apt packages).
# - docker: Local Docker containers using Dockerfile.sandbox. Useful for development.
# Override at runtime with SANDBOX_PROVIDER env var.
provider: "daytona"
# Daytona cloud sandbox configuration
daytona:
base_url: "https://app.daytona.io/api"
auto_stop_interval: 3600 # 1 hour in seconds
auto_archive_interval: 604800 # 7 days in seconds — keep workspaces "stopped" (fast restart) longer before moving to cold storage
auto_delete_interval: 7776000 # 90 days in seconds — total lifetime before a dormant sandbox is garbage-collected
python_version: "3.12"
# Hash will be automatically appended based on dependencies
snapshot_enabled: true # Recommended: Enable snapshot-based sandbox creation
snapshot_name: "langalpha" # Base name for snapshot (hash will be appended)
snapshot_auto_create: true # Automatically create snapshot if it doesn't exist
# Platform-secret catalog — Daytona provider only (ignored under docker/memory).
# Each entry delivers a backend-owned API key into sandboxes as a managed
# Secret: the sandbox env var holds an opaque placeholder, and Daytona
# substitutes the real value into outbound HTTPS request *headers* for the
# allowlisted hosts only. Declared per deployment (not code); an entry is the
# opt-in and works in OSS mode too. Uncomment the FMP example to enable —
# boot then fail-closes unless DAYTONA_SECRET_NAMESPACE and the source env
# var are set.
# platform_secrets:
# - source_env_var: "FMP_API_KEY" # read from the host .env
# sandbox_env_var: "FMP_API_KEY" # placeholder-bearing var in the sandbox
# name_suffix: "platform-fmp-api-key" # Secret name = <DAYTONA_SECRET_NAMESPACE>-<suffix>
# description: "Platform FMP API key"
# hosts: ["financialmodelingprep.com"] # egress substitution allowlist
# Docker local sandbox configuration
# Ignored when provider is "daytona". Override with DOCKER_SANDBOX_* env vars.
docker:
image: "langalpha-sandbox:latest" # Build with: docker build -f Dockerfile.sandbox -t langalpha-sandbox:latest .
memory_limit: "4g"
cpu_count: 2.0
network_mode: "bridge" # "bridge" (default) or "none" for full isolation
dev_mode: false # true = bind-mount host dir (fast local dev, files shared instantly)
# host_work_dir: "/tmp/langalpha-sandbox-work" # Required when dev_mode is true
# volumes: # Extra mounts in "host:container[:ro]" format
# - "/data/datasets:/mnt/datasets:ro"
mcp:
# The servers that ship with LangAlpha are declared by the bundles under
# plugins/, one Agent Plugins package per group -- see plugins/README.md.
# This list is yours: servers you add, and overrides for the ones we ship.
#
# A name no bundle declares ADDS a server. A name a bundle already declares
# OVERRIDES it -- the keys you write here are laid over the shipped server
# and every key you leave out keeps its shipped value. Switching one off is
# two lines, and changing one field never means restating the command:
#
# - name: "price_data" # stop launching a server we ship
# enabled: false
#
# - name: "yf_price" # keep ours, show the agent every tool
# tool_exposure_mode: "detailed"
#
# - name: "fundamentals" # keep our description, run your own build
# command: "uvx"
# args: ["--from", "my-fundamentals-mcp", "serve"]
#
# Overriding here rather than editing plugins/ keeps your choice out of a
# file the next pull overwrites. `env` and `headers` replace the shipped map
# whole rather than merging key by key, so you can take a variable away as
# well as add one.
#
# Third-party stdio servers: always launch isolated -- uvx/npx with pinned
# versions -- never from the shared environment, whose dependency pins
# (including the mcp SDK) will strand them on the next SDK major. Builtin
# servers under plugins/ are era-proof via _bootstrap and run with
# `uv run python`.
servers:
# A no-op override, here so the shape is in front of you. Flip it to
# false and price_data stops launching on the next restart.
- name: "price_data"
enabled: true
# Adding a server of your own looks like this:
# - name: "tavily"
# description: "Web search engine for finding current information online"
# instruction: "Use for web searches, news, research, and real-time information."
# tool_exposure_mode: "summary"
# transport: "stdio"
# command: "npx"
# args: ["-y", "tavily-mcp@latest"]
# env:
# TAVILY_API_KEY: "${TAVILY_API_KEY}"
# Tool discovery settings
tool_discovery_enabled: true
lazy_load: true # Load tools on-demand
cache_duration: 300 # Cache tool metadata for 5 minutes
filesystem:
# Filesystem access configuration for first-class filesystem tools.
# These tools provide direct file and directory operations without code generation.
# Working directory inside the sandbox — the root for all agent file operations.
# Agent sees virtual paths like /results/file.txt which map to {working_directory}/results/file.txt.
# Must match the home directory of the sandbox user (created in Dockerfile.sandbox).
# Changing this value triggers automatic workspace migration: existing workspaces will
# back up files to PostgreSQL and recreate their sandbox with the new working directory
# on next reconnect (tracked via sandbox_config_hash in workspaces.config JSONB).
working_directory: "/home/workspace"
# allowed_directories and denied_directories are auto-derived from working_directory:
# allowed: [working_directory, "/tmp"]
# denied: [working_directory/_internal]
# Override only if you need custom values.
enable_path_validation: true # Validate paths against allowed_directories
storage:
# Cloud storage provider for image/chart/file uploads and avator
# All credentials and settings are loaded from .env file
# Options: s3, r2, oss, none (to disable uploads)
provider: "s3"
logging:
# Logging configuration
level: "INFO" # DEBUG, INFO, WARNING, ERROR, CRITICAL
file: "logs/ptc.log"
agent:
background_auto_wait: false
# Subagent configuration
subagents:
# List of enabled subagents (available: research, general-purpose, data-prep, equity-analyst, report-builder, plus user-defined)
enabled:
- general-purpose
- research
- data-prep
- equity-analyst
- report-builder
# User-defined subagents (optional). Each key becomes a subagent name.
# definitions:
# equity-analyst:
# description: "Specialized equity research analyst for deep-dive company analysis"
# mode: ptc # ptc or flash
# role_prompt: |
# You are a senior equity research analyst...
# tools: [execute_code, filesystem, finance, web_search]
# skills: [automation] # Runtime-loaded via SkillsMiddleware
# preload_skills: [creating-financial-models, xlsx] # Injected into prompt
# max_iterations: 15
# sections:
# workspace_paths: true
# tool_guide: true
# data_processing: true
# visualizations: true
# Compaction Middleware Configuration
# Automatically compacts conversation context when approaching token limits
compaction:
enabled: true # Enable/disable context compaction
token_threshold: 120000 # Trigger full compaction when messages exceed this token count
keep_messages: 10 # Preserve last N messages after compaction
truncate_args_trigger_messages: 40 # Truncate tool args when >= N messages (Tier 1)
truncate_args_keep_messages: 10 # Protect last N messages from arg truncation
truncate_args_max_length: 2000 # Per-arg-value max chars before truncation
# =============================================================================
# TOOL CONFIGURATION
# =============================================================================
# Search API Configuration
# Options: tavily, bocha, serper, exa, parallel
search_api: tavily
# Crawler Configuration
# Web crawling with SafeCrawlerWrapper for circuit breaker and fault tolerance.
# Backend: "scrapling" uses tiered HTTP→browser→stealth fetching.
crawler:
backend: "router" # Options: "scrapling", "router" (adds PDF/YouTube/X extraction)
# Stage-level concurrency caps (replace the old max_concurrent_crawls).
# Tier 1 (curl_cffi) is cheap, Tier 2/3 (browsers) bound RAM at ~400MB each.
http_concurrency: 20 # Max concurrent Tier-1 HTTP fetches
browser_concurrency: 6 # Max concurrent Tier-2/3 browser fetches (~2.4GB RAM ceiling)
page_timeout: 60000 # Page load timeout in ms (overall, all tiers)
# Circuit breaker settings — applied to both per-host and global infra breakers.
# Per-host: a failing Reuters cannot poison Wikipedia.
# Global infra: trips on cross-cutting failures (browser crash, DNS).
circuit_breaker:
failure_threshold: 5
recovery_timeout: 60
success_threshold: 2
# Queue limit — admission-control counter. Returns queue_full immediately when
# in-flight exceeds max_size; no separate slot wait.
queue:
max_size: 100
# Web Fetch Configuration
# Settings for web_fetch tool's sitemap-aware content extraction
web_fetch:
sitemap_enabled: true # Sitemap fetching for URL suggestions
sitemap_max_urls: 100 # Max URLs to fetch from sitemap
sitemap_max_examples: 3 # Max example URLs per path prefix in summary
sitemap_timeout: 10 # Timeout in seconds for sitemap fetch