Repository navigation
Expand file tree
/
Copy pathmain.py
More file actions
3494 lines (3195 loc) · 165 KB
/
Copy pathmain.py
File metadata and controls
3494 lines (3195 loc) · 165 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
import os
os.environ["OPENBLAS_NUM_THREADS"] = "1"
os.environ["MKL_NUM_THREADS"] = "1"
os.environ["NUMEXPR_NUM_THREADS"] = "1"
os.environ["OMP_NUM_THREADS"] = "1"
import platform as _platform
import subprocess as _subprocess
import warnings
os.environ["QT_LOGGING_RULES"] = "*.debug=false;qt.qpa.*=false;qt.text.*=false;qt.qpa.window=false"
warnings.filterwarnings("ignore")
try:
import colorama
colorama.just_fix_windows_console()
except Exception:
try:
import ctypes
kernel32 = ctypes.windll.kernel32
hStdOut = kernel32.GetStdHandle(-11)
mode = ctypes.c_ulong()
kernel32.GetConsoleMode(hStdOut, ctypes.byref(mode))
kernel32.SetConsoleMode(hStdOut, mode.value | 0x0004)
except Exception:
pass
# ── Nuclear: force CREATE_NO_WINDOW on EVERY subprocess call on Windows ───────
# This patches Popen itself, so no per-file flag is needed anywhere.
if _platform.system() == "Windows":
_OrigPopen = _subprocess.Popen
class _Popen(_OrigPopen):
def __init__(self, args, **kw):
kw["creationflags"] = kw.get("creationflags", 0) | _subprocess.CREATE_NO_WINDOW
kw.pop("startupinfo", None) # drop any stale/shared STARTUPINFO
super().__init__(args, **kw)
_subprocess.Popen = _Popen
# ─────────────────────────────────────────────────────────────────────────────
import asyncio
import re
import threading
import queue
import time
import json
import sys
import traceback
from collections import deque
from datetime import datetime
from pathlib import Path
import sounddevice as sd
# sounddevice versions that still assign to ndarray.shape trigger a NumPy 2.5
# deprecation warning from inside the stream callback. It is a dependency warning,
# not an application failure; keep TITAN's console clean until sounddevice is updated.
warnings.filterwarnings("ignore", message=r"Setting the shape on a NumPy array has been deprecated in NumPy 2\.5\.", category=DeprecationWarning, module=r"sounddevice")
warnings.filterwarnings("ignore", category=DeprecationWarning, module=r"sounddevice")
from google import genai
from google.genai import types
from ui import TitanUI
from memory.memory_manager import (
load_memory, update_memory, format_memory_for_prompt,
save_session_summary, pop_last_session,
)
from actions.file_processor import file_processor
from actions.flight_finder import flight_finder
from actions.open_app import open_app
from actions.weather_report import weather_action
from actions.send_message import send_message
from actions.reminder import reminder
from actions.computer_settings import computer_settings
from actions.screen_processor import _capture_camera, _capture_screen
from actions.youtube_video import youtube_video
from actions.desktop import desktop_control
from actions.browser_control import browser_control
from actions.file_controller import file_controller
from actions.code_helper import code_helper
from actions.dev_agent import dev_agent
from actions.web_search import web_search as web_search_action
from actions.computer_control import computer_control
from actions.game_updater import game_updater
from actions.shadow_link import shadow_link_control
from actions.system_monitor import SystemMonitor, get_system_status
from actions.proactive import ProactiveEngine
from actions.background_monitor import (
add_monitor, remove_monitor, list_monitors, check_all as monitor_check_all,
)
from actions.voice_face_id import (
voice_face_id, verify_voice, startup_authenticate,
handle_security_command, get_security_config as get_sec_config,
)
from actions.web_search import _news as _fetch_news_sync
from memory.config_manager import (
get_brief_enabled,
get_input_device, get_output_device,
save_input_device, save_output_device,
)
from memory.work_pad import (
run_work_pad,
format_for_prompt as format_pad_for_prompt,
track_long_task,
absorb_work_output,
load_pad,
_next_step,
)
from task_control import set_ui_hooks
from core.plugin_loader import discover_plugins
from core.skill_registry import discover_skills
from core.exec import run_command, RUN_COMMAND_DECLARATION
from core.agent_loop import main_agent_loop
from core.session_log import session_logger
from core.todo_engine import todo_engine, TODO_WRITE_DECLARATION, TODO_READ_DECLARATION
from core.goal_manager import goal_manager, GOAL_TOOLS_DECLARATIONS
from core.scheduler import scheduler, SCHEDULE_DECLARATION
from core.interaction import interaction_engine, ASK_USER_DECLARATION
from core.plan_mode import plan_mode, PLAN_MODE_DECLARATIONS
from core.jobs import job_registry, JOB_TOOLS_DECLARATIONS
from core.fs_tools import (
read_file, write_file, str_replace_editor, grep_search, glob_search,
FS_TOOLS_DECLARATIONS,
)
from core.code_runner import run_python_code, PYTHON_EVAL_DECLARATION
from core.web_tools import web_fetch, WEB_FETCH_DECLARATION
from core.nvidia_brain import run_brain_turn
from core.task_workers import task_workers, TASK_WORKER_DECLARATIONS, running_in_worker
import core.confirm as confirm
import core.audio_devices as audio_devices
from core.undo import undo_last, peek as undo_peek, push_undo, clear as undo_clear
# core.workflow_engine (WORKFLOW_DECLARATION/ralph_loop) and core.subagent_engine
# (SUBAGENT_TOOLS_DECLARATIONS) are intentionally NOT imported anymore. Both were
# earlier, half-built "do complex work" paths that ended up coexisting with
# core.task_workers — workflow_start was pure decoration (ralph_loop just
# returns a string, no actual execution ever happens), and invoke_subagent
# duplicated task_workers.start() under a different name. Having 3 different
# tool names that all mean "do multi-step work", with only some of them
# wired to real execution, is exactly what caused "works sometimes, does
# nothing other times" — the model was picking between them essentially at
# random. task_workers.py is the one real, verification-gated implementation;
# everything now points at it exclusively. See TASK_WORKER_DECLARATIONS below.
def _finalize(raw):
if raw is None:
return "⚠️ No result came back from the tool."
return str(raw).strip() or "Done."
def get_base_dir():
if getattr(sys, "frozen", False):
return Path(sys.executable).parent
return Path(__file__).resolve().parent
BASE_DIR = get_base_dir()
API_CONFIG_PATH = BASE_DIR / "config" / "api_keys.json"
PROMPT_PATH = BASE_DIR / "core" / "prompt.txt"
# ── Full-Brain mode ─────────────────────────────────────────────────────
# When True: Gemini Live gets NO tools and NO planning system prompt — it's
# reduced to STT-in/TTS-out. Every user utterance is instead routed to
# core/nvidia_brain.run_brain_turn(), which calls NVIDIA NIM to plan and
# execute tool calls. Gemini only speaks the final text NVIDIA hands back.
# Flip to False (or set "full_brain_mode": false in config/api_keys.json)
# to go back to Gemini deciding + calling tools itself, same as before.
def _full_brain_enabled() -> bool:
try:
cfg = json.loads(API_CONFIG_PATH.read_text(encoding="utf-8"))
return bool(cfg.get("full_brain_mode", False))
except Exception:
return False
_SILENT_RELAY_PROMPT = (
"You are a text-to-speech relay, not an assistant. You have NO tools "
"and must never claim to have done anything.\n"
"Rules:\n"
"1. You will receive messages starting with '[SPEAK_NOW]' — when you "
"get one, speak that exact text aloud, in your own natural voice/pacing, "
"without adding new facts, opinions, or offers.\n"
"2. If you receive audio/speech from the user directly (not a "
"[SPEAK_NOW] message), do NOT answer it yourself and do NOT guess an "
"answer. Just stay silent / say a brief neutral filler like 'mm' at "
"most — a separate planning system is handling the real answer and "
"will send it to you shortly as [SPEAK_NOW].\n"
"3. Never invent information. Never call a function. Never say a task "
"is done unless it appeared inside a [SPEAK_NOW] message."
)
LIVE_MODELS = [
"models/gemini-2.5-flash-native-audio-preview-12-2025",
"models/gemini-3.1-flash-live-preview",
"models/gemini-2.0-flash-live-001",
]
_live_model_idx = 0
def _get_live_model() -> str:
global _live_model_idx
return LIVE_MODELS[_live_model_idx % len(LIVE_MODELS)]
def _rotate_live_model() -> str:
global _live_model_idx
_live_model_idx += 1
m = LIVE_MODELS[_live_model_idx % len(LIVE_MODELS)]
print(f"[TITAN] 🔄 Switching to model: {m}")
return m
CHANNELS = 1
SEND_SAMPLE_RATE = 16000
RECEIVE_SAMPLE_RATE = 24000
CHUNK_SIZE = 512 # 32 ms @ 16 kHz (was 1024 = 64 ms)
# ── Debug toggle: set True to see VAD / first-audio timings ──
_DEBUG_AUDIO = False # was True — per-chunk prints stall the event loop
_dbg_chunk_count = 0
# Local VAD — stream speech only, then tell Gemini "user stopped"
# Real speech in your logs is RMS 800–1900. Titan speaker leak is 3000+.
# 18-chunk (~0.6s) end-silence was cutting you mid-sentence → "say that again".
_VAD_START_RMS = 640.0
_VAD_HOLD_RMS = 360.0
_VAD_PREROLL = 8 # ~256 ms before speech start
_VAD_END_SILENCE = 42 # ~1.34 s quiet → end of turn
_VAD_MIN_SPEECH = 14 # need ~450 ms of real speech before we may end
_SEND_QUEUE_MAX = 16
_ECHO_TAIL_S = 1.25 # swallow speaker tail after Titan stops
_VAD_SUPPRESS_S = 1.40 # after ESC / interrupt, ignore mic this long
_END_DEBOUNCE_S = 1.40 # don't send audio_stream_end twice in a row
_AUDIO_MIME = "audio/pcm;rate=16000"
# Talk-while-working: higher than keyboard/fan, lower than a real sentence.
_VAD_BUSY_START_RMS = 880.0
_VAD_BUSY_HOLD_RMS = 420.0
_VAD_BUSY_MIN_SPEECH = 16 # ~0.5 s so a click does not cancel the job
_WORK_TOOLS = {
"file_controller", "run_command", "load_skill", "code_helper",
"file_processor", "work_pad", "dev_agent", "web_search",
}
# ── Boss / worker power split ────────────────────────────────────────────────
# The BOSS (Gemini Live) owns the voice: listen, triage, speak, and steer
# workers. It keeps every fast one-shot action - "open Chrome", "who is X",
# "what's my CPU at" must stay instant, and routing those through a worker
# would add an NVIDIA round-trip plus worker spawn to a 200 ms task.
#
# The WORKER (core/task_workers.py, NVIDIA brain) owns everything that builds
# something: writing files, running commands, loading skill instructions,
# planning, verifying. It has strictly MORE power than the boss - the only
# thing it cannot do is speak.
#
# This split is the enforcement mechanism, not a suggestion. Previously
# prompt.txt merely asked the boss to hand off complex work, so whenever the
# boss felt capable it would call write_file + run_command itself, skip the
# skill entirely, and emit a bare default-template deck. Removing those tools
# from the boss makes handoff the only physically available path.
WORKER_ONLY_TOOLS = {
# instruction loading - the boss cannot act on skills, so it must not hold
# them (and the old deferred-load path bound the skill to the WRONG
# transcript, starting workers with tasks like the single word "no")
"load_skill",
# authoring / execution
"write_file", "str_replace_editor", "run_command", "python_eval",
"code_helper", "dev_agent", "file_processor", "file_controller",
"game_updater",
# planning + completion gating (already worker-scoped, listed for clarity)
"task_set_plan", "verify_task_result",
"enter_plan_mode", "exit_plan_mode", "set_goal", "complete_goal",
}
def _dbg(tag: str, msg: str):
if _DEBUG_AUDIO:
from datetime import datetime
ts = datetime.now().strftime("%H:%M:%S.%f")[:-3]
print(f"[{ts}] [{tag}] {msg}")
def _get_api_key() -> str:
with open(API_CONFIG_PATH, "r", encoding="utf-8") as f:
return json.load(f)["gemini_api_key"]
def get_live_clock() -> str:
"""True local wall clock from this PC — not the frozen session-start time."""
now = datetime.now().astimezone()
off = now.strftime("%z")
off_h = f"{off[:3]}:{off[3:]}" if off else ""
tz = now.tzname() or "local"
return (
f"LIVE PC CLOCK (do not guess — this is the real time right now)\n"
f"Date: {now.strftime('%A, %B %d, %Y')}\n"
f"Time: {now.strftime('%I:%M:%S %p')} ({now.strftime('%H:%M:%S')})\n"
f"Timezone: {tz} (UTC{off_h})\n"
f"ISO: {now.isoformat(timespec='seconds')}"
)
def _load_system_prompt() -> str:
prompt_text = ""
try:
prompt_text = PROMPT_PATH.read_text(encoding="utf-8")
except Exception:
prompt_text = (
"You are TITAN, an advanced AI assistant. "
"Be concise, direct, and always use the provided tools to complete tasks. "
"Never simulate or guess results — always call the appropriate tool."
)
try:
from actions.file_controller import _windows_shell_folder
desk_p = _windows_shell_folder("Desktop") or (Path.home() / "Desktop")
down_p = _windows_shell_folder("Downloads") or (Path.home() / "Downloads")
docs_p = _windows_shell_folder("Documents") or (Path.home() / "Documents")
prompt_text += (
f"\n\nUSER DIRECTORIES (ALWAYS use these exact paths when saving or finding files):\n"
f"- Desktop: {str(desk_p).replace(chr(92), '/')}\n"
f"- Downloads: {str(down_p).replace(chr(92), '/')}\n"
f"- Documents: {str(docs_p).replace(chr(92), '/')}\n"
f"Never invent paths like 'E:/auto/shadowos/Desktop' — the real Desktop is at {str(desk_p).replace(chr(92), '/')}.\n"
)
except Exception:
pass
try:
pad_txt = format_pad_for_prompt()
if pad_txt:
prompt_text += "\n\n" + pad_txt
except Exception:
pass
try:
job_txt = format_job_for_prompt()
if job_txt:
prompt_text += "\n\n" + job_txt
except Exception:
pass
prompt_text += (
"\n\nTASK CHECKLIST PROTOCOL: Use `todo_write` as your dynamic checklist to plan and track multi-step tasks. "
"When starting a complex task (documents, presentations, multi-step code), call `todo_write` (action='set_plan', title='...', steps=['step 1', 'step 2', ...]) "
"to break down what needs to be done. As you finish each step, call `todo_write` (action='update_step', step_id=N, status='completed') "
"so the live checklist on the Workpad is always updated for sir.\n"
)
prompt_text += (
"\n\nCLOCK RULE: The date/time printed at session start goes STALE. "
"Whenever the user asks the time, date, today, tomorrow, or you need the time for a reminder, "
"you MUST call get_clock first and speak THAT result. Never invent or remember the time.\n"
)
prompt_text += (
"\n\nREASONING RULE: You respond fast and directly for normal conversation and tool calls — "
"do not overthink simple requests. But if the user asks for something that genuinely needs "
"multi-step planning, comparison, or reasoning through constraints (e.g. scheduling, strategy, "
"complex decisions, multi-part math), call the deep_think tool instead of answering off the cuff. "
"Briefly tell the user you're thinking it through first, then call deep_think and speak its result "
"naturally in your own voice — don't read it robotically."
)
prompt_text += (
"\n\nMID-TASK TALK: If sir talks while a tool is running, the job STOPS and does not save. "
"You will then hear what he said. Follow THAT. "
"stop / wait / don't touch → do not call the same tool again. "
"A specific change ('don't touch the cover', 'change the title') → only that change. "
"Do not restart the old job unless he says continue.\n"
"DOC QUALITY: Professional/academic reports = Times New Roman 12pt 1.5 ON THE TEMPLATE "
"(via the docx skill). NEVER a generic navy/Segoe AI report. After write, verify formatting and fonts.\n"
"STEP-BY-STEP WORKFLOW:\n"
"For any complex, creative, or multi-step request (documents, presentations, analysis, coding):\n"
"1. First Plan on Workpad: Call `todo_write` (action='set_plan', title='...', steps=[...]) to break down the task.\n"
"2. Execute Step-by-Step: Systematically execute each step using your real tools (load_skill, run_command, file_controller).\n"
"3. Mark Progress: Update step statuses via `todo_write` (action='update_step', step_id=N, status='completed').\n"
"4. Verify & Deliver: Confirm that the output is ready before reporting to sir.\n"
)
prompt_text += (
"\n\nCOMPLETION HONESTY — THIS IS A HARD RULE, NOT A STYLE PREFERENCE:\n"
"Never say a file/task is done, saved, updated, ready, improved, or 'तैयार है' / 'बना दिया है' "
"unless a tool call (run_command, file_controller, etc.) actually ran and SUCCEEDED in "
"*this same turn or the immediately preceding one*. Saying it's done without that is a lie to sir, "
"not politeness — never do it, even if he sounds impatient or angry.\n"
"If work is not actually finished yet: say plainly that you are calling the tool now, then CALL IT — "
"in the same turn. Do not respond with only reassurance ('थोड़ा समय लगेगा', 'प्रक्रिया चल रही है', "
"'कर रहा हूँ') with no tool call behind it. A turn with no tool call and no new information is a "
"wasted turn sir can hear — if you are not ready to speak the real answer, you are ready to call a tool.\n"
"If sir asks 'did you actually do something' / 'kuch kiya kya' / 'दिख नहीं रहा': answer honestly from "
"the ACTUAL last tool result, not from what you said earlier. If the last real attempt failed or you "
"never called the tool, say so directly ('माफ़ कीजिए sir, अभी तक नहीं बना — अभी बनाता हूँ') and then "
"immediately call the tool — don't apologize in words only and stall again.\n"
"Every one of your spoken turns about an in-progress file job must either (a) contain a tool call, or "
"(b) report the exit_code/result of a tool call that just ran. Never a bare reassurance sentence with "
"neither.\n"
)
return prompt_text
_CTRL_RE = re.compile(r"<ctrl\d+>", re.IGNORECASE)
def _looks_tool_error(text: str) -> bool:
t = (text or "").strip()
if not t:
return False
head = t[:90].lower()
return (
t.startswith("❌")
or head.startswith("error")
or " failed" in head
or "traceback" in t.lower()
or "unexpected" in head
or "got an unexpected" in t.lower()
)
def _clean_transcript(text: str) -> str:
text = _CTRL_RE.sub("", text)
text = re.sub(r"[\x00-\x08\x0b-\x1f]", "", text)
return text.strip()
TOOL_DECLARATIONS = [
{
"name": "ui_automation",
"description": (
"Interacts with Windows desktop apps programmatically via Windows Accessibility UIA tree "
"WITHOUT moving or stealing the user's physical mouse. "
"Actions: 'click' (click button/element), 'type' (paste text into input box), "
"'get_text' (read text from element), 'dump_tree' (inspect UI accessibility tree)."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "click | type | get_text | dump_tree"},
"app_name": {"type": "STRING", "description": "Target application name (e.g. 'Calculator', 'Notepad', 'Chrome')"},
"element_name": {"type": "STRING", "description": "Title or button label to interact with (e.g. 'Seven', 'Edit', 'Submit')"},
"text": {"type": "STRING", "description": "Text to type into an edit box or search query for dump_tree"},
"index": {"type": "INTEGER", "description": "Index if multiple elements match the same name (default 0)"}
},
"required": ["action", "app_name"]
}
},
{
"name": "open_app",
"description": (
"Opens any application on the computer. "
"Use this whenever the user asks to open, launch, or start any app, "
"website, or program. Always call this tool — never just say you opened it."
),
"parameters": {
"type": "OBJECT",
"properties": {
"app_name": {
"type": "STRING",
"description": "Exact name of the application (e.g. 'WhatsApp', 'Chrome', 'Spotify')"
}
},
"required": ["app_name"]
}
},
{
"name": "web_search",
"description": (
"Searches the web. Use for ANY question about current facts, events, prices, "
"or topics — always prefer this over guessing. "
"Modes: 'search' (default), 'news' (latest headlines on a topic), "
"'research' (deep comprehensive answer), 'price' (product cost lookup), "
"'compare' (side-by-side comparison of items)."
),
"parameters": {
"type": "OBJECT",
"properties": {
"query": {"type": "STRING", "description": "Search query or topic"},
"mode": {"type": "STRING", "description": "search | news | research | price | compare"},
"items": {"type": "ARRAY", "items": {"type": "STRING"}, "description": "Items to compare (compare mode)"},
"aspect": {"type": "STRING", "description": "Comparison aspect: price | specs | reviews | features"},
},
"required": ["query"]
}
},
{
"name": "system_status",
"description": (
"Returns real-time system metrics: CPU usage, RAM, GPU load, CPU temperature, "
"uptime, and process count. Use when the user asks about computer performance, "
"temperature, memory, or resource usage."
),
"parameters": {
"type": "OBJECT",
"properties": {},
}
},
{
"name": "set_titan_microphone",
"description": "Mutes or unmutes TITAN's own microphone. Call this when the user says 'mute yourself', 'stop listening', 'unmute yourself', or 'start listening'. This controls TITAN only, not the Windows system volume.",
"parameters": {
"type": "OBJECT",
"properties": {
"muted": {"type": "BOOLEAN", "description": "True to stop TITAN listening; false to resume listening."}
},
"required": ["muted"]
}
},
{
"name": "undo_last_action",
"description": (
"Reverses the most recent reversible action TITAN performed (a setting change, "
"a mic mute, etc). Call this the instant the user says 'undo', 'undo that', 'put "
"it back', or 'go back'. Runs immediately — never ask for confirmation first."
),
"parameters": {
"type": "OBJECT",
"properties": {
"peek_only": {
"type": "BOOLEAN",
"description": "True to just report what WOULD be undone, without doing it (e.g. user asks 'what can you undo').",
}
},
}
},
{
"name": "set_titan_audio_device",
"description": (
"Lists or switches which microphone or speakers TITAN uses. Call with action='list' "
"when the user asks what microphones/speakers are available, or action='set' with "
"kind and device_name to switch TITAN to a specific device. The change applies the "
"next time TITAN reconnects (usually within a few seconds)."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "'list' or 'set'"},
"kind": {"type": "STRING", "description": "'input' (microphone) or 'output' (speakers)"},
"device_name": {"type": "STRING", "description": "Exact device name from the list — required for action='set'."},
},
"required": ["action"]
}
},
{
"name": "get_clock",
"description": (
"Returns this PC's LIVE local date, time, weekday and timezone RIGHT NOW. "
"MUST be called for any time/date question or when computing reminder times. "
"Never guess the time from memory or from the session-start clock."
),
"parameters": {"type": "OBJECT", "properties": {}, "required": []}
},
TODO_WRITE_DECLARATION,
TODO_READ_DECLARATION,
SCHEDULE_DECLARATION,
*GOAL_TOOLS_DECLARATIONS,
ASK_USER_DECLARATION,
*PLAN_MODE_DECLARATIONS,
*JOB_TOOLS_DECLARATIONS,
*TASK_WORKER_DECLARATIONS,
*FS_TOOLS_DECLARATIONS,
PYTHON_EVAL_DECLARATION,
WEB_FETCH_DECLARATION,
{
"name": "weather_report",
"description": "Gives the weather report to user",
"parameters": {
"type": "OBJECT",
"properties": {
"city": {"type": "STRING", "description": "City name"}
},
"required": ["city"]
}
},
{
"name": "send_message",
"description": "Sends a text message via WhatsApp, Telegram, or other messaging platform.",
"parameters": {
"type": "OBJECT",
"properties": {
"receiver": {"type": "STRING", "description": "Recipient contact name"},
"message_text": {"type": "STRING", "description": "The message to send"},
"platform": {"type": "STRING", "description": "Platform: WhatsApp, Telegram, etc."}
},
"required": ["receiver", "message_text", "platform"]
}
},
{
"name": "reminder",
"description": "Sets a timed reminder using Task Scheduler.",
"parameters": {
"type": "OBJECT",
"properties": {
"date": {"type": "STRING", "description": "Date in YYYY-MM-DD format"},
"time": {"type": "STRING", "description": "Time in HH:MM format (24h)"},
"message": {"type": "STRING", "description": "Reminder message text"}
},
"required": ["date", "time", "message"]
}
},
{
"name": "youtube_video",
"description": (
"Controls YouTube. Use for: playing videos, summarizing a video's content, "
"getting video info, or showing trending videos."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "play | summarize | get_info | trending (default: play)"},
"query": {"type": "STRING", "description": "Search query for play action"},
"save": {"type": "BOOLEAN", "description": "Save summary to Notepad (summarize only)"},
"region": {"type": "STRING", "description": "Country code for trending e.g. TR, US"},
"url": {"type": "STRING", "description": "Video URL for get_info action"},
},
"required": []
}
},
{
"name": "screen_process",
"description": (
"Captures the screen or webcam image and lets you analyze it. "
"MUST be called when user asks what is on screen, what you see, "
"look at camera, analyze my screen, etc. "
"You have NO visual ability without this tool. "
"After the image is captured it is sent directly to you — describe what you see and answer the user's question. "
"When using camera: the live view stays open until user says close it or calls close_camera."
),
"parameters": {
"type": "OBJECT",
"properties": {
"angle": {"type": "STRING", "description": "'screen' to capture display, 'camera' for webcam. Default: 'screen'"},
"text": {"type": "STRING", "description": "The question or instruction about the captured image"}
},
"required": ["text"]
}
},
{
"name": "close_camera",
"description": (
"Closes the live camera view shown on screen. "
"Call when user says: close camera, stop camera, turn off camera, "
"kamerayı kapat, kapat, creepy, etc."
),
"parameters": {"type": "OBJECT", "properties": {}, "required": []}
},
{
"name": "computer_settings",
"description": (
"Controls the computer: volume, brightness, window management, keyboard shortcuts, "
"typing text on screen, closing apps, fullscreen, dark mode, WiFi, restart, shutdown, "
"scrolling, tab management, zoom, screenshots, lock screen, refresh/reload page. "
"Use for ANY single computer control command."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "The action to perform"},
"description": {"type": "STRING", "description": "Natural language description of what to do"},
"value": {"type": "STRING", "description": "Optional value: volume level, text to type, etc."}
},
"required": []
}
},
{
"name": "browser_control",
"description": (
"Controls any web browser. Use for: opening websites, searching the web, "
"clicking elements, filling forms, scrolling, screenshots, navigation, any web-based task. "
"Simple open/search requests launch the user's own browser normally (their real profile "
"and logged-in accounts); interactive actions (click, type, fill_form...) attach an "
"automation browser. "
"Always pass the 'browser' parameter when the user specifies a browser (e.g. 'open in Edge', "
"'use Firefox', 'open Chrome'). Multiple browsers can run simultaneously."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "go_to | search | click | type | scroll | fill_form | smart_click | smart_type | get_text | get_url | press | new_tab | close_tab | screenshot | back | forward | reload | switch | list_browsers | close | close_all"},
"browser": {"type": "STRING", "description": "Target browser: chrome | edge | firefox | opera | operagx | brave | vivaldi | safari. Omit to use the currently active browser."},
"url": {"type": "STRING", "description": "URL for go_to / new_tab action"},
"query": {"type": "STRING", "description": "Search query for search action"},
"engine": {"type": "STRING", "description": "Search engine: google | bing | duckduckgo | yandex (default: google)"},
"selector": {"type": "STRING", "description": "CSS selector for click/type"},
"text": {"type": "STRING", "description": "Text to click or type"},
"description": {"type": "STRING", "description": "Element description for smart_click/smart_type"},
"direction": {"type": "STRING", "description": "up | down for scroll"},
"amount": {"type": "INTEGER", "description": "Scroll amount in pixels (default: 500)"},
"key": {"type": "STRING", "description": "Key name for press action (e.g. Enter, Escape, F5)"},
"path": {"type": "STRING", "description": "Save path for screenshot"},
"incognito": {"type": "BOOLEAN", "description": "Open in private/incognito mode"},
"clear_first": {"type": "BOOLEAN", "description": "Clear field before typing (default: true)"},
},
"required": ["action"]
}
},
{
"name": "file_controller",
"description": "Manages files and folders: list, create, delete, move, copy, rename, read, write, find, disk usage.",
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "list | create_file | create_folder | delete | move | copy | rename | read | write | find | largest | disk_usage | organize_desktop | info"},
"path": {"type": "STRING", "description": "File/folder path or shortcut: desktop, downloads, documents, home"},
"destination": {"type": "STRING", "description": "Destination path for move/copy"},
"new_name": {"type": "STRING", "description": "New name for rename"},
"content": {"type": "STRING", "description": "Content for create_file/write"},
"name": {"type": "STRING", "description": "File name to search for"},
"extension": {"type": "STRING", "description": "File extension to search (e.g. .pdf)"},
"count": {"type": "INTEGER", "description": "Number of results for largest"},
},
"required": ["action"]
}
},
{
"name": "desktop_control",
"description": "Controls the desktop: wallpaper, organize, clean, list, stats.",
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "wallpaper | wallpaper_url | organize | clean | list | stats | task"},
"path": {"type": "STRING", "description": "Image path for wallpaper"},
"url": {"type": "STRING", "description": "Image URL for wallpaper_url"},
"mode": {"type": "STRING", "description": "by_type or by_date for organize"},
"task": {"type": "STRING", "description": "Natural language desktop task"},
},
"required": ["action"]
}
},
{
"name": "code_helper",
"description": "Writes, edits, explains, runs, or builds code files.",
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "write | edit | explain | run | build | auto (default: auto)"},
"description": {"type": "STRING", "description": "What the code should do or what change to make"},
"language": {"type": "STRING", "description": "Programming language (default: python)"},
"output_path": {"type": "STRING", "description": "Where to save the file"},
"file_path": {"type": "STRING", "description": "Path to existing file for edit/explain/run/build"},
"code": {"type": "STRING", "description": "Raw code string for explain"},
"args": {"type": "STRING", "description": "CLI arguments for run/build"},
"timeout": {"type": "INTEGER", "description": "Execution timeout in seconds (default: 30)"},
},
"required": ["action"]
}
},
{
"name": "dev_agent",
"description": "Builds complete multi-file projects from scratch: plans, writes files, installs deps, opens VSCode, runs and fixes errors.",
"parameters": {
"type": "OBJECT",
"properties": {
"description": {"type": "STRING", "description": "What the project should do"},
"language": {"type": "STRING", "description": "Programming language (default: python)"},
"project_name": {"type": "STRING", "description": "Optional project folder name"},
"timeout": {"type": "INTEGER", "description": "Run timeout in seconds (default: 30)"},
},
"required": ["description"]
}
},
{
"name": "computer_control",
"description": "FALLBACK raw mouse/keyboard control. ALWAYS USE 'ui_automation' FIRST for clicking or typing in desktop apps (Notepad, Calculator, Word) so the mouse is NOT hijacked. Use computer_control ONLY for global hotkeys, scrolling, or raw fallback mouse movement.",
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "type | smart_type | click | double_click | right_click | hotkey | press | scroll | move | copy | paste | screenshot | wait | clear_field | focus_window | screen_find | screen_click | random_data | user_data"},
"text": {"type": "STRING", "description": "Text to type or paste"},
"x": {"type": "INTEGER", "description": "X coordinate"},
"y": {"type": "INTEGER", "description": "Y coordinate"},
"keys": {"type": "STRING", "description": "Key combination e.g. 'ctrl+c'"},
"key": {"type": "STRING", "description": "Single key e.g. 'enter'"},
"direction": {"type": "STRING", "description": "up | down | left | right"},
"amount": {"type": "INTEGER", "description": "Scroll amount (default: 3)"},
"seconds": {"type": "NUMBER", "description": "Seconds to wait"},
"title": {"type": "STRING", "description": "Window title for focus_window"},
"description": {"type": "STRING", "description": "Element description for screen_find/screen_click"},
"type": {"type": "STRING", "description": "Data type for random_data"},
"field": {"type": "STRING", "description": "Field for user_data: name|email|city"},
"clear_first": {"type": "BOOLEAN", "description": "Clear field before typing (default: true)"},
"path": {"type": "STRING", "description": "Save path for screenshot"},
},
"required": ["action"]
}
},
{
"name": "shadow_link",
"description": (
"Controls the user's REAL open Chrome browser tab via the Shadow-Link extension bridge. "
"Use for: getting current Chrome tab URL, navigating open Chrome tab, clicking web elements, "
"typing text into Chrome inputs, scrolling, and extracting page content."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "get_url | navigate | click | type | scroll | extract"},
"url": {"type": "STRING", "description": "Target URL for navigate"},
"text": {"type": "STRING", "description": "Text to type into input or search"},
"selector": {"type": "STRING", "description": "CSS selector for click"}
},
"required": []
}
},
{
"name": "game_updater",
"description": (
"THE ONLY tool for ANY Steam or Epic Games request. "
"Use for: installing, downloading, updating games, listing installed games, "
"checking download status, scheduling updates. "
"ALWAYS call directly for any Steam/Epic/game request. "
"NEVER use browser_control or web_search for Steam/Epic."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {"type": "STRING", "description": "update | install | list | download_status | schedule | cancel_schedule | schedule_status (default: update)"},
"platform": {"type": "STRING", "description": "steam | epic | both (default: both)"},
"game_name": {"type": "STRING", "description": "Game name (partial match supported)"},
"app_id": {"type": "STRING", "description": "Steam AppID for install (optional)"},
"hour": {"type": "INTEGER", "description": "Hour for scheduled update 0-23 (default: 3)"},
"minute": {"type": "INTEGER", "description": "Minute for scheduled update 0-59 (default: 0)"},
"shutdown_when_done": {"type": "BOOLEAN", "description": "Shut down PC when download finishes"},
},
"required": []
}
},
{
"name": "flight_finder",
"description": "Searches Google Flights and speaks the best options.",
"parameters": {
"type": "OBJECT",
"properties": {
"origin": {"type": "STRING", "description": "Departure city or airport code"},
"destination": {"type": "STRING", "description": "Arrival city or airport code"},
"date": {"type": "STRING", "description": "Departure date (any format)"},
"return_date": {"type": "STRING", "description": "Return date for round trips"},
"passengers": {"type": "INTEGER", "description": "Number of passengers (default: 1)"},
"cabin": {"type": "STRING", "description": "economy | premium | business | first"},
"save": {"type": "BOOLEAN", "description": "Save results to Notepad"},
},
"required": ["origin", "destination", "date"]
}
},
{
"name": "manage_monitor",
"description": (
"Add, remove, or list background monitoring topics. "
"TITAN checks these topics once a day and alerts the user when there is a new development. "
"Use 'add' when the user says 'monitor X', 'track X', 'follow X'. "
"Use 'remove' when the user says 'stop monitoring X'. "
"Use 'list' when the user asks what is being monitored. "
"Do NOT add crypto, financial, or trading topics."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {
"type": "STRING",
"description": "add | remove | list",
},
"topic": {
"type": "STRING",
"description": "Topic to monitor or stop monitoring (e.g. 'space exploration', 'AI news')",
},
},
"required": ["action"],
},
},
{
"name": "voice_face_id",
"description": (
"Voice ID and Face ID Security System. "
"Use action='enroll_face' to capture webcam and register owner face. "
"Use action='enroll_voice' to record 3s of mic and register owner voice. "
"Use action='status' to check security status. "
"Use action='toggle' with enable=True to turn ON locks (voice/face/gate). "
"Use action='toggle' with enable=False to turn OFF locks (requires face verification). "
"You do NOT know the passcode. Never make up rules about passcode format. "
"For passcode setup, tell user to type 'set pin XXXX' in the UI input box."
),
"parameters": {
"type": "OBJECT",
"properties": {
"action": {
"type": "STRING",
"description": "enroll_face | enroll_voice | status | toggle"
},
"mode": {
"type": "STRING",
"description": "voice | face | gate | both (for toggle action)"
},
"enable": {
"type": "BOOLEAN",
"description": "True to turn ON lock, False to turn OFF lock (requires face auth)"
}
},
"required": ["action"]
}
},
{
"name": "shutdown_titan",
"description": (
"Shuts down the assistant completely. "
"Call this when the user expresses intent to end the conversation, "
"close the assistant, say goodbye, or stop Titan. "
"The user can say this in ANY language."
),
"parameters": {
"type": "OBJECT",
"properties": {},
}
},
{
"name": "file_processor",
"description": (
"Processes any file that the user has uploaded or dropped onto the interface. "
"Use this when the user refers to an uploaded file and wants an action on it. "
"Supports: images (describe/ocr/resize/compress/convert), "
"PDFs (summarize/extract_text/to_word), "
"Word docs & text files (summarize/fix/reformat/translate), "
"CSV/Excel (analyze/stats/filter/sort/convert), "
"JSON/XML (validate/format/analyze), "
"code files (explain/review/fix/optimize/run/document/test), "
"audio (transcribe/trim/convert/info), "
"video (trim/extract_audio/extract_frame/compress/transcribe/info), "
"archives (list/extract), "
"presentations (summarize/extract_text). "
"ALWAYS call this tool when a file has been uploaded and the user gives a command about it. "
"If the user's command is ambiguous, pick the most logical action for that file type."
),
"parameters": {
"type": "OBJECT",
"properties": {
"file_path": {
"type": "STRING",
"description": "Full path to the uploaded file. Leave empty to use the currently uploaded file."
},
"action": {
"type": "STRING",
"description": (
"What to do with the file. Examples by type:\n"
"image: describe | ocr | resize | compress | convert | info\n"
"pdf: summarize | extract_text | to_word | info\n"
"docx/txt: summarize | fix | reformat | translate_hint | word_count | to_bullet\n"
"csv/excel: analyze | stats | filter | sort | convert | info\n"
"json: validate | format | analyze | to_csv\n"
"code: explain | review | fix | optimize | run | document | test\n"
"audio: transcribe | trim | convert | info\n"
"video: trim | extract_audio | extract_frame | compress | transcribe | info | convert\n"
"archive: list | extract\n"
"pptx: summarize | extract_text | analyze"
)
},
"instruction": {
"type": "STRING",
"description": "Free-form instruction if action doesn't cover it. E.g. 'translate this to Turkish', 'find all email addresses'"
},
"format": {
"type": "STRING",
"description": "Target format for conversion. E.g. 'mp3', 'pdf', 'csv', 'png'"
},
"width": {"type": "INTEGER", "description": "Target width for image resize"},
"height": {"type": "INTEGER", "description": "Target height for image resize"},
"scale": {"type": "NUMBER", "description": "Scale factor for image resize (e.g. 0.5)"},
"quality": {"type": "INTEGER", "description": "Quality 1-100 for image/video compress"},
"start": {"type": "STRING", "description": "Start time for trim: seconds or HH:MM:SS"},
"end": {"type": "STRING", "description": "End time for trim: seconds or HH:MM:SS"},
"timestamp": {"type": "STRING", "description": "Timestamp for video frame extraction HH:MM:SS"},
"column": {"type": "STRING", "description": "Column name for CSV filter/sort"},
"value": {"type": "STRING", "description": "Filter value for CSV filter"},
"condition": {"type": "STRING", "description": "Filter condition: equals|contains|gt|lt"},
"ascending": {"type": "BOOLEAN", "description": "Sort order for CSV sort (default: true)"},
"save": {"type": "BOOLEAN", "description": "Save result to file (default: true)"},
"destination": {"type": "STRING", "description": "Output folder for archive extract"},
},
"required": []
}
},
{
"name": "deep_think",
"description": (
"Use ONLY when you need to reason/plan through something yourself before answering out loud — "
"e.g. 'plan my exam schedule for the next 2 weeks', 'work out the best strategy considering X and Y', "
"complex math, or multi-constraint decisions. This does NOT execute any tools, write files, or run "
"commands — it only thinks and hands you back text to speak. If the task actually needs real work "
"done (files created, commands run, multi-step execution) use start_task_worker instead, not this. "
"Do NOT use for simple facts, small talk, or anything you can already answer directly. "
"Tell the user briefly that you're thinking it through before calling this."
),
"parameters": {
"type": "OBJECT",
"properties": {
"task": {
"type": "STRING",
"description": "The full question or planning task to reason through, with all relevant context/constraints included."
}
},
"required": ["task"]
}
},
{
"name": "save_memory",
"description": (
"Save an important personal fact about the user to long-term memory. "
"Call this silently whenever the user reveals something worth remembering: "
"name, age, city, job, preferences, hobbies, relationships, projects, or future plans. "
"Do NOT call for: weather, reminders, searches, or one-time commands. "
"Do NOT announce that you are saving — just call it silently. "