summary refs log tree commit diff
path: root/modules/agent.nix
blob: 9cc063667cf405fc48bfea0f0ce7186e4321cd91 (plain)
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
{ config, hermes-agent, ... }:

{
  imports = [
    hermes-agent.nixosModules.default
  ];

  age.secrets.renard-agent = {
    file = ../secrets/renard-agent.age;
    owner = "renard";
    group = "renard";
    mode = "0440";
  };

  services.hermes-agent = {
    enable = true;

    # the agent has it's own user on the host under
    # which it runs and can change state at.
    user = "renard";
    group = "renard";
    stateDir = "/home/renard";

    extraDependencyGroups = [ "messaging" ];

    environmentFiles = [
      # required secrets:
      # * OPENROUTER_API_KEY
      # * DISCORD_BOT_TOKEN
      # * DISCORD_ALLOWED_USERS
      config.age.secrets.renard-agent.path
    ];

    settings = {
      model = {
        provider = "openrouter";
        # my current favorite budget model, with recent price
        # changes by far the cheapest for the performance,
        # with other chinese models also being close.
        # only modality is text, requires auxiliary vision model!
        # native price: $0.44/m in, $0.87/m out.
        default = "deepseek/deepseek-v4-pro";
      };

      agent = {
        reasoning_effort = "high"; # 'high' rated highest in benchmarks, tune to taste!
        max_turns = 90;
      };

      auxiliary.vision = {
        # auxiliary vision model for image comprehension.
        # this qwen model is built for this exact use-case and is extremely
        # cheap for the performance (though below gemini), however,
        # it does not do any reasoning on the visual input!
        # possible upgrade path for reasoning: qwen3-vl-30b-a3b-thinking
        # native price: $0.13/m in, $1.56/m out.
        provider = "openrouter";
        model = "qwen/qwen3-vl-32b-instruct";
        timeout = 120;
      };

      toolsets = [ "all" ];

      terminal = {
        backend = "local";
        persistent_shell = true;
        timeout = 180;
      };

      memory = {
        memory_enabled = true;
        user_profile_enabled = true;
      };

      # slightly higher limits and targets than default for better
      # long session preservation.
      compression = {
        enabled = true;
        threshold = 0.75;
        target_ratio = 0.35;
        protect_last_n = 40;
      };

      # progressive message editing, a little silly in discord but looks like
      # the classic llm chats.
      streaming = {
        enabled = true;
      };
      display = {
        streaming = true; # small regression with streaming.enabled, keep both on.
        tool_progress = "new"; # only show fresh new tool calls, not every call.
      };

      discord = {
        require_mention = false;
        thread_require_mention = false;
        auto_thread = false;
        reactions = true;
        history_backfill = true;
        allow_mentions = {
          everyone = false;
          roles = false;
          users = true;
          replied_user = true;
        };
      };

      unauthorized_dm_behavior = "ignore"; # ignore strangers

      timezone = "Europe/Berlin";
      group_sessions_per_user = false;
    };
  };
}