Coverage for src/lilbee/providers/fleet/swap_config.py: 100%

48 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Generate one swap group's llama-swap config. 

2 

3Each group runs behind its own llama-swap process with its own config, so a 

4reload of one group never touches another's loaded servers. See 

5docs/architecture.md (llama-swap) for the supervisor/proxy design. 

6""" 

7 

8from __future__ import annotations 

9 

10import json 

11import shlex 

12import subprocess 

13import sys 

14from typing import TYPE_CHECKING 

15 

16from lilbee.providers.fleet.readback import MEMORY_FLAG, engine_log_env 

17from lilbee.providers.roles import WorkerRole 

18 

19if TYPE_CHECKING: 

20 from collections.abc import Mapping 

21 from pathlib import Path 

22 

23 from lilbee.providers.fleet.launch import InstanceLaunch 

24 

25# One group holds this process's members. Swap disabled keeps them co-resident (a 

26# role's replicas); enabled makes llama-swap evict one to load another. 

27_GROUP_NAME = "lilbee" 

28# Cold-load ceiling floor per role; the member's weights scale it up from here. 

29# An embedder's weights stream in seconds, so its floor covers only process start 

30# and backend init: waiting out a chat-sized ten minutes on a crash-looping or 

31# wedged embed server holds every ingest caller for that long. 

32_COLD_LOAD_FLOOR_S: dict[WorkerRole, int] = { 

33 WorkerRole.CHAT: 600, 

34 WorkerRole.VISION: 600, 

35 WorkerRole.RERANK: 600, 

36 WorkerRole.EMBED: 120, 

37} 

38# Conservative cold-load disk rate; a slow network volume streams well under this. 

39_COLD_LOAD_BYTES_PER_S = 150 * 1024 * 1024 

40_LOG_LEVEL = "info" 

41# Matches the --host every server argv binds (adapters); "localhost" would have 

42# llama-swap dial [::1] first, where another process could hold the same port. 

43_PROXY_URL_TEMPLATE = "http://127.0.0.1:{port}" 

44# Shared with the swap manager, whose orphan-server sweep matches this flag's 

45# value in survivor cmdlines. 

46PORT_FLAG = "--port" 

47 

48# llama-swap config keys. 

49_KEY_HEALTH_TIMEOUT = "healthCheckTimeout" 

50_KEY_LOG_LEVEL = "logLevel" 

51_KEY_MODELS = "models" 

52_KEY_CMD = "cmd" 

53_KEY_PROXY = "proxy" 

54_KEY_TTL = "ttl" 

55_KEY_ENV = "env" 

56_KEY_GROUPS = "groups" 

57_KEY_SWAP = "swap" 

58_KEY_EXCLUSIVE = "exclusive" 

59_KEY_PERSISTENT = "persistent" 

60_KEY_MEMBERS = "members" 

61 

62 

63def build_swap_config( 

64 launches: list[InstanceLaunch], 

65 member_ports: Mapping[str, int], 

66 *, 

67 swap: bool = False, 

68 ttl_seconds: int = 0, 

69 engine_log_dir: Path | None = None, 

70) -> str: 

71 """Render a llama-swap config (JSON, which is valid YAML) for *launches*. 

72 

73 Each launch becomes a model whose id is its replica model id and whose 

74 command is the llama-server argv plus the explicit port from *member_ports*; 

75 one group holds them behind this group's proxy endpoint. ``swap`` makes the 

76 members evict each other on load, so only one is resident at a time; the 

77 default keeps them co-resident. ``ttl_seconds`` is llama-swap's idle unload 

78 timer per member; 0 keeps weights loaded forever. Ports are allocated fresh per start (never 

79 llama-swap's fixed ``startPort`` range) so a previous instance's lingering 

80 server can't collide with the new fleet's bind. 

81 

82 ``engine_log_dir`` points each engine at its own log there, at the verbosity 

83 that reports what it allocated. llama-swap forwards none of the upstream's 

84 output, so for an engine without ``GET /memory`` that file is the only place 

85 those numbers exist, and reading them back is what tells the planner whether 

86 its estimate held (:mod:`lilbee.providers.fleet.readback`). A launch carrying 

87 ``--memory`` serves the same numbers over HTTP instead and gets no log env: 

88 the trace-level file is exactly what the endpoint retires. 

89 """ 

90 models: dict[str, object] = {} 

91 for launch in launches: 

92 port = member_ports[launch.model_id] 

93 entry: dict[str, object] = { 

94 _KEY_CMD: _command_line(launch.argv, port), 

95 _KEY_PROXY: _PROXY_URL_TEMPLATE.format(port=port), 

96 _KEY_TTL: ttl_seconds, 

97 } 

98 env = dict(launch.env_overrides) 

99 if engine_log_dir is not None and MEMORY_FLAG not in launch.argv: 

100 env.update(engine_log_env(engine_log_dir, launch.model_id)) 

101 if env: 

102 entry[_KEY_ENV] = [f"{key}={value}" for key, value in env.items()] 

103 models[launch.model_id] = entry 

104 config: dict[str, object] = { 

105 _KEY_HEALTH_TIMEOUT: _health_check_timeout_s(launches), 

106 _KEY_LOG_LEVEL: _LOG_LEVEL, 

107 _KEY_MODELS: models, 

108 _KEY_GROUPS: { 

109 _GROUP_NAME: { 

110 _KEY_SWAP: swap, 

111 _KEY_EXCLUSIVE: False, 

112 _KEY_PERSISTENT: True, 

113 _KEY_MEMBERS: [launch.model_id for launch in launches], 

114 } 

115 }, 

116 } 

117 return json.dumps(config, indent=2) 

118 

119 

120def cold_load_timeout_s(weights_bytes: int, role: WorkerRole) -> int: 

121 """Cold-load ceiling for one member's weights at a conservative disk rate, floored per role. 

122 

123 The single source of the scaling formula: llama-swap's health-check timeout 

124 and the provider's per-client request timeout both derive from it, so a model 

125 whose load llama-swap would wait out can never time out the client first. 

126 """ 

127 return max(_COLD_LOAD_FLOOR_S[role], weights_bytes // _COLD_LOAD_BYTES_PER_S) 

128 

129 

130def _health_check_timeout_s(launches: list[InstanceLaunch]) -> int: 

131 """Slowest member's cold-load ceiling; the timeout is proxy-global in 

132 llama-swap, so the slowest possible load sets it.""" 

133 return max( 

134 (cold_load_timeout_s(launch.weights_bytes, launch.role) for launch in launches), 

135 default=max(_COLD_LOAD_FLOOR_S.values()), 

136 ) 

137 

138 

139def _command_line(argv: list[str], port: int) -> str: 

140 """Shell command for a member: the role argv plus its explicit port. 

141 

142 Quoting must match how llama-swap splits the command back into argv: MS 

143 rules on Windows (POSIX single quotes would stay literal in the paths and 

144 the spawn fails with "file does not exist"), POSIX everywhere else. 

145 """ 

146 if sys.platform == "win32": 

147 rendered = subprocess.list2cmdline(argv) 

148 else: 

149 rendered = shlex.join(argv) 

150 return f"{rendered} {PORT_FLAG} {port}"