Coverage for src/lilbee/core/config/enums.py: 100%

60 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""StrEnum types used by :mod:`lilbee.config`.""" 

2 

3from enum import StrEnum 

4 

5 

6class ReasoningMode(StrEnum): 

7 """How a chat API surface presents a reasoning model's thinking. 

8 

9 ``separate`` reports thinking in the wire format's own reasoning channel: 

10 ``reasoning_content`` on ``/v1/chat/completions``, a ``thinking`` block on 

11 ``/v1/messages``. ``inline`` presents thinking as ordinary answer text with 

12 the ``<think>`` markers stripped, for clients that never render the 

13 reasoning channel. ``off`` asks the model not to think, which only reaches 

14 the template on a locally served model; ``/v1/messages`` also drops 

15 thinking a model produces anyway, ``/v1/chat/completions`` still reports it 

16 in ``reasoning_content``. 

17 """ 

18 

19 SEPARATE = "separate" 

20 INLINE = "inline" 

21 OFF = "off" 

22 

23 

24class FtsLanguage(StrEnum): 

25 """Snowball stemmer languages LanceDB's FTS accepts (lancedb.index.lang_mapping). 

26 

27 Listed here so config validation does not import lancedb. 

28 """ 

29 

30 ARABIC = "Arabic" 

31 DANISH = "Danish" 

32 DUTCH = "Dutch" 

33 ENGLISH = "English" 

34 FINNISH = "Finnish" 

35 FRENCH = "French" 

36 GERMAN = "German" 

37 GREEK = "Greek" 

38 HUNGARIAN = "Hungarian" 

39 ITALIAN = "Italian" 

40 NORWEGIAN = "Norwegian" 

41 PORTUGUESE = "Portuguese" 

42 ROMANIAN = "Romanian" 

43 RUSSIAN = "Russian" 

44 SPANISH = "Spanish" 

45 SWEDISH = "Swedish" 

46 TAMIL = "Tamil" 

47 TURKISH = "Turkish" 

48 

49 

50class ChatMode(StrEnum): 

51 """How chat turns route through retrieval. ``search`` uses retrieval; ``chat`` skips it.""" 

52 

53 SEARCH = "search" 

54 CHAT = "chat" 

55 

56 

57class LlmProvider(StrEnum): 

58 """Inference backend that ``create_provider`` builds. 

59 

60 ``auto`` prefix-routes: native GGUF refs to the local llama-server engine, 

61 remote-prefixed refs (``ollama/``, ``openai/``, ...) to the SDK backend. 

62 ``remote`` forces the SDK backend. 

63 """ 

64 

65 AUTO = "auto" 

66 REMOTE = "remote" 

67 

68 

69class RerankerType(StrEnum): 

70 """How the reranker GGUF is served. ``auto`` detects by architecture.""" 

71 

72 AUTO = "auto" 

73 CROSS_ENCODER = "cross_encoder" 

74 LLM = "llm" 

75 

76 

77class CrawlRenderMode(StrEnum): 

78 """How a crawl fetches pages. ``http`` uses no browser; ``browser`` runs Chromium with JS.""" 

79 

80 HTTP = "http" 

81 BROWSER = "browser" 

82 

83 

84class ClustererBackend(StrEnum): 

85 """Known wiki clusterer backends.""" 

86 

87 EMBEDDING = "embedding" 

88 CONCEPTS = "concepts" 

89 

90 

91class WikiEntityMode(StrEnum): 

92 """Strategy used to extract entities for the wiki. 

93 

94 The extractor emits typed NER entities only. Concept pages are 

95 proposed by the LLM inside the per-source batched call in 

96 :mod:`lilbee.wiki.generation`. The enum values reflect that 

97 extractor responsibility. 

98 """ 

99 

100 NER_ENTITIES = "ner_entities" 

101 NER_CONCEPTS_PLUS_LLM_TYPES = "ner_concepts_plus_llm_types" 

102 LLM_TAGGED = "llm_tagged" 

103 

104 

105class TableModel(StrEnum): 

106 """xberg's table structure recognition model, used when layout detection is on. 

107 

108 The ``slanet_*`` variants are the docling-parity lineage; ``tatr`` is xberg's 

109 older default. ``disabled`` skips structure recognition. 

110 """ 

111 

112 DISABLED = "disabled" 

113 TATR = "tatr" 

114 SLANET_AUTO = "slanet_auto" 

115 SLANET_PLUS = "slanet_plus" 

116 SLANET_WIRED = "slanet_wired" 

117 SLANET_WIRELESS = "slanet_wireless" 

118 

119 

120class OcrPageStrategy(StrEnum): 

121 """Which PDF pages xberg OCRs: failed native text only, or also pages graded as scans.""" 

122 

123 AUTO = "auto" 

124 SCANNED_PAGES = "scanned_pages" 

125 

126 

127class KvCacheType(StrEnum): 

128 """KV cache element type. ``q8_0`` / ``q4_0`` require flash attention.""" 

129 

130 F16 = "f16" 

131 F32 = "f32" 

132 Q8_0 = "q8_0" 

133 Q4_0 = "q4_0" 

134 

135 

136# Bytes per KV element for memory budgeting, from llama.cpp's block layouts: 

137# q8_0 stores 32 elements in 34 bytes (2-byte scale + 32 data bytes) and q4_0 

138# stores 32 elements in 18 bytes (2-byte scale + 16 bytes of packed nibbles). 

139# Rounding q4_0 up to a whole byte charged its cache at almost twice the real 

140# size and nearly halved the context window the dynamic picker granted. 

141KV_CACHE_TYPE_BYTES: dict[KvCacheType, float] = { 

142 KvCacheType.F16: 2.0, 

143 KvCacheType.F32: 4.0, 

144 KvCacheType.Q8_0: 34 / 32, 

145 KvCacheType.Q4_0: 18 / 32, 

146}