Coverage for src/lilbee/core/config/defaults.py: 100%

40 statements  

« prev     ^ index     » next       coverage.py v7.15.2, created at 2026-09-28 17:20 +0000

1"""Default values and constants for :mod:`lilbee.core.config`. 

2 

3Holds frozen literal data: the config file name, directory ignore lists, the 

4NER label allow-list, LanceDB table names, the default HTTP timeout and context 

5size, the crawl URL exclusion patterns (grouped per category), the default RAG 

6and general system prompts, and the CORS allow-origin regex. 

7""" 

8 

9from __future__ import annotations 

10 

11from collections.abc import Mapping 

12from types import MappingProxyType 

13 

14DEFAULT_IGNORE_DIRS = frozenset( 

15 { 

16 "node_modules", 

17 "__pycache__", 

18 "venv", 

19 "build", 

20 "dist", 

21 "target", 

22 "vendor", 

23 "_build", 

24 "coverage", 

25 "htmlcov", 

26 } 

27) 

28 

29# spaCy NER labels that map onto something wiki-shaped. Excludes 

30# QUANTITY / ORDINAL / CARDINAL / DATE / TIME / MONEY / PERCENT / 

31# LANGUAGE / LAW because pages for "42" or "2021" are never useful, and 

32# NORP (nationalities / political / religious groups) because its surfaces 

33# are adjectival (Saturnian, American) and make poor page subjects; opt 

34# back in via the config override. FAC (buildings / airports) stays: 

35# corpora routinely surface them as wiki-worthy topics. 

36DEFAULT_ALLOWED_NER_LABELS = frozenset( 

37 {"PERSON", "ORG", "GPE", "LOC", "EVENT", "WORK_OF_ART", "PRODUCT", "FAC"} 

38) 

39 

40# Timeout for backend catalog / management HTTP calls. 

41DEFAULT_HTTP_TIMEOUT = 30.0 

42 

43# Safe default + cap for chat-mode n_ctx; full 128K+ training contexts OOM laptops. 

44DEFAULT_NUM_CTX = 8192 

45CONFIG_FILE_NAME = "config.toml" 

46 

47CHUNKS_TABLE = "chunks" 

48SOURCES_TABLE = "_sources" 

49CITATIONS_TABLE = "_citations" 

50MEMORIES_TABLE = "_memories" 

51META_TABLE = "_meta" 

52PAGE_TEXTS_TABLE = "_page_texts" 

53CONCEPT_NODES_TABLE = "concept_nodes" 

54CONCEPT_EDGES_TABLE = "concept_edges" 

55CHUNK_CONCEPTS_TABLE = "chunk_concepts" 

56ENTITIES_TABLE = "entities" 

57ENTITY_SCHEMA_TABLE = "_entity_schema" 

58# Per-(subject, source) wiki mention evidence. The wiki stub index is a 

59# corpus-wide aggregate over this table, so a subject named below the floor in 

60# each separately-synced source still crosses it once its rows are all present. 

61WIKI_MENTIONS_TABLE = "_wiki_mentions" 

62 

63# Tables an ingest writes per source, and the column holding the source key. 

64INGEST_SOURCE_COLUMNS: Mapping[str, str] = MappingProxyType( 

65 { 

66 CHUNKS_TABLE: "source", 

67 PAGE_TEXTS_TABLE: "source", 

68 CHUNK_CONCEPTS_TABLE: "chunk_source", 

69 ENTITIES_TABLE: "source", 

70 CITATIONS_TABLE: "source_filename", 

71 SOURCES_TABLE: "filename", 

72 } 

73) 

74 

75# Default URL-exclusion regexes for recursive crawls. Grouped by source 

76# CMS / category. User overrides come from LILBEE_CRAWL_EXCLUDE_PATTERNS 

77# (newline-separated) or config.toml. 

78 

79# WordPress scaffolding: admin UIs, APIs, RPC, numeric permalinks, Elementor. 

80_WP_EXCLUDE: tuple[str, ...] = ( 

81 r"/wp-admin/", 

82 r"/wp-login(\.php)?", 

83 r"/wp-json/", 

84 r"/xmlrpc\.php", 

85 r"/wp-cron\.php", 

86 r"/wp-includes/", 

87 r"/wp-content/uploads/", 

88 r"\?p=\d+", 

89 r"\?page_id=\d+", 

90 r"\?cat=\d+", 

91 r"/elementor-\d+", 

92 r"\?elementor_library", 

93) 

94 

95# Pagination and archive permalinks (WP + other CMSes share this shape). 

96_ARCHIVE_EXCLUDE: tuple[str, ...] = ( 

97 r"/page/\d+/?$", 

98 r"\?paged?=\d+", 

99 r"/20\d{2}(/\d{2}(/\d{2})?)?/?$", 

100 r"/tag/", 

101 r"/category/", 

102 r"/author/", 

103 r"/archives?/?$", 

104 r"/comment-page-\d+", 

105) 

106 

107# Syndication feeds (content-duplicated in HTML pages). 

108_FEED_EXCLUDE: tuple[str, ...] = ( 

109 r"/feed/?$", 

110 r"/feed/atom/?$", 

111 r"/feed/rdf/?$", 

112 r"/comments/feed/?$", 

113 r"/rss/?$", 

114) 

115 

116# Duplicate views of the same canonical page (AMP, print, preview). 

117_DUPLICATE_VIEW_EXCLUDE: tuple[str, ...] = ( 

118 r"/amp/?$", 

119 r"\?amp=", 

120 r"\?print=", 

121 r"/print/?$", 

122 r"\?preview=", 

123) 

124 

125# WP attachment URLs (point at media, not content pages). 

126_ATTACHMENT_EXCLUDE: tuple[str, ...] = ( 

127 r"/attachment/", 

128 r"\?attachment_id=", 

129) 

130 

131# Regexes against the whole URL, not globs, so a bare prefix also matches 

132# longer words: /cart excluded /cartography. Require a segment boundary. 

133_PATH_BOUNDARY = r"(?:/|\?|#|$)" 

134 

135 

136def _whole_segments(*paths: str) -> tuple[str, ...]: 

137 """Anchor each path prefix so it matches a whole segment, not a word.""" 

138 return tuple(path + _PATH_BOUNDARY for path in paths) 

139 

140 

141# Auth and account flows (generic across CMSes and e-commerce platforms). 

142_AUTH_EXCLUDE: tuple[str, ...] = _whole_segments( 

143 r"/login", 

144 r"/logout", 

145 r"/register", 

146 r"/signup", 

147 r"/signin", 

148 r"/account", 

149 r"/profile", 

150 r"/password-reset", 

151 r"/forgot-password", 

152) 

153_AUTH_EXCLUDE = (*_AUTH_EXCLUDE, r"/my-account/") 

154 

155# E-commerce transactional flows (cart / checkout / compare / etc.). 

156_ECOMMERCE_EXCLUDE: tuple[str, ...] = _whole_segments( 

157 r"/cart", 

158 r"/checkout", 

159 r"/wishlist", 

160 r"/orders?", 

161 r"/compare", 

162) 

163_ECOMMERCE_EXCLUDE = ( 

164 *_ECOMMERCE_EXCLUDE, 

165 r"/products\.json", 

166 r"/collections/.+/products/.+\?page=", 

167) 

168 

169# Marketing / tracking query parameters (utm_*, fbclid, gclid, etc.). 

170# Vendor campaign tokens only. Dropping ?utm_source= is free (the canonical 

171# URL is in the frontier too), but ?ref= and ?share= are ordinary content 

172# links on docs and forum platforms. 

173_TRACKING_EXCLUDE: tuple[str, ...] = ( 

174 ( 

175 r"[?&](" 

176 r"utm_[a-z_]+" 

177 r"|fbclid|gclid|msclkid|yclid" 

178 r"|mc_cid|mc_eid" 

179 r"|_hsenc|_hsmi|hsCtaTracking" 

180 r"|mkt_tok|mkt_[a-z_]+" 

181 r"|trk|trkInfo" 

182 r"|dm_i" 

183 r"|vero_id|vero_conv" 

184 r"|oly_anon_id|oly_enc_id" 

185 r"|igshid" 

186 r"|pk_campaign|pk_source|pk_medium|pk_[a-z_]+" 

187 r"|_ga" 

188 r"|affiliate|aff_id|aff_ref|aff|partner" 

189 r"|srsltid" 

190 r"|replytocom" 

191 r")=" 

192 ), 

193) 

194 

195# Site-meta URLs and non-HTML resources; skipped before fetch. 

196_META_EXCLUDE: tuple[str, ...] = ( 

197 r"/sitemap[^/]*\.xml", 

198 r"/robots\.txt", 

199 r"/humans\.txt", 

200 r"/favicon\.ico", 

201 r"/\.well-known/", 

202 r"\.(jpe?g|png|gif|webp|avif|svg|ico|pdf|docx?|xlsx?|pptx?|zip|tar|gz|mp3|mp4|webm|ogg|ttf|woff2?|css|js|map|json|xml)(\?.*)?$", 

203) 

204 

205# Mediawiki/Wikipedia navlinks that dominate BFS before the article body. 

206_MEDIAWIKI_EXCLUDE: tuple[str, ...] = ( 

207 r"/wiki/Main_Page$", 

208 r"/wiki/Wikipedia:", 

209 r"/wiki/Portal:", 

210 r"/wiki/Help:", 

211 r"/wiki/Special:", 

212 r"/wiki/Category:", 

213 r"/wiki/Template:", 

214 r"/wiki/Template_talk:", 

215 r"/wiki/Talk:", 

216 r"/wiki/File:", 

217 r"/wiki/File_talk:", 

218 r"/wiki/User:", 

219 r"/wiki/User_talk:", 

220 r"/w/index\.php", 

221) 

222 

223DEFAULT_CRAWL_EXCLUDE_PATTERNS: tuple[str, ...] = ( 

224 *_WP_EXCLUDE, 

225 *_ARCHIVE_EXCLUDE, 

226 *_FEED_EXCLUDE, 

227 *_DUPLICATE_VIEW_EXCLUDE, 

228 *_ATTACHMENT_EXCLUDE, 

229 *_AUTH_EXCLUDE, 

230 *_ECOMMERCE_EXCLUDE, 

231 *_TRACKING_EXCLUDE, 

232 *_META_EXCLUDE, 

233 *_MEDIAWIKI_EXCLUDE, 

234) 

235 

236 

237DEFAULT_RAG_SYSTEM_PROMPT = ( 

238 "You are a precise assistant answering from the user's own documents. " 

239 "Ground every claim in the numbered context passages and nothing else; if " 

240 "they don't cover the question, say so plainly instead of guessing or " 

241 "answering from general knowledge. Synthesize across passages rather than " 

242 "leaning on one, and if they disagree, note the conflict. Cite inline by " 

243 "placing the passage number in brackets right after the claim it supports " 

244 "(e.g. [1] or [2][5]), and cite only passages you actually used. Do not " 

245 "write a Sources, References, or Bibliography list at the end; the app adds " 

246 "the real source list for you. Prefer exact values, names, and short quotes " 

247 "from the context over paraphrase. Handle any material: prose, notes, " 

248 "tables, transcripts, or code; for code, prefer a working example. When " 

249 "asked how to do something, lay the answer out as ordered steps; if the " 

250 "context covers the procedure only partially, give the steps it contains " 

251 "and name what's missing rather than glossing over the gap. Match the " 

252 "answer's length to the question: exhaustive requests deserve every " 

253 "relevant detail the context offers." 

254) 

255 

256DEFAULT_GENERAL_SYSTEM_PROMPT = ( 

257 "You are a helpful, direct assistant. Answer the user's question from " 

258 "general knowledge. Keep responses concise unless asked to elaborate. " 

259 "For code, prefer working examples over abstract explanations." 

260) 

261 

262# CORS allow-origin regex: Obsidian (desktop + iOS) and localhost loopback. 

263# Mutating endpoints still require auth regardless of origin. 

264DEFAULT_CORS_ORIGIN_REGEX = ( 

265 r"^(app://obsidian\.md" 

266 r"|capacitor://localhost" 

267 r"|https?://localhost(:\d+)?" 

268 r"|https?://127\.0\.0\.1(:\d+)?" 

269 r"|https?://\[::1\](:\d+)?)$" 

270)