Coverage for src/lilbee/core/config/defaults.py: 100%
40 statements
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
« prev ^ index » next coverage.py v7.15.2, created at 2026-09-28 17:20 +0000
1"""Default values and constants for :mod:`lilbee.core.config`.
3Holds frozen literal data: the config file name, directory ignore lists, the
4NER label allow-list, LanceDB table names, the default HTTP timeout and context
5size, the crawl URL exclusion patterns (grouped per category), the default RAG
6and general system prompts, and the CORS allow-origin regex.
7"""
9from __future__ import annotations
11from collections.abc import Mapping
12from types import MappingProxyType
14DEFAULT_IGNORE_DIRS = frozenset(
15 {
16 "node_modules",
17 "__pycache__",
18 "venv",
19 "build",
20 "dist",
21 "target",
22 "vendor",
23 "_build",
24 "coverage",
25 "htmlcov",
26 }
27)
29# spaCy NER labels that map onto something wiki-shaped. Excludes
30# QUANTITY / ORDINAL / CARDINAL / DATE / TIME / MONEY / PERCENT /
31# LANGUAGE / LAW because pages for "42" or "2021" are never useful, and
32# NORP (nationalities / political / religious groups) because its surfaces
33# are adjectival (Saturnian, American) and make poor page subjects; opt
34# back in via the config override. FAC (buildings / airports) stays:
35# corpora routinely surface them as wiki-worthy topics.
36DEFAULT_ALLOWED_NER_LABELS = frozenset(
37 {"PERSON", "ORG", "GPE", "LOC", "EVENT", "WORK_OF_ART", "PRODUCT", "FAC"}
38)
40# Timeout for backend catalog / management HTTP calls.
41DEFAULT_HTTP_TIMEOUT = 30.0
43# Safe default + cap for chat-mode n_ctx; full 128K+ training contexts OOM laptops.
44DEFAULT_NUM_CTX = 8192
45CONFIG_FILE_NAME = "config.toml"
47CHUNKS_TABLE = "chunks"
48SOURCES_TABLE = "_sources"
49CITATIONS_TABLE = "_citations"
50MEMORIES_TABLE = "_memories"
51META_TABLE = "_meta"
52PAGE_TEXTS_TABLE = "_page_texts"
53CONCEPT_NODES_TABLE = "concept_nodes"
54CONCEPT_EDGES_TABLE = "concept_edges"
55CHUNK_CONCEPTS_TABLE = "chunk_concepts"
56ENTITIES_TABLE = "entities"
57ENTITY_SCHEMA_TABLE = "_entity_schema"
58# Per-(subject, source) wiki mention evidence. The wiki stub index is a
59# corpus-wide aggregate over this table, so a subject named below the floor in
60# each separately-synced source still crosses it once its rows are all present.
61WIKI_MENTIONS_TABLE = "_wiki_mentions"
63# Tables an ingest writes per source, and the column holding the source key.
64INGEST_SOURCE_COLUMNS: Mapping[str, str] = MappingProxyType(
65 {
66 CHUNKS_TABLE: "source",
67 PAGE_TEXTS_TABLE: "source",
68 CHUNK_CONCEPTS_TABLE: "chunk_source",
69 ENTITIES_TABLE: "source",
70 CITATIONS_TABLE: "source_filename",
71 SOURCES_TABLE: "filename",
72 }
73)
75# Default URL-exclusion regexes for recursive crawls. Grouped by source
76# CMS / category. User overrides come from LILBEE_CRAWL_EXCLUDE_PATTERNS
77# (newline-separated) or config.toml.
79# WordPress scaffolding: admin UIs, APIs, RPC, numeric permalinks, Elementor.
80_WP_EXCLUDE: tuple[str, ...] = (
81 r"/wp-admin/",
82 r"/wp-login(\.php)?",
83 r"/wp-json/",
84 r"/xmlrpc\.php",
85 r"/wp-cron\.php",
86 r"/wp-includes/",
87 r"/wp-content/uploads/",
88 r"\?p=\d+",
89 r"\?page_id=\d+",
90 r"\?cat=\d+",
91 r"/elementor-\d+",
92 r"\?elementor_library",
93)
95# Pagination and archive permalinks (WP + other CMSes share this shape).
96_ARCHIVE_EXCLUDE: tuple[str, ...] = (
97 r"/page/\d+/?$",
98 r"\?paged?=\d+",
99 r"/20\d{2}(/\d{2}(/\d{2})?)?/?$",
100 r"/tag/",
101 r"/category/",
102 r"/author/",
103 r"/archives?/?$",
104 r"/comment-page-\d+",
105)
107# Syndication feeds (content-duplicated in HTML pages).
108_FEED_EXCLUDE: tuple[str, ...] = (
109 r"/feed/?$",
110 r"/feed/atom/?$",
111 r"/feed/rdf/?$",
112 r"/comments/feed/?$",
113 r"/rss/?$",
114)
116# Duplicate views of the same canonical page (AMP, print, preview).
117_DUPLICATE_VIEW_EXCLUDE: tuple[str, ...] = (
118 r"/amp/?$",
119 r"\?amp=",
120 r"\?print=",
121 r"/print/?$",
122 r"\?preview=",
123)
125# WP attachment URLs (point at media, not content pages).
126_ATTACHMENT_EXCLUDE: tuple[str, ...] = (
127 r"/attachment/",
128 r"\?attachment_id=",
129)
131# Regexes against the whole URL, not globs, so a bare prefix also matches
132# longer words: /cart excluded /cartography. Require a segment boundary.
133_PATH_BOUNDARY = r"(?:/|\?|#|$)"
136def _whole_segments(*paths: str) -> tuple[str, ...]:
137 """Anchor each path prefix so it matches a whole segment, not a word."""
138 return tuple(path + _PATH_BOUNDARY for path in paths)
141# Auth and account flows (generic across CMSes and e-commerce platforms).
142_AUTH_EXCLUDE: tuple[str, ...] = _whole_segments(
143 r"/login",
144 r"/logout",
145 r"/register",
146 r"/signup",
147 r"/signin",
148 r"/account",
149 r"/profile",
150 r"/password-reset",
151 r"/forgot-password",
152)
153_AUTH_EXCLUDE = (*_AUTH_EXCLUDE, r"/my-account/")
155# E-commerce transactional flows (cart / checkout / compare / etc.).
156_ECOMMERCE_EXCLUDE: tuple[str, ...] = _whole_segments(
157 r"/cart",
158 r"/checkout",
159 r"/wishlist",
160 r"/orders?",
161 r"/compare",
162)
163_ECOMMERCE_EXCLUDE = (
164 *_ECOMMERCE_EXCLUDE,
165 r"/products\.json",
166 r"/collections/.+/products/.+\?page=",
167)
169# Marketing / tracking query parameters (utm_*, fbclid, gclid, etc.).
170# Vendor campaign tokens only. Dropping ?utm_source= is free (the canonical
171# URL is in the frontier too), but ?ref= and ?share= are ordinary content
172# links on docs and forum platforms.
173_TRACKING_EXCLUDE: tuple[str, ...] = (
174 (
175 r"[?&]("
176 r"utm_[a-z_]+"
177 r"|fbclid|gclid|msclkid|yclid"
178 r"|mc_cid|mc_eid"
179 r"|_hsenc|_hsmi|hsCtaTracking"
180 r"|mkt_tok|mkt_[a-z_]+"
181 r"|trk|trkInfo"
182 r"|dm_i"
183 r"|vero_id|vero_conv"
184 r"|oly_anon_id|oly_enc_id"
185 r"|igshid"
186 r"|pk_campaign|pk_source|pk_medium|pk_[a-z_]+"
187 r"|_ga"
188 r"|affiliate|aff_id|aff_ref|aff|partner"
189 r"|srsltid"
190 r"|replytocom"
191 r")="
192 ),
193)
195# Site-meta URLs and non-HTML resources; skipped before fetch.
196_META_EXCLUDE: tuple[str, ...] = (
197 r"/sitemap[^/]*\.xml",
198 r"/robots\.txt",
199 r"/humans\.txt",
200 r"/favicon\.ico",
201 r"/\.well-known/",
202 r"\.(jpe?g|png|gif|webp|avif|svg|ico|pdf|docx?|xlsx?|pptx?|zip|tar|gz|mp3|mp4|webm|ogg|ttf|woff2?|css|js|map|json|xml)(\?.*)?$",
203)
205# Mediawiki/Wikipedia navlinks that dominate BFS before the article body.
206_MEDIAWIKI_EXCLUDE: tuple[str, ...] = (
207 r"/wiki/Main_Page$",
208 r"/wiki/Wikipedia:",
209 r"/wiki/Portal:",
210 r"/wiki/Help:",
211 r"/wiki/Special:",
212 r"/wiki/Category:",
213 r"/wiki/Template:",
214 r"/wiki/Template_talk:",
215 r"/wiki/Talk:",
216 r"/wiki/File:",
217 r"/wiki/File_talk:",
218 r"/wiki/User:",
219 r"/wiki/User_talk:",
220 r"/w/index\.php",
221)
223DEFAULT_CRAWL_EXCLUDE_PATTERNS: tuple[str, ...] = (
224 *_WP_EXCLUDE,
225 *_ARCHIVE_EXCLUDE,
226 *_FEED_EXCLUDE,
227 *_DUPLICATE_VIEW_EXCLUDE,
228 *_ATTACHMENT_EXCLUDE,
229 *_AUTH_EXCLUDE,
230 *_ECOMMERCE_EXCLUDE,
231 *_TRACKING_EXCLUDE,
232 *_META_EXCLUDE,
233 *_MEDIAWIKI_EXCLUDE,
234)
237DEFAULT_RAG_SYSTEM_PROMPT = (
238 "You are a precise assistant answering from the user's own documents. "
239 "Ground every claim in the numbered context passages and nothing else; if "
240 "they don't cover the question, say so plainly instead of guessing or "
241 "answering from general knowledge. Synthesize across passages rather than "
242 "leaning on one, and if they disagree, note the conflict. Cite inline by "
243 "placing the passage number in brackets right after the claim it supports "
244 "(e.g. [1] or [2][5]), and cite only passages you actually used. Do not "
245 "write a Sources, References, or Bibliography list at the end; the app adds "
246 "the real source list for you. Prefer exact values, names, and short quotes "
247 "from the context over paraphrase. Handle any material: prose, notes, "
248 "tables, transcripts, or code; for code, prefer a working example. When "
249 "asked how to do something, lay the answer out as ordered steps; if the "
250 "context covers the procedure only partially, give the steps it contains "
251 "and name what's missing rather than glossing over the gap. Match the "
252 "answer's length to the question: exhaustive requests deserve every "
253 "relevant detail the context offers."
254)
256DEFAULT_GENERAL_SYSTEM_PROMPT = (
257 "You are a helpful, direct assistant. Answer the user's question from "
258 "general knowledge. Keep responses concise unless asked to elaborate. "
259 "For code, prefer working examples over abstract explanations."
260)
262# CORS allow-origin regex: Obsidian (desktop + iOS) and localhost loopback.
263# Mutating endpoints still require auth regardless of origin.
264DEFAULT_CORS_ORIGIN_REGEX = (
265 r"^(app://obsidian\.md"
266 r"|capacitor://localhost"
267 r"|https?://localhost(:\d+)?"
268 r"|https?://127\.0\.0\.1(:\d+)?"
269 r"|https?://\[::1\](:\d+)?)$"
270)