feat: knowledge-base memory, Dockerfile, docker-compose, CI/release workflows, PR template

This commit is contained in:
copilot-swe-agent[bot]
2026-07-25 12:29:21 +00:00
committed by GitHub
parent c74c8b0d8b
commit 867542dd49
11 changed files with 543 additions and 22 deletions
+82 -7
View File
@@ -39,6 +39,20 @@ _FLUSH_SYSTEM_PROMPT = (
"Be precise. Omit pleasantries."
)
_TAGS_SYSTEM_PROMPT = (
"You are a keyword tagger for a knowledge base. "
"Extract 5–8 short, lowercase keyword tags from the following conversation summary. "
"Tags should represent the main topics, entities, and concepts discussed. "
"Return ONLY a comma-separated list of tags with no other text or punctuation. "
"Example output: api design, authentication, database schema, user roles, caching"
)
_KB_CONTEXT_HEADER = (
"The following are relevant past conversation summaries from your knowledge base. "
"Use them as background context if they relate to the current question, "
"but do not repeat their contents unless directly asked."
)
def _thread_key(update: Update) -> tuple[int, int] | None:
"""Return the (chat_id, thread_id) key if the message is part of a thread, else None."""
@@ -83,7 +97,7 @@ async def start_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) -> N
"/help – show available commands\n"
"/clear – reset conversation history\n"
"/flush – summarise and archive this thread's memory\n"
"/recall – retrieve archived thread summaries\n"
"/recall [query] – retrieve archived thread summaries\n"
"/analyse – run a manual API analysis right now",
parse_mode=ParseMode.MARKDOWN,
)
@@ -103,7 +117,7 @@ async def help_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) -> No
"/clear – reset conversation history for this context\n"
"/flush – summarise the current thread, store the summary, and compress memory\n"
" _(only available inside a message thread)_\n"
"/recall – show the stored summary for this thread, or list all summaries\n"
"/recall [query] – show this thread's summary, list all summaries, or search by keyword\n"
"/analyse – trigger an immediate API analysis and proposal",
parse_mode=ParseMode.MARKDOWN,
)
@@ -183,12 +197,20 @@ async def flush_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) -> N
system_prompt=_FLUSH_SYSTEM_PROMPT,
)
# Extract keyword tags for knowledge-base indexing (second LLM call, lightweight)
tags_raw = await llm.chat(
f"Summary to tag:\n\n{summary_text}",
system_prompt=_TAGS_SYSTEM_PROMPT,
)
tags = [t.strip().lower() for t in tags_raw.split(",") if t.strip()][:10]
message_count = sum(1 for m in history if m["role"] == "user")
thread_summary = ThreadSummary(
chat_id=chat_id,
thread_id=thread_id,
summary=summary_text,
message_count=message_count,
tags=tags,
)
store.save(thread_summary)
@@ -206,10 +228,14 @@ async def flush_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) -> N
async def recall_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) -> None:
"""Handle /recall – retrieve stored thread summaries.
"""Handle /recall [query] – retrieve stored thread summaries.
Inside a thread: shows the stored summary for this thread (if any).
Outside a thread: lists all stored summaries (newest first).
With a query argument (e.g. ``/recall api design``): searches all stored
summaries whose tags overlap with the query keywords and returns matches.
Without a query:
- Inside a thread: shows the stored summary for this thread (if any).
- Outside a thread: lists all stored summaries (newest first).
"""
settings: Settings = context.bot_data["settings"]
store: ThreadMemoryStore = context.bot_data["thread_store"]
@@ -218,6 +244,28 @@ async def recall_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) ->
if user is None or not _is_allowed(user.id, settings):
return
# If the user supplied a keyword query, search the knowledge base
args: list[str] = context.args or [] # type: ignore[assignment]
if args:
query = " ".join(args).strip()
results = store.search(query)
if not results:
await update.message.reply_text( # type: ignore[union-attr]
f"No memories found matching *{query}*. "
"Try a different keyword or use /flush to add more summaries.",
parse_mode=ParseMode.MARKDOWN,
)
return
lines = [f"\U0001f50d *Knowledge base search: {query}*\n"]
for s in results:
date = s.flushed_at[:10]
first_line = s.summary.split("\n")[0][:80]
lines.append(f"\u2022 Thread `{s.thread_id}` ({date}): {first_line}\u2026")
if s.tags:
lines.append(f" \U0001f3f7 {', '.join(s.tags)}")
await _send_long(update, "\n".join(lines))
return
key = _thread_key(update)
if key is not None:
chat_id, thread_id = key
@@ -243,6 +291,8 @@ async def recall_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) ->
date = s.flushed_at[:10]
first_line = s.summary.split("\n")[0][:80]
lines.append(f"\u2022 Thread `{s.thread_id}` ({date}): {first_line}\u2026")
if s.tags:
lines.append(f" \U0001f3f7 {', '.join(s.tags)}")
await _send_long(update, "\n".join(lines))
@@ -267,14 +317,37 @@ async def analyse_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) ->
await _send_long(update, proposal.format_for_telegram())
def _with_kb_context(
history: list[dict[str, str]],
relevant: list[ThreadSummary],
) -> list[dict[str, str]]:
"""Prepend relevant knowledge-base summaries as a transient system context message.
The returned list is a *new* list — the original *history* is not mutated.
The injected message is never appended to the stored history, so it does not
permanently consume the context window.
"""
if not relevant:
return history
snippets = [f"[Thread {s.thread_id}] {s.summary[:400]}" for s in relevant[:3]]
kb_msg = _KB_CONTEXT_HEADER + "\n\n" + "\n\n---\n\n".join(snippets)
return [{"role": "system", "content": kb_msg}, *history]
async def message_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) -> None:
"""Handle plain text messages – forward to LLM and reply.
Thread messages: history is stored unbounded under the (chat_id, thread_id) key.
Non-thread messages: history is capped at _MAX_HISTORY turns per user.
Before each LLM call the knowledge base is searched for summaries whose tags
overlap with keywords in the current message. Any matches are injected as
transient context — they are NOT stored in the rolling history, so they do
not permanently consume the context window.
"""
settings: Settings = context.bot_data["settings"]
llm: LLMClient = context.bot_data["llm"]
store: ThreadMemoryStore = context.bot_data["thread_store"]
user = update.effective_user
if user is None or not _is_allowed(user.id, settings):
return
@@ -287,13 +360,15 @@ async def message_handler(update: Update, context: ContextTypes.DEFAULT_TYPE) ->
if key is not None:
# Thread message: unbounded history
history = _thread_history[key]
reply = await llm.chat(text, history=history)
call_history = _with_kb_context(history, store.search(text))
reply = await llm.chat(text, history=call_history)
history.append({"role": "user", "content": text})
history.append({"role": "assistant", "content": reply})
else:
# Non-thread message: capped history per user
history = _history[user.id]
reply = await llm.chat(text, history=history)
call_history = _with_kb_context(history, store.search(text))
reply = await llm.chat(text, history=call_history)
history.append({"role": "user", "content": text})
history.append({"role": "assistant", "content": reply})
if len(history) > _MAX_HISTORY * 2: