Compare commits
61
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
15359e2ae3 | ||
|
|
09d15d83de
|
||
|
|
19b1505815 | ||
|
|
3cd0654306
|
||
|
|
a4fb413151
|
||
|
|
7b160be3b7
|
||
|
|
733241554f
|
||
|
|
92e9f3dcc2
|
||
|
|
2fbe47e936 | ||
|
|
f2b95935bb | ||
|
|
bce439921f | ||
|
|
2de2d0fe3a | ||
|
|
cf795bbc35 | ||
|
|
a6ed20451a | ||
|
|
7fd32b3024 | ||
|
|
a88d233c6b | ||
|
|
2abc39e3ac | ||
|
|
f430998137 | ||
|
|
8dceb79d91 | ||
|
|
6c5b0f778d | ||
|
|
37ed8fd0f9 | ||
|
|
0594ea54aa | ||
|
|
60f7473297 | ||
|
|
ec69e8e4f7 | ||
|
|
62b1175aeb | ||
|
|
41f804a1eb | ||
|
|
f50d076164 | ||
|
|
fc4f9c5053 | ||
|
|
e3262cd366 | ||
|
|
341f3d8623 | ||
|
|
e2c29204fa | ||
|
|
f0e6a0cb52 | ||
|
|
7f0b0376d1 | ||
|
|
44b5ea6a68 | ||
|
|
a49457094d | ||
|
|
9296fda390 | ||
|
|
bb0d9090f3 | ||
|
|
703a2384e7 | ||
|
|
4b3f00c325 | ||
|
|
38dfe404d1 | ||
|
|
347ac63f86 | ||
|
|
506758f67d | ||
|
|
f0572ba9fb | ||
|
|
4686f3fae0 | ||
|
|
ea1c8cfb13 | ||
|
|
9ca7578d28 | ||
|
|
64b466c4ac | ||
|
|
49174de9ff | ||
|
|
59f9f01c69 | ||
|
|
a7eae4b09f | ||
|
|
c466b04a25 | ||
|
|
431e5c63aa | ||
|
|
6e117e3ce9 | ||
|
|
9a9228bc07 | ||
|
|
2dd371408f | ||
|
|
0005ad1fd3 | ||
|
|
446978704d | ||
|
|
f24bd5b361 | ||
|
|
4d5c27cfaa | ||
|
|
d45f0be314 | ||
|
|
e1a24aff20 |
@@ -0,0 +1,57 @@
|
|||||||
|
name: Create Blog Article if new notes exist
|
||||||
|
on:
|
||||||
|
schedule:
|
||||||
|
- cron: "15 18 * * *"
|
||||||
|
push:
|
||||||
|
branches:
|
||||||
|
- master
|
||||||
|
jobs:
|
||||||
|
prepare_blog_drafts_and_push:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
steps:
|
||||||
|
- name: Checkout repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Install dependencies
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
apt update && apt upgrade -y
|
||||||
|
apt install rustc cargo python-is-python3 pip python3-venv python3-virtualenv libmagic-dev git -y
|
||||||
|
virtualenv .venv
|
||||||
|
source .venv/bin/activate
|
||||||
|
pip install --upgrade pip
|
||||||
|
pip install -r requirements.txt
|
||||||
|
git config --global user.name "Blog Creator"
|
||||||
|
git config --global user.email "ridgway.infrastructure@gmail.com"
|
||||||
|
git config --global push.autoSetupRemote true
|
||||||
|
|
||||||
|
- name: Create .env
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
echo "TRILIUM_HOST=${{ vars.TRILIUM_HOST }}" > .env
|
||||||
|
echo "TRILIUM_PORT='${{ vars.TRILIUM_PORT }}'" >> .env
|
||||||
|
echo "TRILIUM_PROTOCOL='${{ vars.TRILIUM_PROTOCOL }}'" >> .env
|
||||||
|
echo "TRILIUM_PASS='${{ secrets.TRILIUM_PASS }}'" >> .env
|
||||||
|
echo "TRILIUM_TOKEN='${{ secrets.TRILIUM_TOKEN }}'" >> .env
|
||||||
|
echo "OLLAMA_PROTOCOL='${{ vars.OLLAMA_PROTOCOL }}'" >> .env
|
||||||
|
echo "OLLAMA_HOST='${{ vars.OLLAMA_HOST }}'" >> .env
|
||||||
|
echo "OLLAMA_PORT='${{ vars.OLLAMA_PORT }}'" >> .env
|
||||||
|
echo "EMBEDDING_MODEL='${{ vars.EMBEDDING_MODEL }}'" >> .env
|
||||||
|
echo "EDITOR_MODEL='${{ vars.EDITOR_MODEL }}'" >> .env
|
||||||
|
export PURE='["${{ vars.CONTENT_CREATOR_MODELS_1 }}", "${{ vars.CONTENT_CREATOR_MODELS_2 }}", "${{ vars.CONTENT_CREATOR_MODELS_3 }}", "${{ vars.CONTENT_CREATOR_MODELS_4 }}"]'
|
||||||
|
echo "CONTENT_CREATOR_MODELS='$PURE'" >> .env
|
||||||
|
echo "GIT_PROTOCOL='${{ vars.GIT_PROTOCOL }}'" >> .env
|
||||||
|
echo "GIT_REMOTE='${{ vars.GIT_REMOTE }}'" >> .env
|
||||||
|
echo "GIT_USER='${{ vars.GIT_USER }}'" >> .env
|
||||||
|
echo "GIT_PASS='${{ secrets.GIT_PASS }}'" >> .env
|
||||||
|
echo "N8N_SECRET='${{ secrets.N8N_SECRET }}'" >> .env
|
||||||
|
echo "N8N_WEBHOOK_URL='${{ vars.N8N_WEBHOOK_URL }}'" >> .env
|
||||||
|
echo "CHROMA_HOST='${{ vars.CHROMA_HOST }}'" >> .env
|
||||||
|
echo "CHROMA_PORT='${{ vars.CHROMA_PORT }}'" >> .env
|
||||||
|
echo "OLLAMA_API_KEY='${{ secrets.OLLAMA_API_KEY }}'" >> .env
|
||||||
|
|
||||||
|
- name: Create Blogs
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
source .venv/bin/activate
|
||||||
|
python src/main.py
|
||||||
@@ -2,3 +2,9 @@
|
|||||||
__pycache__
|
__pycache__
|
||||||
.venv
|
.venv
|
||||||
.aider*
|
.aider*
|
||||||
|
.vscode
|
||||||
|
.zed
|
||||||
|
pyproject.toml
|
||||||
|
.ropeproject
|
||||||
|
generated_files/*
|
||||||
|
pyright*
|
||||||
|
|||||||
+6
-2
@@ -7,8 +7,12 @@ ENV PYTHONUNBUFFERED 1
|
|||||||
|
|
||||||
ADD src/ /blog_creator
|
ADD src/ /blog_creator
|
||||||
|
|
||||||
RUN apt-get update && apt-get install -y rustc cargo python-is-python3 pip python3.12-venv libmagic-dev
|
RUN apt-get update && apt-get install -y rustc cargo python-is-python3 pip python3-venv libmagic-dev git
|
||||||
|
# Need to set up git here or we get funky errors
|
||||||
|
RUN git config --global user.name "Blog Creator"
|
||||||
|
RUN git config --global user.email "ridgway.infrastructure@gmail.com"
|
||||||
|
RUN git config --global push.autoSetupRemote true
|
||||||
|
#Get a python venv going as well cause safety
|
||||||
RUN python -m venv /opt/venv
|
RUN python -m venv /opt/venv
|
||||||
ENV PATH="/opt/venv/bin:$PATH"
|
ENV PATH="/opt/venv/bin:$PATH"
|
||||||
|
|
||||||
|
|||||||
@@ -1,55 +1,290 @@
|
|||||||
## BLOG CREATOR
|
# Blog Creator
|
||||||
|
|
||||||
This creator requires you to use a working Trilium Instance and create a .env file with the following
|
An automated blog generation system that uses CrewAI agents to research, write, and edit blog posts from Trilium notes.
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
|
||||||
|
The system uses three CrewAI crews orchestrated by a Flow:
|
||||||
|
|
||||||
|
1. **Research Crew** - A critical researcher agent with web search capabilities investigates the topic and produces verified findings
|
||||||
|
2. **Writing Crew** - Four creative journalist agents write draft blog articles in parallel, each with different creative styles
|
||||||
|
3. **Editor Crew** - A critical editor loads the drafts into a vector database, queries for relevant context, and produces the final polished document with metadata
|
||||||
|
|
||||||
|
## Requirements
|
||||||
|
|
||||||
|
- Python 3.10 or later
|
||||||
|
- Ollama server running with required models
|
||||||
|
- ChromaDB server for vector storage
|
||||||
|
- Trilium notes instance
|
||||||
|
- Gitea instance (for automated workflows)
|
||||||
|
- n8n instance (for notifications)
|
||||||
|
|
||||||
|
## Environment Variables
|
||||||
|
|
||||||
|
Create a `.env` file in the project root with the following variables:
|
||||||
|
|
||||||
```
|
```
|
||||||
TRILIUM_HOST
|
# Trilium Configuration
|
||||||
TRILIUM_PORT
|
TRILIUM_HOST=
|
||||||
TRILIUM_PROTOCOL
|
TRILIUM_PORT=
|
||||||
TRILIUM_PASS
|
TRILIUM_PROTOCOL=https
|
||||||
|
TRILIUM_PASS=
|
||||||
|
TRILIUM_TOKEN=
|
||||||
|
|
||||||
|
# Ollama Configuration
|
||||||
|
OLLAMA_PROTOCOL=http
|
||||||
|
OLLAMA_HOST=
|
||||||
|
OLLAMA_PORT=11434
|
||||||
|
EMBEDDING_MODEL=nomic-embed-text
|
||||||
|
EDITOR_MODEL=llama3.1:8b
|
||||||
|
CONTENT_CREATOR_MODELS=["phi4-mini:latest", "qwen3:1.7b", "gemma3:latest"]
|
||||||
|
|
||||||
|
# ChromaDB Configuration
|
||||||
|
CHROMA_HOST=chroma
|
||||||
|
CHROMA_PORT=8000
|
||||||
|
|
||||||
|
# Git Configuration
|
||||||
|
GIT_USER=
|
||||||
|
GIT_PASS=
|
||||||
|
GIT_PROTOCOL=https
|
||||||
|
GIT_REMOTE=git.aridgwayweb.com/armistace/blog.git
|
||||||
|
|
||||||
|
# Notification Configuration
|
||||||
|
N8N_SECRET=
|
||||||
|
N8N_WEBHOOK_URL=
|
||||||
|
|
||||||
|
# Ollama Web Search (required for researcher agent)
|
||||||
|
OLLAMA_API_KEY=
|
||||||
```
|
```
|
||||||
|
|
||||||
This container is going to be what I use to trigger a blog creation event
|
### CONTENT_CREATOR_MODELS Format
|
||||||
|
|
||||||
To do this we will
|
The `CONTENT_CREATOR_MODELS` variable should be a JSON array of Ollama model names. Each model will be used by one of the three journalist agents. Example:
|
||||||
|
|
||||||
1. Download a Note from Trillium (I need to work out how to choose this, maybe something with a tag and then this can add a tag when it's used? each note is a seperate post, a tag to indicate if it's ready as well?)
|
|
||||||
|
|
||||||
`SELECT NOTES WHERE blog_tag = true AND used_tag = false AND ready_tag = true?`
|
|
||||||
|
|
||||||
2. Check if the ollama server is available (it's currently on a box that may not be on)
|
|
||||||
|
|
||||||
- If not on stop
|
|
||||||
|
|
||||||
3. `git pull git.aridgwayweb.com/blog`
|
|
||||||
|
|
||||||
- set up git creds: git.name = ai git.email = ridgwayinfrastructure@gmail.com get git password stored (create service user in gitea for this)
|
|
||||||
|
|
||||||
- `git config set upstream Auto true`
|
|
||||||
|
|
||||||
4. cd /src/content
|
|
||||||
|
|
||||||
5. take the information from the trillium note and prepare a 500 word blog post, insert the following at the top
|
|
||||||
|
|
||||||
```
|
```
|
||||||
Title: <title>
|
CONTENT_CREATOR_MODELS=["llama3.1:8b", "qwen2.5:7b", "phi4:latest"]
|
||||||
Date: <date post created>
|
|
||||||
Modified: <date post created>
|
|
||||||
Category: <this will come from a tag on the post (category: <category>)
|
|
||||||
Tags: <ai generated tags>, ai_content, not_human_content
|
|
||||||
Slug: <have ai write slug?>
|
|
||||||
Authors: <model name>.ai
|
|
||||||
Summary: <have ai write a 10 word summary of the post
|
|
||||||
```
|
```
|
||||||
|
|
||||||
6. write it to `<title>.md`
|
### OLLAMA_API_KEY
|
||||||
|
|
||||||
7. `git checkout -b <title>`
|
The researcher agent uses Ollama's native web search API. Create an API key from your Ollama account (https://ollama.com) and add it to your `.env` file. This uses your existing Ollama subscription for web searches.
|
||||||
|
|
||||||
8. `git add .`
|
## Project Structure
|
||||||
|
|
||||||
9. `git commit -m "<have ai write a git commit about the post>"`
|
```
|
||||||
|
blog_creator/
|
||||||
|
├── .env # Environment variables (create this)
|
||||||
|
├── .gitea/workflows/deploy.yml # Gitea Actions workflow
|
||||||
|
├── docker-compose.yml # Local development setup
|
||||||
|
├── requirements.txt # Python dependencies
|
||||||
|
├── README.md # This file
|
||||||
|
└── src/
|
||||||
|
├── main.py # Entry point
|
||||||
|
└── ai_generators/
|
||||||
|
├── ollama_md_generator.py # Main interface (used by main.py)
|
||||||
|
├── blog_flow.py # CrewAI Flow orchestrator
|
||||||
|
├── crews/
|
||||||
|
│ ├── research_crew/ # Researcher agent with web search
|
||||||
|
│ ├── writing_crew/ # Three journalist agents
|
||||||
|
│ └── editor_crew/ # Editor agent with metadata generation
|
||||||
|
└── tools/
|
||||||
|
```
|
||||||
|
|
||||||
10. `git push`
|
## Local Development Setup
|
||||||
|
|
||||||
11. Send notification via n8n to matrix for me to review?
|
### Using Docker Compose
|
||||||
|
|
||||||
|
1. Clone the repository and navigate to the project directory
|
||||||
|
|
||||||
|
2. Create your `.env` file with all required variables
|
||||||
|
|
||||||
|
3. Start the services:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker-compose up -d
|
||||||
|
```
|
||||||
|
|
||||||
|
This starts:
|
||||||
|
- `blog_creator` - The main application container
|
||||||
|
- `chroma` - ChromaDB vector database
|
||||||
|
|
||||||
|
4. The container will run `main.py` automatically on startup. To run manually:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
docker-compose exec blog_creator python src/main.py
|
||||||
|
```
|
||||||
|
|
||||||
|
### Manual Setup (without Docker)
|
||||||
|
|
||||||
|
1. Install system dependencies:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
apt update && apt install -y rustc cargo python-is-python3 pip python3-venv libmagic-dev git
|
||||||
|
```
|
||||||
|
|
||||||
|
2. Create and activate a virtual environment:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python -m venv .venv
|
||||||
|
source .venv/bin/activate
|
||||||
|
```
|
||||||
|
|
||||||
|
3. Install Python dependencies:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
pip install -r requirements.txt
|
||||||
|
```
|
||||||
|
|
||||||
|
4. Configure Git:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
git config --global user.name "Blog Creator"
|
||||||
|
git config --global user.email "your-email@example.com"
|
||||||
|
git config --global push.autoSetupRemote true
|
||||||
|
```
|
||||||
|
|
||||||
|
5. Run the application:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
python src/main.py
|
||||||
|
```
|
||||||
|
|
||||||
|
## How It Works
|
||||||
|
|
||||||
|
### Trilium Integration
|
||||||
|
|
||||||
|
The system fetches notes from Trilium that are tagged for blog creation. Each note becomes one blog post. The note content is used as the basis for the AI-generated article.
|
||||||
|
|
||||||
|
### Blog Generation Flow
|
||||||
|
|
||||||
|
1. **Research Phase** - The researcher agent investigates the topic using web search, critically evaluates claims, and produces verified findings
|
||||||
|
|
||||||
|
2. **Writing Phase** - Three journalist agents write creative drafts in parallel, each with different temperature and top_p settings for variety
|
||||||
|
|
||||||
|
3. **Editor Phase** - The editor:
|
||||||
|
- Chunks and embeds all drafts into ChromaDB
|
||||||
|
- Queries the vector database for relevant context
|
||||||
|
- Generates the final polished document with metadata header
|
||||||
|
|
||||||
|
### Output Format
|
||||||
|
|
||||||
|
Each blog post includes a metadata header followed by the markdown body:
|
||||||
|
|
||||||
|
```
|
||||||
|
Title: Designing and Building an AI Enhanced CCTV System
|
||||||
|
Date: 2026-02-02 20:00
|
||||||
|
Modified: 2026-02-02 20:00
|
||||||
|
Category: Homelab
|
||||||
|
Tags: proxmox, hardware, self host, homelab, ai_content, not_human_content
|
||||||
|
Slug: ai-enhanced-cctv
|
||||||
|
Authors: phi4-mini.ai, qwen3.ai, gemma3.ai
|
||||||
|
Summary: Home CCTV Security has become a bastion of cloud subscription awfulness. This blog describes creating your own AI enhanced system.
|
||||||
|
|
||||||
|
<full markdown blog body follows>
|
||||||
|
```
|
||||||
|
|
||||||
|
The metadata fields are generated as follows:
|
||||||
|
- **Title** - From the Trilium note title
|
||||||
|
- **Date/Modified** - Current datetime when generated
|
||||||
|
- **Category** - AI-generated single word (e.g., Homelab, DevOps, Security)
|
||||||
|
- **Tags** - AI-generated relevant tags plus `ai_content, not_human_content`
|
||||||
|
- **Slug** - AI-generated URL-friendly slug
|
||||||
|
- **Authors** - Derived from CONTENT_CREATOR_MODELS (model name + `.ai`)
|
||||||
|
- **Summary** - AI-generated 15-25 word summary
|
||||||
|
|
||||||
|
### Git Workflow
|
||||||
|
|
||||||
|
After generation, the blog post is:
|
||||||
|
1. Committed to a new branch named after the slug
|
||||||
|
2. Pushed to the configured Git remote
|
||||||
|
3. A notification is sent via n8n to Matrix for review
|
||||||
|
|
||||||
|
## Gitea Actions Workflow
|
||||||
|
|
||||||
|
The `.gitea/workflows/deploy.yml` file defines an automated workflow that:
|
||||||
|
|
||||||
|
- Runs on a schedule (daily at 18:15 UTC) or on push to master branch
|
||||||
|
- Installs all dependencies
|
||||||
|
- Creates the `.env` file from Gitea secrets and variables
|
||||||
|
- Runs the blog generation script
|
||||||
|
|
||||||
|
### Setting Up Gitea Variables
|
||||||
|
|
||||||
|
In your Gitea repository settings, configure the following:
|
||||||
|
|
||||||
|
**Variables** (Repository Settings -> Variables):
|
||||||
|
- `TRILIUM_HOST` - Your Trilium server hostname
|
||||||
|
- `TRILIUM_PORT` - Trilium port
|
||||||
|
- `TRILIUM_PROTOCOL` - http or https
|
||||||
|
- `OLLAMA_PROTOCOL` - http or https
|
||||||
|
- `OLLAMA_HOST` - Ollama server hostname
|
||||||
|
- `OLLAMA_PORT` - Ollama port (default 11434)
|
||||||
|
- `EMBEDDING_MODEL` - Embedding model name
|
||||||
|
- `EDITOR_MODEL` - Editor/Researcher model name
|
||||||
|
- `CONTENT_CREATOR_MODELS_1` through `CONTENT_CREATOR_MODELS_4` - Individual model names (the workflow joins these into an array)
|
||||||
|
- `GIT_PROTOCOL` - https or ssh
|
||||||
|
- `GIT_REMOTE` - Git repository URL
|
||||||
|
- `GIT_USER` - Git username for pushing
|
||||||
|
- `N8N_WEBHOOK_URL` - n8n webhook URL for notifications
|
||||||
|
- `CHROMA_HOST` - ChromaDB hostname
|
||||||
|
- `CHROMA_PORT` - ChromaDB port
|
||||||
|
|
||||||
|
**Secrets** (Repository Settings -> Secrets):
|
||||||
|
- `TRILIUM_PASS` - Trilium password
|
||||||
|
- `TRILIUM_TOKEN` - Trilium API token
|
||||||
|
- `GIT_PASS` - Git password or personal access token
|
||||||
|
- `N8N_SECRET` - n8n webhook secret key
|
||||||
|
- `OLLAMA_API_KEY` - Ollama API key for web search
|
||||||
|
|
||||||
|
### Workflow Triggers
|
||||||
|
|
||||||
|
The workflow runs automatically when:
|
||||||
|
- A push is made to the master branch
|
||||||
|
- The scheduled cron time is reached (18:15 UTC daily)
|
||||||
|
|
||||||
|
To trigger manually, push any change to master or modify the cron schedule in `.gitea/workflows/deploy.yml`.
|
||||||
|
|
||||||
|
## Customizing Agent Behavior
|
||||||
|
|
||||||
|
Agent personalities and task instructions are defined in YAML files under `src/ai_generators/crews/*/config/`. You can modify these without changing Python code:
|
||||||
|
|
||||||
|
- `research_crew/config/agents.yaml` - Researcher role, goal, backstory
|
||||||
|
- `research_crew/config/tasks.yaml` - Research task description
|
||||||
|
- `writing_crew/config/agents.yaml` - Four journalist personalities
|
||||||
|
- `writing_crew/config/tasks.yaml` - Writing task descriptions
|
||||||
|
- `editor_crew/config/agents.yaml` - Editor role, goal, backstory
|
||||||
|
- `editor_crew/config/tasks.yaml` - Editing task and metadata format
|
||||||
|
|
||||||
|
After editing YAML files, restart the application or container to apply changes.
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
### Ollama Connection Errors
|
||||||
|
|
||||||
|
Ensure the Ollama server is running and accessible from the blog_creator container. Check `OLLAMA_HOST` and `OLLAMA_PORT` in your `.env` file.
|
||||||
|
|
||||||
|
### ChromaDB Connection Errors
|
||||||
|
|
||||||
|
Verify ChromaDB is running and the `CHROMA_HOST` and `CHROMA_PORT` variables are correct. In Docker Compose, use `chroma` as the host name.
|
||||||
|
|
||||||
|
### Ollama Web Search Errors
|
||||||
|
|
||||||
|
If the researcher agent fails with web search errors, check that `OLLAMA_API_KEY` is set correctly. Verify your Ollama subscription is active and has web search access.
|
||||||
|
|
||||||
|
### Empty Output
|
||||||
|
|
||||||
|
If blog posts are generated but empty, check:
|
||||||
|
- Ollama models are downloaded and available
|
||||||
|
- `CONTENT_CREATOR_MODELS` contains valid model names
|
||||||
|
- Sufficient timeout for model inference (default is 30 minutes per operation)
|
||||||
|
|
||||||
|
### Git Push Failures
|
||||||
|
|
||||||
|
Verify `GIT_USER` and `GIT_PASS` are correct and the user has write access to the remote repository. Check that the remote URL in `GIT_REMOTE` is accessible.
|
||||||
|
|
||||||
|
## Development Notes
|
||||||
|
|
||||||
|
- The `main.py` entry point should not be modified for normal operation
|
||||||
|
- All AI generation logic is in `src/ai_generators/`
|
||||||
|
- The Flow pattern allows easy addition of new crews or steps
|
||||||
|
- Vector database collections are named `blog_{title}_{random_id}` and persist across runs
|
||||||
@@ -1,3 +1,7 @@
|
|||||||
|
networks:
|
||||||
|
net:
|
||||||
|
driver: bridge
|
||||||
|
|
||||||
services:
|
services:
|
||||||
blog_creator:
|
blog_creator:
|
||||||
build:
|
build:
|
||||||
@@ -8,4 +12,33 @@ services:
|
|||||||
- .env
|
- .env
|
||||||
volumes:
|
volumes:
|
||||||
- ./generated_files/:/blog_creator/generated_files
|
- ./generated_files/:/blog_creator/generated_files
|
||||||
|
networks:
|
||||||
|
- net
|
||||||
|
|
||||||
|
chroma:
|
||||||
|
image: chromadb/chroma
|
||||||
|
container_name: chroma
|
||||||
|
volumes:
|
||||||
|
# Be aware that indexed data are located in "/chroma/chroma/"
|
||||||
|
# Default configuration for persist_directory in chromadb/config.py
|
||||||
|
# Read more about deployments: https://docs.trychroma.com/deployment
|
||||||
|
- chroma-data:/chroma/chroma
|
||||||
|
#command: "--host 0.0.0.0 --port 8000 --proxy-headers --log-config chromadb/log_config.yml --timeout-keep-alive 30"
|
||||||
|
environment:
|
||||||
|
- IS_PERSISTENT=TRUE
|
||||||
|
restart: unless-stopped # possible values are: "no", always", "on-failure", "unless-stopped"
|
||||||
|
ports:
|
||||||
|
- "8000:8000"
|
||||||
|
healthcheck:
|
||||||
|
# Adjust below to match your container port
|
||||||
|
test:
|
||||||
|
["CMD", "curl", "-f", "http://localhost:8000/api/v2/heartbeat"]
|
||||||
|
interval: 30s
|
||||||
|
timeout: 10s
|
||||||
|
retries: 3
|
||||||
|
networks:
|
||||||
|
- net
|
||||||
|
|
||||||
|
volumes:
|
||||||
|
chroma-data:
|
||||||
|
driver: local
|
||||||
|
|||||||
@@ -0,0 +1,2 @@
|
|||||||
|
*
|
||||||
|
!.gitignore
|
||||||
@@ -1,83 +0,0 @@
|
|||||||
<think>
|
|
||||||
Alright, I've got this query from someone who wants to create an Ollama Blog Writer using Python. Let me break down what they're asking for.
|
|
||||||
|
|
||||||
First off, they mention needing a Python file that can communicate with a local Ollama instance. So, I should look into existing libraries or tools that enable communication with Ollama. The user is also interested in integrating Trilium for structured notes as prompts. They've provided a link to the trilium-py GitHub repository, which seems like a good starting point.
|
|
||||||
|
|
||||||
Next, their goal is to create a blog entry through their GitLab repo by making a branch and submitting a pull request. They want the PR content approved before proceeding further. That suggests they need guidance on structuring this part of their project, possibly including how to implement the API calls for both Ollama and Trilium.
|
|
||||||
|
|
||||||
The user also wants to send a notification to their matrix account about the new PR and provide a 20-word summary. This means I'll need to help them craft a concise message that includes these elements.
|
|
||||||
|
|
||||||
Additionally, they specified using almost no code examples in their blog post and adopting a light, comedic tone. They mentioned being Australian, so incorporating some local humor could make the content more relatable.
|
|
||||||
|
|
||||||
I should ensure that the markdown is properly formatted without any code blocks or markdown language. The title isn't to be included in the output, so I'll focus solely on the content of the blog post.
|
|
||||||
|
|
||||||
Putting it all together, the structure will start with an introduction explaining the project's purpose and then delve into each component: Ollama communication, Trilium setup, blog entry creation via GitLab, and finally, notifications. Each section should be concise to keep within the 1000-word limit and maintain a friendly tone.
|
|
||||||
|
|
||||||
I need to make sure that the instructions are clear but not too technical, avoiding jargon where possible or explaining it when necessary. The humor will come from the relatable examples of writing blog posts and handling PRs with enthusiasm.
|
|
||||||
</think>
|
|
||||||
|
|
||||||
# Creating an Ollama Blog Writer
|
|
||||||
|
|
||||||
Alright, fellow tech enthusiasts! Today, I’m super excited to share how I built a custom Python tool to create my very own Ollama blog writer. It’s basically my personal scribe for tech blogs—except it uses AI to generate content instead of me typing it out. Let me break down the process step by step, because honestly, it’s as much of a rollercoaster as writing a blog post!
|
|
||||||
|
|
||||||
## Step 1: Communicating with Ollama
|
|
||||||
|
|
||||||
First things first, I needed to connect my Python script to a running Ollama instance. Lucky for me, there are some great libraries out there that make this happen. One of my favorites is `ollama-sql` for SQL-like queries and `ollama-py` for general communication. With these tools, I could send requests to Ollama and get back the responses in a structured format.
|
|
||||||
|
|
||||||
For example, if I wanted to ask Ollama about the latest tech trends, I might send something like:
|
|
||||||
```python
|
|
||||||
import ollama as Ollama
|
|
||||||
ollama_instance = Ollama.init()
|
|
||||||
response = ollama_instance.query("What are the top AI developments this year?")
|
|
||||||
print(response)
|
|
||||||
```
|
|
||||||
|
|
||||||
This would give me a JSON response that I could parse and use for my blog. Easy peasy!
|
|
||||||
|
|
||||||
## Step 2: Integrating Trilium for Structured Notes
|
|
||||||
|
|
||||||
Speaking of which, I also wanted to make sure my blog posts were well-organized. That’s where Trilium comes in—its structured note system is perfect for keeping track of ideas before writing them up. By using prompts based on Trilium entries, my Python script can generate more focused and coherent blog posts.
|
|
||||||
|
|
||||||
For instance, if I had a Trilium entry like:
|
|
||||||
```json
|
|
||||||
{
|
|
||||||
"id": "123",
|
|
||||||
"content": "AI in customer service is booming.",
|
|
||||||
"type": "thought"
|
|
||||||
}
|
|
||||||
```
|
|
||||||
I could use that as a prompt to generate something like:
|
|
||||||
*"In the rapidly evolving landscape of AI applications, customer service has taken a quantum leap with AI-powered platforms...."*
|
|
||||||
|
|
||||||
Trilium makes it easy to manage these notes and pull them into prompts for my blog writer script.
|
|
||||||
|
|
||||||
## Step 3: Creating Blog Entries in My GitLab Repo
|
|
||||||
|
|
||||||
Now, here’s where things get interesting (and slightly nerve-wracking). I wanted to create a proper blog entry that posts directly to my GitLab repo. So, I forked the [aridgwayweb/blog](https://git.aridgwayweb.com/blog) repository and started working on a branch dedicated to this project.
|
|
||||||
|
|
||||||
In my `create_blog_entry.py` script, I used GitLab’s API to create a new entry. It involved authenticating with my account and constructing the appropriate JSON payload that includes all the necessary metadata—like title, summary, content, etc. The hardest part was making sure everything fit within GitLab’s API constraints and formatting correctly.
|
|
||||||
|
|
||||||
Here’s an excerpt of what I sent:
|
|
||||||
```python
|
|
||||||
import gitlab
|
|
||||||
gl = gitlab.Gitlab('gitlab.com', 'your_api_key')
|
|
||||||
entry = gl.entries.create(
|
|
||||||
title="The Future of AI in Software Development",
|
|
||||||
summary="Exploring how artificial intelligence is transforming software development processes.",
|
|
||||||
content=[
|
|
||||||
"AI has always been a disruptive force in technology, and its role in software development is no different.",
|
|
||||||
"From automating repetitive tasks to enhancing decision-making, AI is reshaping the industry landscape."
|
|
||||||
]
|
|
||||||
)
|
|
||||||
```
|
|
||||||
|
|
||||||
And then I notified myself that it was done!
|
|
||||||
|
|
||||||
## Step 4: Sending Notifications via Matrix
|
|
||||||
|
|
||||||
Finally, after everything was up and running, I sent a quick notification to my matrix account about the new pull request. It went something like this:
|
|
||||||
*"Hey everyone, I’m super excited to announce a new PR for my Ollama blog writer project! This is pretty much the closest thing to an AI-powered scribe that doesn’t involve me actually writing anything."*
|
|
||||||
|
|
||||||
Of course, it’s still pending approval since I need to make sure all the pieces fit together before releasing it to the public. But hey, at least I’ve got a solid foundation to build on!
|
|
||||||
|
|
||||||
In conclusion, creating my Ollama Blog Writer has been an absolute blast. It combines my love for tech with Python and AI in ways I never imagined. Now, if only I could find a way to automate writing blog *reviews*…
|
|
||||||
@@ -1,46 +0,0 @@
|
|||||||
<think>
|
|
||||||
Okay, so I'm trying to wrap my head around this PowerBI experience for a data product. Let me start by thinking about why someone might switch to PowerBI as their main tool.
|
|
||||||
|
|
||||||
First, the blog title says it's about API performance. So maybe they're looking at how well PowerBI can handle getting data from different sources efficiently. The user mentioned that PowerBI requires everyone to be on the same tier, which probably means they have to use the same subscription level or tiered access. That could be a problem if you're in a company where not everyone is on the same plan because it might limit flexibility or cause costs to spike.
|
|
||||||
|
|
||||||
Next, pulling data with PowerBI seems limited. They say it's only useful for small tables. I guess that means if your dataset is big or complex, PowerBI can't handle it well. Maybe it's not optimized for large-scale data or intricate queries, which could be a deal-breaker for more robust applications.
|
|
||||||
|
|
||||||
Then there's the issue of being part of the Microsoft ecosystem. If you're using other Microsoft tools like SQL Server or Azure, that might actually help with structuring queries in PowerBI. But if you're outside this ecosystem, it fails. Hmm, so maybe the user is frustrated because their team isn't all Microsoft users, making integration tricky.
|
|
||||||
|
|
||||||
Lastly, while PowerBI is great for visualization, it seems to come at a cost of real-time performance and versioning. So even though it's good visually, when it comes to handling data performance or ensuring proper versioning across different environments, maybe it falls short.
|
|
||||||
|
|
||||||
Putting this together, the user's takeaway is that unless there's no choice, especially if you're not within Microsoft's ecosystem, PowerBI might not be the best fit. It could lead to failures and hurt their product's performance. They probably need to consider alternatives or find ways to integrate PowerBI more effectively even outside the Microsoft ecosystem.
|
|
||||||
</think>
|
|
||||||
|
|
||||||
# The Curious Case of PowerBI in Data Product Development
|
|
||||||
|
|
||||||
Alright, let me spill the beans on my latest adventure with PowerBI—spoiler alert: it wasn’t all smooth sailing. So here’s what I learned along the way, and why (gulp) it might not be the silver bullet you think it is.
|
|
||||||
|
|
||||||
## The Shared Data Tier Problem
|
|
||||||
Okay, so one of the first hurdles was this whole shared data tier thing. Let me tell ya, it felt like a non-starter for most companies out there. Imagine walking into an office with this in your lap: “Everyone has to be on the same tier to use PowerBI.” Yeah, sounds like a lot of bureaucracy just to get some data flowing. But then I started thinking—what if they’re not? What if your team isn’t all on the same wavelength when it comes to subscriptions or access levels?
|
|
||||||
|
|
||||||
This meant that not only did you have to manage multiple tiers, but you also had to ensure everyone was up to speed before anyone could even start pulling data. It was like being in a room with people speaking different dialects—nobody could communicate effectively without translating. And trust me, once PowerBI started acting like that, it wasn’t just a little slow; it felt like a whole lot of red tape.
|
|
||||||
|
|
||||||
## Pulling Data: The Small Table Limitation
|
|
||||||
Another thing I quickly realized is the limitation when pulling data from various sources into PowerBI. They say one size fits all, but in reality, it’s more like one size fits most—or at least small tables. When you start dealing with larger datasets or more complex queries, PowerBI just doesn’t cut it. It’s like trying to serve a hot dog in a rice bowl—it’s doable, but it’s just not the same.
|
|
||||||
|
|
||||||
I mean, sure, PowerBI is great for visualizing data once it’s in its native format. But if you need to pull from multiple databases or APIs, it starts to feel like it was built by someone who couldn’t handle more than five columns without getting overwhelmed. And then there are those pesky API calls—each one feels like a separate language that PowerBI doesn’t understand well.
|
|
||||||
|
|
||||||
## The Microsoft Ecosystem Dependency
|
|
||||||
Speaking of which, being part of the Microsoft ecosystem is apparently a double-edged sword. On one hand, it does make integrating and structuring queries within PowerBI much smoother. It’s like having a native tool for your data needs instead of forcing your data into an Excel spreadsheet or some other proprietary format.
|
|
||||||
|
|
||||||
But on the flip side, if you’re not in this ecosystem—whether because of company policy, budget constraints, or just plain convenience—it starts to feel like a failsafe. Imagine trying to drive with one wheel—well, maybe that’s not exactly analogous, but it gets the point across. Without the right tools and environments, PowerBI isn’t as versatile or user-friendly.
|
|
||||||
|
|
||||||
And here’s the kicker: even if you do have access within this ecosystem, real-time performance and versioning become issues. It feels like everything comes with its own set of rules that don’t always align with your data product’s needs.
|
|
||||||
|
|
||||||
## The Visualization vs. Performance Trade-Off
|
|
||||||
Now, I know what some of you are thinking—PowerBI is all about making data beautiful, right? And it does a fantastic job at that. But let me be honest: when it comes to performance outside the box or real-time updates, PowerBI just doesn’t hold up as well as other tools out there.
|
|
||||||
|
|
||||||
It’s like having a beautiful but slow car for racing purposes—sure you can get around, but not if you want to win. Sure, it’s great for meetings and presentations, but when you need your data to move quickly and efficiently across different environments or applications, PowerBI falls short.
|
|
||||||
|
|
||||||
## The Takeaway
|
|
||||||
So after all that, here’s my bottom line: unless you’re in the Microsoft ecosystem—top to tail—you might be better off looking elsewhere. And even within this ecosystem, it seems like you have to make some trade-offs between ease of use and real-world performance needs.
|
|
||||||
|
|
||||||
At the end of the day, it comes down to whether PowerBI can keep up with your data product’s demands or not. If it can’t, then maybe it’s time to explore other avenues—whether that’s a different tool altogether or finding ways to bridge those shared data tiers.
|
|
||||||
|
|
||||||
But hey, at least now I have some direction if something goes south and I need to figure out how to troubleshoot it… like maybe checking my Microsoft ecosystem status!
|
|
||||||
@@ -2,3 +2,8 @@ ollama
|
|||||||
trilium-py
|
trilium-py
|
||||||
gitpython
|
gitpython
|
||||||
PyGithub
|
PyGithub
|
||||||
|
chromadb
|
||||||
|
crewai
|
||||||
|
crewai-tools
|
||||||
|
PyJWT
|
||||||
|
dotenv
|
||||||
|
|||||||
@@ -0,0 +1,318 @@
|
|||||||
|
"""
|
||||||
|
CrewAI Flow that orchestrates the blog-generation pipeline.
|
||||||
|
|
||||||
|
Flow
|
||||||
|
----
|
||||||
|
1. **Research crew** – a critical researcher with web-search investigates the
|
||||||
|
topic and produces verified findings.
|
||||||
|
2. **Writing crew** – four creative journalists write draft blog articles
|
||||||
|
in parallel (async tasks).
|
||||||
|
3. **Editor crew** – a critical editor loads the journalist drafts into
|
||||||
|
ChromaDB, queries for the most relevant context, and produces the final
|
||||||
|
polished markdown document complete with a metadata header (Title, Date,
|
||||||
|
Category, Tags, Slug, Authors, Summary).
|
||||||
|
|
||||||
|
The ChromaDB integration is preserved from the original implementation: each
|
||||||
|
journalist draft is chunked, embedded, and stored in a collection; the editor
|
||||||
|
receives the top-N most relevant chunks as context.
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
import random
|
||||||
|
import re
|
||||||
|
import string
|
||||||
|
from datetime import datetime
|
||||||
|
|
||||||
|
import chromadb
|
||||||
|
from crewai.flow.flow import Flow, listen, start
|
||||||
|
from ollama import Client
|
||||||
|
from pydantic import BaseModel, ConfigDict
|
||||||
|
|
||||||
|
from ai_generators.crews.editor_crew.editor_crew import EditorCrew
|
||||||
|
from ai_generators.crews.research_crew.research_crew import ResearchCrew
|
||||||
|
from ai_generators.crews.writing_crew.writing_crew import WritingCrew
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# State
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
class BlogFlowState(BaseModel):
|
||||||
|
"""Structured state for the blog generation flow."""
|
||||||
|
|
||||||
|
model_config = ConfigDict(arbitrary_types_allowed=True)
|
||||||
|
|
||||||
|
title: str = ""
|
||||||
|
inner_title: str = ""
|
||||||
|
content: str = ""
|
||||||
|
research_findings: str = ""
|
||||||
|
drafts: list[str] = []
|
||||||
|
final_document: str = ""
|
||||||
|
date: str = ""
|
||||||
|
authors: str = ""
|
||||||
|
category: str = ""
|
||||||
|
|
||||||
|
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
# Flow
|
||||||
|
# ---------------------------------------------------------------------------
|
||||||
|
|
||||||
|
|
||||||
|
class BlogFlow(Flow[BlogFlowState]):
|
||||||
|
"""Orchestrate researcher → journalists → editor via CrewAI Flows.
|
||||||
|
|
||||||
|
Usage::
|
||||||
|
|
||||||
|
flow = BlogFlow()
|
||||||
|
result = flow.kickoff(inputs={
|
||||||
|
"title": "my_blog_slug",
|
||||||
|
"inner_title": "My Blog Title",
|
||||||
|
"content": "<original content>",
|
||||||
|
})
|
||||||
|
print(result) # final markdown document
|
||||||
|
"""
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Helpers – Ollama / ChromaDB / embedding utilities
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _get_ollama_url() -> str:
|
||||||
|
return (
|
||||||
|
f"{os.environ['OLLAMA_PROTOCOL']}://"
|
||||||
|
f"{os.environ['OLLAMA_HOST']}:{os.environ['OLLAMA_PORT']}"
|
||||||
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _get_chroma_client() -> chromadb.HttpClient:
|
||||||
|
chroma_port = int(os.environ["CHROMA_PORT"])
|
||||||
|
return chromadb.HttpClient(host=os.environ["CHROMA_HOST"], port=chroma_port)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _get_ollama_client() -> Client:
|
||||||
|
return Client(host=BlogFlow._get_ollama_url())
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _id_generator(size: int = 6) -> str:
|
||||||
|
return "".join(
|
||||||
|
random.choice(string.ascii_uppercase + string.digits) for _ in range(size)
|
||||||
|
)
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _split_into_chunks(text: str, chunk_size: int = 100) -> list[str]:
|
||||||
|
words = re.findall(r"\S+", text)
|
||||||
|
chunks: list[str] = []
|
||||||
|
current_chunk: list[str] = []
|
||||||
|
word_count = 0
|
||||||
|
for word in words:
|
||||||
|
current_chunk.append(word)
|
||||||
|
word_count += 1
|
||||||
|
if word_count >= chunk_size:
|
||||||
|
chunks.append(" ".join(current_chunk))
|
||||||
|
current_chunk = []
|
||||||
|
word_count = 0
|
||||||
|
if current_chunk:
|
||||||
|
chunks.append(" ".join(current_chunk))
|
||||||
|
return chunks
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _get_embeddings(chunks: list[str]) -> list[list[float]]:
|
||||||
|
ollama_client = BlogFlow._get_ollama_client()
|
||||||
|
embed_model = os.environ["EMBEDDING_MODEL"]
|
||||||
|
try:
|
||||||
|
embeds = ollama_client.embed(model=embed_model, input=chunks)
|
||||||
|
return embeds.get("embeddings", []) # type: ignore[no-any-return]
|
||||||
|
except Exception as exc:
|
||||||
|
print(f"Error generating embeddings: {exc}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
def _load_drafts_to_vector_db(self, drafts: list[str]) -> chromadb.Collection:
|
||||||
|
"""Load journalist drafts into a new ChromaDB collection and return it."""
|
||||||
|
chroma = self._get_chroma_client()
|
||||||
|
collection_name = (
|
||||||
|
f"blog_{self.state.title.lower().replace(' ', '_')}_{self._id_generator()}"
|
||||||
|
)
|
||||||
|
collection = chroma.get_or_create_collection(name=collection_name)
|
||||||
|
|
||||||
|
for i, draft in enumerate(drafts):
|
||||||
|
model_name = f"journalist_{i + 1}"
|
||||||
|
chunks = self._split_into_chunks(draft)
|
||||||
|
if not chunks or all(chunk.strip() == "" for chunk in chunks):
|
||||||
|
print(f"Skipping {model_name} – no content generated")
|
||||||
|
continue
|
||||||
|
print(f"Generating embeddings for {model_name}")
|
||||||
|
embeds = self._get_embeddings(chunks)
|
||||||
|
if not embeds:
|
||||||
|
print(f"Skipping {model_name} – no embeddings generated")
|
||||||
|
continue
|
||||||
|
if len(embeds) != len(chunks):
|
||||||
|
min_length = min(len(embeds), len(chunks))
|
||||||
|
chunks = chunks[:min_length]
|
||||||
|
embeds = embeds[:min_length]
|
||||||
|
if min_length == 0:
|
||||||
|
print(f"Skipping {model_name} – no valid content/embeddings pairs")
|
||||||
|
continue
|
||||||
|
ids = [model_name + str(j) for j in range(len(chunks))]
|
||||||
|
metadata = [{"model_agent": model_name} for _ in chunks]
|
||||||
|
print(f"Loading into collection for {model_name}")
|
||||||
|
collection.add(
|
||||||
|
documents=chunks,
|
||||||
|
embeddings=embeds, # type: ignore[arg-type]
|
||||||
|
ids=ids,
|
||||||
|
metadatas=metadata, # type: ignore[arg-type]
|
||||||
|
)
|
||||||
|
return collection
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _query_vector_db(collection: chromadb.Collection, query_text: str) -> str:
|
||||||
|
"""Query the ChromaDB collection and return the most relevant
|
||||||
|
document chunks joined as a single string."""
|
||||||
|
ollama_client = BlogFlow._get_ollama_client()
|
||||||
|
embed_model = os.environ["EMBEDDING_MODEL"]
|
||||||
|
try:
|
||||||
|
embed_result = ollama_client.embed(model=embed_model, input=query_text)
|
||||||
|
query_embed = embed_result.get("embeddings", [])
|
||||||
|
if not query_embed:
|
||||||
|
print(
|
||||||
|
"Warning: Failed to generate query embeddings, "
|
||||||
|
"falling back to empty list"
|
||||||
|
)
|
||||||
|
query_embed = [[]]
|
||||||
|
except Exception as exc:
|
||||||
|
print(f"Error generating query embeddings: {exc}")
|
||||||
|
query_embed = [[]]
|
||||||
|
|
||||||
|
try:
|
||||||
|
query_result = collection.query(
|
||||||
|
query_embeddings=query_embed,
|
||||||
|
n_results=100, # type: ignore[arg-type]
|
||||||
|
)
|
||||||
|
documents = query_result.get("documents", [])
|
||||||
|
if documents and len(documents) > 0 and len(documents[0]) > 0:
|
||||||
|
return "\n\n".join(documents[0])
|
||||||
|
print("Warning: No relevant documents found in collection")
|
||||||
|
return "No relevant information found in drafts."
|
||||||
|
except Exception as exc:
|
||||||
|
print(f"Error querying collection: {exc}")
|
||||||
|
return "No relevant information found in drafts due to query error."
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Flow steps
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
|
||||||
|
@start()
|
||||||
|
def research(self) -> str:
|
||||||
|
"""Run the research crew to investigate the blog topic."""
|
||||||
|
print("=" * 60)
|
||||||
|
print("RESEARCH PHASE – investigating topic")
|
||||||
|
print("=" * 60)
|
||||||
|
|
||||||
|
result = (
|
||||||
|
ResearchCrew()
|
||||||
|
.crew()
|
||||||
|
.kickoff(
|
||||||
|
inputs={
|
||||||
|
"inner_title": self.state.inner_title,
|
||||||
|
"content": self.state.content,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.state.research_findings = result.raw
|
||||||
|
print("Research phase complete")
|
||||||
|
return result.raw
|
||||||
|
|
||||||
|
@listen(research)
|
||||||
|
def write_drafts(self, research_findings: str) -> list[str]:
|
||||||
|
"""Run the writing crew (4 journalists in parallel) and collect
|
||||||
|
their draft outputs."""
|
||||||
|
print("=" * 60)
|
||||||
|
print("WRITING PHASE – 4 journalists drafting in parallel")
|
||||||
|
print("=" * 60)
|
||||||
|
|
||||||
|
result = (
|
||||||
|
WritingCrew()
|
||||||
|
.crew()
|
||||||
|
.kickoff(
|
||||||
|
inputs={
|
||||||
|
"inner_title": self.state.inner_title,
|
||||||
|
"content": self.state.content,
|
||||||
|
"research_findings": research_findings,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
# Collect all draft outputs from the crew's task outputs
|
||||||
|
drafts: list[str] = []
|
||||||
|
for task_output in result.tasks_output:
|
||||||
|
drafts.append(task_output.raw)
|
||||||
|
|
||||||
|
self.state.drafts = drafts
|
||||||
|
print(f"Writing phase complete – {len(drafts)} drafts produced")
|
||||||
|
return drafts
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _compute_authors() -> str:
|
||||||
|
"""Build an author string from the CONTENT_CREATOR_MODELS env var.
|
||||||
|
|
||||||
|
Each model name is stripped of any tag suffix (e.g. ``:latest``)
|
||||||
|
and ``.ai`` is appended. Multiple models are joined with ``', '``.
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
models = json.loads(os.environ["CONTENT_CREATOR_MODELS"])
|
||||||
|
except (KeyError, json.JSONDecodeError):
|
||||||
|
models = []
|
||||||
|
authors = ", ".join(m.split(":")[0].split("/")[-1] + ".ai" for m in models)
|
||||||
|
return authors or "unknown.ai"
|
||||||
|
|
||||||
|
@listen(write_drafts)
|
||||||
|
def edit_final(self, drafts: list[str]) -> str:
|
||||||
|
"""Load journalist drafts into the vector DB, query for the most
|
||||||
|
relevant context, and run the editor crew to produce the final
|
||||||
|
polished document with a metadata header."""
|
||||||
|
print("=" * 60)
|
||||||
|
print("EDITOR PHASE – producing final document")
|
||||||
|
print("=" * 60)
|
||||||
|
|
||||||
|
# ---- Compute date and authors for the metadata header ----
|
||||||
|
if not self.state.date:
|
||||||
|
self.state.date = datetime.now().strftime("%Y-%m-%d %H:%M")
|
||||||
|
self.state.authors = self._compute_authors()
|
||||||
|
if not self.state.category:
|
||||||
|
self.state.category = "<pick one word that best describes the topic, e.g. Homelab, DevOps, Security, Networking>"
|
||||||
|
|
||||||
|
# ---- Vector DB integration ----
|
||||||
|
print("Loading drafts into vector database")
|
||||||
|
collection = self._load_drafts_to_vector_db(drafts)
|
||||||
|
|
||||||
|
# Build the editor's brief so we can query the vector DB with it
|
||||||
|
editor_brief = (
|
||||||
|
f"You are an editor taking information from 3 Software "
|
||||||
|
f"Developers and Data experts writing a 5000 word blog article. "
|
||||||
|
f"You like when they use almost no code examples. "
|
||||||
|
f"You are also Australian. The title for the blog is "
|
||||||
|
f"{self.state.inner_title}. "
|
||||||
|
f"The basis for the content of the blog is: "
|
||||||
|
f"<blog>{self.state.content}</blog>"
|
||||||
|
)
|
||||||
|
draft_context = self._query_vector_db(collection, editor_brief)
|
||||||
|
print("Showing pertinent info from drafts used in final edited edition")
|
||||||
|
|
||||||
|
# ---- Editor crew ----
|
||||||
|
result = (
|
||||||
|
EditorCrew()
|
||||||
|
.crew()
|
||||||
|
.kickoff(
|
||||||
|
inputs={
|
||||||
|
"inner_title": self.state.inner_title,
|
||||||
|
"content": self.state.content,
|
||||||
|
"draft_context": draft_context,
|
||||||
|
"date": self.state.date,
|
||||||
|
"authors": self.state.authors,
|
||||||
|
"category": self.state.category,
|
||||||
|
}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.state.final_document = result.raw
|
||||||
|
print("Editor phase complete")
|
||||||
|
return result.raw
|
||||||
@@ -0,0 +1,20 @@
|
|||||||
|
editor:
|
||||||
|
role: >
|
||||||
|
Critical Blog Editor
|
||||||
|
goal: >
|
||||||
|
Produce the final, polished ~5000-word version of a blog about {inner_title},
|
||||||
|
complete with a metadata header (Title, Date, Category, Tags, Slug, Authors,
|
||||||
|
Summary)
|
||||||
|
backstory: >
|
||||||
|
You are an editor taking information from 3 Software Developers and
|
||||||
|
Data experts writing a 5000 word blog article. You like when they use
|
||||||
|
almost no code examples. You are also Australian. The content may have
|
||||||
|
light comedic elements; you are more professional and will attempt to
|
||||||
|
tone these down. You are critical of repeated sentences, inconsistencies,
|
||||||
|
and weak arguments. You ensure the final document is cohesive,
|
||||||
|
well-structured, and publication-ready. You never leave placeholder
|
||||||
|
text — every section must contain finished content. You always begin
|
||||||
|
your output with a plain-text metadata block (Title, Date, Modified,
|
||||||
|
Category, Tags, Slug, Authors, Summary) followed by a blank line and
|
||||||
|
then the full markdown body. You generate sensible Category, Tags,
|
||||||
|
Slug and Summary values based on the blog content.
|
||||||
@@ -0,0 +1,45 @@
|
|||||||
|
edit_task:
|
||||||
|
description: >
|
||||||
|
Generate the final, 5000 word blog post using this information
|
||||||
|
from the journalist drafts:
|
||||||
|
<context>{draft_context}</context>
|
||||||
|
|
||||||
|
You are an editor taking information from 3 Software Developers and
|
||||||
|
Data experts writing a 5000 word blog article. You like when they use
|
||||||
|
almost no code examples. You are also Australian. The content may have
|
||||||
|
light comedic elements; you are more professional and will attempt to
|
||||||
|
tone these down. As this person produce the final version of this blog
|
||||||
|
as a markdown document keeping in mind the context provided by the
|
||||||
|
previous drafts. You are to produce the content not placeholders for
|
||||||
|
further editors. The title for the blog is {inner_title}. Avoid
|
||||||
|
repeated sentences. The basis for the content of the blog is:
|
||||||
|
<blog>{content}</blog>
|
||||||
|
|
||||||
|
IMPORTANT: The output MUST start with a metadata block in exactly this
|
||||||
|
format, followed by a blank line, then the blog body. Do not wrap the
|
||||||
|
metadata block in code fences or any other markup. Generate sensible
|
||||||
|
values for Category, Tags, Slug and Summary based on the blog content.
|
||||||
|
|
||||||
|
Title: {inner_title}
|
||||||
|
Date: {date}
|
||||||
|
Modified: {date}
|
||||||
|
Category: {category}
|
||||||
|
Tags: <generate 3-5 short lowercase tags relevant to the content>, ai_content, not_human_content
|
||||||
|
Slug: <generate a short URL-friendly slug using lowercase words separated by hyphens>
|
||||||
|
Authors: {authors}
|
||||||
|
Summary: <write a single sentence summary of roughly 15-25 words>
|
||||||
|
|
||||||
|
After the metadata block and blank line, write the full blog body in
|
||||||
|
markdown. Do not repeat the title as a heading in the body.
|
||||||
|
|
||||||
|
- Only output the metadata block and then the markdown body.
|
||||||
|
- Do not wrap in markdown code fences.
|
||||||
|
- Do not provide a commentary on the drafts in the context.
|
||||||
|
- Produce real content, not placeholders for further editors.
|
||||||
|
- Avoid repeated sentences.
|
||||||
|
expected_output: >
|
||||||
|
A metadata block (Title, Date, Modified, Category, Tags, Slug, Authors,
|
||||||
|
Summary) followed by a blank line and then a polished ~5000-word markdown
|
||||||
|
blog article about {inner_title}. No commentary. No placeholders. Cohesive
|
||||||
|
and publication-ready.
|
||||||
|
agent: editor
|
||||||
@@ -0,0 +1,51 @@
|
|||||||
|
"""Editor crew – produces the final polished blog document."""
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
from crewai import LLM, Agent, Crew, Process, Task
|
||||||
|
from crewai.project import CrewBase, agent, crew, task
|
||||||
|
|
||||||
|
|
||||||
|
def _get_ollama_url() -> str:
|
||||||
|
return (
|
||||||
|
f"{os.environ['OLLAMA_PROTOCOL']}://"
|
||||||
|
f"{os.environ['OLLAMA_HOST']}:{os.environ['OLLAMA_PORT']}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@CrewBase
|
||||||
|
class EditorCrew:
|
||||||
|
"""Crew with a single critical editor who produces the final blog."""
|
||||||
|
|
||||||
|
agents_config = "config/agents.yaml"
|
||||||
|
tasks_config = "config/tasks.yaml"
|
||||||
|
|
||||||
|
@agent
|
||||||
|
def editor(self) -> Agent:
|
||||||
|
return Agent(
|
||||||
|
config=self.agents_config["editor"], # type: ignore[index]
|
||||||
|
llm=LLM(
|
||||||
|
model=f"ollama/{os.environ['EDITOR_MODEL']}",
|
||||||
|
base_url=_get_ollama_url(),
|
||||||
|
temperature=0.6,
|
||||||
|
top_p=0.5,
|
||||||
|
),
|
||||||
|
verbose=True,
|
||||||
|
max_iter=30,
|
||||||
|
respect_context_window=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
@task
|
||||||
|
def edit_task(self) -> Task:
|
||||||
|
return Task(
|
||||||
|
config=self.tasks_config["edit_task"], # type: ignore[index]
|
||||||
|
)
|
||||||
|
|
||||||
|
@crew
|
||||||
|
def crew(self) -> Crew:
|
||||||
|
return Crew(
|
||||||
|
agents=self.agents,
|
||||||
|
tasks=self.tasks,
|
||||||
|
process=Process.sequential,
|
||||||
|
verbose=True,
|
||||||
|
)
|
||||||
@@ -0,0 +1,15 @@
|
|||||||
|
researcher:
|
||||||
|
role: >
|
||||||
|
Critical Technology Researcher
|
||||||
|
goal: >
|
||||||
|
Research and critically evaluate information related to {inner_title}
|
||||||
|
backstory: >
|
||||||
|
You are a skeptical, thorough technology researcher with years of
|
||||||
|
experience in Software Development and DevOps. You never accept
|
||||||
|
information at face value and always cross-reference claims with
|
||||||
|
multiple sources. You are particularly critical of hype, marketing
|
||||||
|
language, and unsubstantiated technical claims. You prefer primary
|
||||||
|
sources, official documentation, and peer-reviewed material over
|
||||||
|
blog posts and opinion pieces. When conflicting information is found
|
||||||
|
you clearly note the discrepancy and provide both viewpoints with
|
||||||
|
credibility assessments.
|
||||||
@@ -0,0 +1,23 @@
|
|||||||
|
research_task:
|
||||||
|
description: >
|
||||||
|
Research the topic: {inner_title}
|
||||||
|
|
||||||
|
The original content to research and expand upon is:
|
||||||
|
<blog>{content}</blog>
|
||||||
|
|
||||||
|
Your task is to:
|
||||||
|
1. Search the web for current, accurate information related to this topic.
|
||||||
|
2. Critically evaluate the claims made in the original content.
|
||||||
|
3. Find supporting or contradicting evidence from reputable sources.
|
||||||
|
4. Identify any outdated information, common misconceptions, or factual errors.
|
||||||
|
5. Provide a comprehensive research summary with verified facts, clearly
|
||||||
|
distinguishing between confirmed information and areas of uncertainty.
|
||||||
|
|
||||||
|
Be thorough and skeptical. Only include information you can verify from
|
||||||
|
reliable sources. Flag anything that seems exaggerated or unverified.
|
||||||
|
expected_output: >
|
||||||
|
A comprehensive research report with verified facts, source citations,
|
||||||
|
and credibility assessments. Clearly distinguish between confirmed
|
||||||
|
information and areas of uncertainty. Include supporting and
|
||||||
|
contradicting evidence where found.
|
||||||
|
agent: researcher
|
||||||
@@ -0,0 +1,54 @@
|
|||||||
|
"""Research crew – investigates a blog topic using web search."""
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
from crewai import LLM, Agent, Crew, Process, Task
|
||||||
|
from crewai.project import CrewBase, agent, crew, task
|
||||||
|
|
||||||
|
from ai_generators.tools import OllamaWebSearchTool
|
||||||
|
|
||||||
|
|
||||||
|
def _get_ollama_url() -> str:
|
||||||
|
return (
|
||||||
|
f"{os.environ['OLLAMA_PROTOCOL']}://"
|
||||||
|
f"{os.environ['OLLAMA_HOST']}:{os.environ['OLLAMA_PORT']}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
@CrewBase
|
||||||
|
class ResearchCrew:
|
||||||
|
"""Crew that researches a blog topic with a critical, web-searching
|
||||||
|
researcher agent."""
|
||||||
|
|
||||||
|
agents_config = "config/agents.yaml"
|
||||||
|
tasks_config = "config/tasks.yaml"
|
||||||
|
|
||||||
|
@agent
|
||||||
|
def researcher(self) -> Agent:
|
||||||
|
return Agent(
|
||||||
|
config=self.agents_config["researcher"], # type: ignore[index]
|
||||||
|
tools=[OllamaWebSearchTool()],
|
||||||
|
llm=LLM(
|
||||||
|
model=f"ollama/{os.environ['EDITOR_MODEL']}",
|
||||||
|
base_url=_get_ollama_url(),
|
||||||
|
temperature=0.3,
|
||||||
|
),
|
||||||
|
verbose=True,
|
||||||
|
max_iter=25,
|
||||||
|
respect_context_window=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
@task
|
||||||
|
def research_task(self) -> Task:
|
||||||
|
return Task(
|
||||||
|
config=self.tasks_config["research_task"], # type: ignore[index]
|
||||||
|
)
|
||||||
|
|
||||||
|
@crew
|
||||||
|
def crew(self) -> Crew:
|
||||||
|
return Crew(
|
||||||
|
agents=self.agents,
|
||||||
|
tasks=self.tasks,
|
||||||
|
process=Process.sequential,
|
||||||
|
verbose=True,
|
||||||
|
)
|
||||||
@@ -0,0 +1,48 @@
|
|||||||
|
journalist_one:
|
||||||
|
role: >
|
||||||
|
Creative Technology Journalist
|
||||||
|
goal: >
|
||||||
|
Write a creative, engaging ~5000-word draft blog article about {inner_title}
|
||||||
|
backstory: >
|
||||||
|
You are a journalist, Software Developer and DevOps expert writing a
|
||||||
|
draft blog article for other tech enthusiasts. You like to use almost no
|
||||||
|
code examples and prefer to talk in a light comedic tone. You are also
|
||||||
|
Australian. You favour vivid analogies and storytelling to explain
|
||||||
|
technical concepts. Your writing is warm, slightly irreverent, and
|
||||||
|
accessible.
|
||||||
|
|
||||||
|
journalist_two:
|
||||||
|
role: >
|
||||||
|
Creative Technology Journalist
|
||||||
|
goal: >
|
||||||
|
Write a creative, engaging ~5000-word draft blog article about {inner_title}
|
||||||
|
backstory: >
|
||||||
|
You are a journalist, Software Developer and DevOps expert writing a
|
||||||
|
draft blog article for other tech enthusiasts. You like to use almost no
|
||||||
|
code examples and prefer to talk in a light comedic tone. You are also
|
||||||
|
Australian. You lean into sharp wit and concise, punchy sentences. You
|
||||||
|
love finding unexpected connections between seemingly unrelated topics.
|
||||||
|
|
||||||
|
journalist_three:
|
||||||
|
role: >
|
||||||
|
Creative Technology Journalist
|
||||||
|
goal: >
|
||||||
|
Write a creative, engaging ~5000-word draft blog article about {inner_title}
|
||||||
|
backstory: >
|
||||||
|
You are a journalist, Software Developer and DevOps expert writing a
|
||||||
|
draft blog article for other tech enthusiasts. You like to use almost no
|
||||||
|
code examples and prefer to talk in a light comedic tone. You are also
|
||||||
|
Australian. You prefer a conversational, meandering style that draws the
|
||||||
|
reader in with personal anecdotes and rhetorical questions.
|
||||||
|
|
||||||
|
journalist_four:
|
||||||
|
role: >
|
||||||
|
Creative Technology Journalist
|
||||||
|
goal: >
|
||||||
|
Write a creative, engaging ~5000-word draft blog article about {inner_title}
|
||||||
|
backstory: >
|
||||||
|
You are a journalist, Software Developer and DevOps expert writing a
|
||||||
|
draft blog article for other tech enthusiasts. You like to use almost no
|
||||||
|
code examples and prefer to talk in a light comedic tone. You are also
|
||||||
|
Australian. You take a methodical, analytical approach with detailed
|
||||||
|
explanations and systematic breakdowns of complex topics.
|
||||||
@@ -0,0 +1,79 @@
|
|||||||
|
write_draft_one:
|
||||||
|
description: >
|
||||||
|
Write a 5000 word draft blog article as a markdown document.
|
||||||
|
The title for the blog is {inner_title}.
|
||||||
|
Do not output the title in the markdown.
|
||||||
|
|
||||||
|
The basis for the content of the blog is:
|
||||||
|
<blog>{content}</blog>
|
||||||
|
|
||||||
|
Research findings to incorporate and validate against:
|
||||||
|
<research>{research_findings}</research>
|
||||||
|
|
||||||
|
Write creatively, with a light comedic tone. You are Australian.
|
||||||
|
Use almost no code examples. Make it engaging for tech enthusiasts.
|
||||||
|
Only output the markdown content — no commentary, no meta-description.
|
||||||
|
expected_output: >
|
||||||
|
A ~5000-word markdown draft blog article about {inner_title}.
|
||||||
|
No title in the output. No commentary or meta-description.
|
||||||
|
agent: journalist_one
|
||||||
|
|
||||||
|
write_draft_two:
|
||||||
|
description: >
|
||||||
|
Write a 5000 word draft blog article as a markdown document.
|
||||||
|
The title for the blog is {inner_title}.
|
||||||
|
Do not output the title in the markdown.
|
||||||
|
|
||||||
|
The basis for the content of the blog is:
|
||||||
|
<blog>{content}</blog>
|
||||||
|
|
||||||
|
Research findings to incorporate and validate against:
|
||||||
|
<research>{research_findings}</research>
|
||||||
|
|
||||||
|
Write creatively, with a light comedic tone. You are Australian.
|
||||||
|
Use almost no code examples. Make it engaging for tech enthusiasts.
|
||||||
|
Only output the markdown content — no commentary, no meta-description.
|
||||||
|
expected_output: >
|
||||||
|
A ~5000-word markdown draft blog article about {inner_title}.
|
||||||
|
No title in the output. No commentary or meta-description.
|
||||||
|
agent: journalist_two
|
||||||
|
|
||||||
|
write_draft_three:
|
||||||
|
description: >
|
||||||
|
Write a 5000 word draft blog article as a markdown document.
|
||||||
|
The title for the blog is {inner_title}.
|
||||||
|
Do not output the title in the markdown.
|
||||||
|
|
||||||
|
The basis for the content of the blog is:
|
||||||
|
<blog>{content}</blog>
|
||||||
|
|
||||||
|
Research findings to incorporate and validate against:
|
||||||
|
<research>{research_findings}</research>
|
||||||
|
|
||||||
|
Write creatively, with a light comedic tone. You are Australian.
|
||||||
|
Use almost no code examples. Make it engaging for tech enthusiasts.
|
||||||
|
Only output the markdown content — no commentary, no meta-description.
|
||||||
|
expected_output: >
|
||||||
|
A ~5000-word markdown draft blog article about {inner_title}.
|
||||||
|
No title in the output. No commentary or meta-description.
|
||||||
|
agent: journalist_three
|
||||||
|
|
||||||
|
write_draft_four:
|
||||||
|
description: >
|
||||||
|
Write a 5000 word draft blog article as a markdown document.
|
||||||
|
The title for the blog is {inner_title}.
|
||||||
|
Do not output the title in the markdown.
|
||||||
|
|
||||||
|
The basis for the content of the blog is:
|
||||||
|
<blog>{content}</blog>
|
||||||
|
|
||||||
|
Research findings to incorporate and validate against:
|
||||||
|
<research>{research_findings}</research>
|
||||||
|
|
||||||
|
Write creatively, with a light comedic tone. You are Australian.
|
||||||
|
Use almost no code examples. Make it engaging for tech enthusiasts.
|
||||||
|
Only output the markdown content — no commentary, no meta-description.
|
||||||
|
expected_output: >
|
||||||
|
A ~5000-word markdown draft blog article about {inner_title}.
|
||||||
|
No title in the output. No commentary or meta-description.
|
||||||
|
agent: journalist_four
|
||||||
@@ -0,0 +1,128 @@
|
|||||||
|
"""Writing crew – three journalists who write creative blog drafts in parallel."""
|
||||||
|
|
||||||
|
import json
|
||||||
|
import os
|
||||||
|
|
||||||
|
from crewai import LLM, Agent, Crew, Process, Task
|
||||||
|
from crewai.project import CrewBase, agent, crew, task
|
||||||
|
|
||||||
|
|
||||||
|
def _get_ollama_url() -> str:
|
||||||
|
return (
|
||||||
|
f"{os.environ['OLLAMA_PROTOCOL']}://"
|
||||||
|
f"{os.environ['OLLAMA_HOST']}:{os.environ['OLLAMA_PORT']}"
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def _get_agent_models() -> list[str]:
|
||||||
|
return json.loads(os.environ["CONTENT_CREATOR_MODELS"])
|
||||||
|
|
||||||
|
|
||||||
|
# Creative-style presets per journalist: (temperature, top_p)
|
||||||
|
_JOURNALIST_PARAMS: dict[int, tuple[float, float]] = {
|
||||||
|
1: (0.70, 0.60), # moderate creativity
|
||||||
|
2: (0.85, 0.50), # high creativity, tighter focus
|
||||||
|
3: (0.60, 0.70), # lower creativity, wider associations
|
||||||
|
4: (0.50, 0.80), # methodical, analytical approach
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
@CrewBase
|
||||||
|
class WritingCrew:
|
||||||
|
"""Crew of three creative journalists who write blog drafts in parallel."""
|
||||||
|
|
||||||
|
agents_config = "config/agents.yaml"
|
||||||
|
tasks_config = "config/tasks.yaml"
|
||||||
|
|
||||||
|
# ---- helpers ----
|
||||||
|
|
||||||
|
@staticmethod
|
||||||
|
def _journalist_llm(index: int) -> LLM:
|
||||||
|
models = _get_agent_models()
|
||||||
|
model = models[index % len(models)]
|
||||||
|
temp, top_p = _JOURNALIST_PARAMS[index + 1]
|
||||||
|
return LLM(
|
||||||
|
model=f"ollama/{model}",
|
||||||
|
base_url=_get_ollama_url(),
|
||||||
|
temperature=temp,
|
||||||
|
top_p=top_p,
|
||||||
|
)
|
||||||
|
|
||||||
|
# ---- agents ----
|
||||||
|
|
||||||
|
@agent
|
||||||
|
def journalist_one(self) -> Agent:
|
||||||
|
return Agent(
|
||||||
|
config=self.agents_config["journalist_one"], # type: ignore[index]
|
||||||
|
llm=self._journalist_llm(0),
|
||||||
|
verbose=True,
|
||||||
|
max_iter=30,
|
||||||
|
respect_context_window=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
@agent
|
||||||
|
def journalist_two(self) -> Agent:
|
||||||
|
return Agent(
|
||||||
|
config=self.agents_config["journalist_two"], # type: ignore[index]
|
||||||
|
llm=self._journalist_llm(1),
|
||||||
|
verbose=True,
|
||||||
|
max_iter=30,
|
||||||
|
respect_context_window=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
@agent
|
||||||
|
def journalist_three(self) -> Agent:
|
||||||
|
return Agent(
|
||||||
|
config=self.agents_config["journalist_three"], # type: ignore[index]
|
||||||
|
llm=self._journalist_llm(2),
|
||||||
|
verbose=True,
|
||||||
|
max_iter=30,
|
||||||
|
respect_context_window=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
@agent
|
||||||
|
def journalist_four(self) -> Agent:
|
||||||
|
return Agent(
|
||||||
|
config=self.agents_config["journalist_four"], # type: ignore[index]
|
||||||
|
llm=self._journalist_llm(3),
|
||||||
|
verbose=True,
|
||||||
|
max_iter=30,
|
||||||
|
respect_context_window=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
# ---- tasks ----
|
||||||
|
|
||||||
|
@task
|
||||||
|
def write_draft_one(self) -> Task:
|
||||||
|
return Task(
|
||||||
|
config=self.tasks_config["write_draft_one"], # type: ignore[index]
|
||||||
|
)
|
||||||
|
|
||||||
|
@task
|
||||||
|
def write_draft_two(self) -> Task:
|
||||||
|
return Task(
|
||||||
|
config=self.tasks_config["write_draft_two"], # type: ignore[index]
|
||||||
|
)
|
||||||
|
|
||||||
|
@task
|
||||||
|
def write_draft_three(self) -> Task:
|
||||||
|
return Task(
|
||||||
|
config=self.tasks_config["write_draft_three"], # type: ignore[index]
|
||||||
|
)
|
||||||
|
|
||||||
|
@task
|
||||||
|
def write_draft_four(self) -> Task:
|
||||||
|
return Task(
|
||||||
|
config=self.tasks_config["write_draft_four"], # type: ignore[index]
|
||||||
|
)
|
||||||
|
|
||||||
|
# ---- crew ----
|
||||||
|
|
||||||
|
@crew
|
||||||
|
def crew(self) -> Crew:
|
||||||
|
return Crew(
|
||||||
|
agents=self.agents,
|
||||||
|
tasks=self.tasks,
|
||||||
|
process=Process.sequential,
|
||||||
|
verbose=True,
|
||||||
|
)
|
||||||
@@ -1,44 +1,181 @@
|
|||||||
|
"""
|
||||||
|
OllamaGenerator – public interface for blog generation.
|
||||||
|
|
||||||
|
This module preserves the same API that ``main.py`` relies on while
|
||||||
|
delegating the heavy lifting to a CrewAI Flow (``blog_flow.BlogFlow``)
|
||||||
|
that orchestrates a researcher, four journalists, and an editor via
|
||||||
|
YAML-configured crews.
|
||||||
|
|
||||||
|
Breaking changes from the previous implementation
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
* ``langchain-ollama`` is no longer required – the ``generate_system_message``
|
||||||
|
helper now talks directly to the Ollama HTTP API via the ``ollama`` client.
|
||||||
|
* Internally, blog generation is driven by CrewAI agents, crews and a Flow
|
||||||
|
rather than by hand-rolled retry loops and thread-pool executors.
|
||||||
|
|
||||||
|
Public interface (unchanged)
|
||||||
|
~~~~~~~~~~~~~~~~~~~~~~~~~~~~
|
||||||
|
* ``OllamaGenerator(title, content, inner_title)``
|
||||||
|
* ``save_to_file(filename)`` – generates the blog and writes it to disk
|
||||||
|
* ``generate_system_message(prompt_system, prompt_human)`` – simple LLM call
|
||||||
|
* ``self.response`` – the final markdown text (populated after ``save_to_file``)
|
||||||
|
"""
|
||||||
|
|
||||||
|
import json
|
||||||
import os
|
import os
|
||||||
|
import time
|
||||||
|
from concurrent.futures import ThreadPoolExecutor, TimeoutError
|
||||||
|
|
||||||
from ollama import Client
|
from ollama import Client
|
||||||
|
|
||||||
|
from ai_generators.blog_flow import BlogFlow
|
||||||
|
|
||||||
|
|
||||||
class OllamaGenerator:
|
class OllamaGenerator:
|
||||||
|
"""Generate a polished blog post from raw content using CrewAI agents.
|
||||||
|
|
||||||
def __init__(self, title: str, content: str, model: str):
|
Parameters
|
||||||
self.title = title
|
----------
|
||||||
self.content = content
|
title : str
|
||||||
ollama_url = f"{os.environ["OLLAMA_PROTOCOL"]}://{os.environ["OLLAMA_HOST"]}:{os.environ["OLLAMA_PORT"]}"
|
An OS-friendly slug used for file names and ChromaDB collection
|
||||||
self.ollama_client = Client(host=ollama_url)
|
names (e.g. ``"my_blog_title"``).
|
||||||
self.ollama_model = model
|
content : str
|
||||||
|
The raw source content that the blog should be based on.
|
||||||
def generate_markdown(self) -> str:
|
inner_title : str
|
||||||
|
The human-readable blog title (used in prompts and output).
|
||||||
prompt = f"""
|
|
||||||
You are a Software Developer and DevOps expert
|
|
||||||
who has transistioned in Developer Relations
|
|
||||||
writing a 1000 word blog for other tech enthusiast.
|
|
||||||
You like to use almost no code examples and prefer to talk
|
|
||||||
in a light comedic tone. You are also Australian
|
|
||||||
As this person write this blog as a markdown document.
|
|
||||||
The title for the blog is {self.title}.
|
|
||||||
Do not output the title in the markdown.
|
|
||||||
The basis for the content of the blog is:
|
|
||||||
{self.content}
|
|
||||||
Only output markdown DO NOT GENERATE AN EXPLANATION
|
|
||||||
"""
|
"""
|
||||||
try:
|
|
||||||
self.response = self.ollama_client.chat(model=self.ollama_model,
|
|
||||||
messages=[
|
|
||||||
{
|
|
||||||
'role': 'user',
|
|
||||||
'content': f'{prompt}',
|
|
||||||
},
|
|
||||||
])
|
|
||||||
return self.response['message']['content']
|
|
||||||
|
|
||||||
except Exception as e:
|
def __init__(
|
||||||
raise Exception(f"Failed to generate markdown: {e}")
|
self,
|
||||||
|
title: str,
|
||||||
|
content: str,
|
||||||
|
inner_title: str,
|
||||||
|
date: str | None = None,
|
||||||
|
category: str | None = None,
|
||||||
|
):
|
||||||
|
self.title = title
|
||||||
|
self.inner_title = inner_title
|
||||||
|
self.content = content
|
||||||
|
self.date = date
|
||||||
|
self.category = category
|
||||||
|
self.response: str | None = None
|
||||||
|
|
||||||
|
# ---- Ollama connection (used by generate_system_message) ----
|
||||||
|
ollama_url = (
|
||||||
|
f"{os.environ['OLLAMA_PROTOCOL']}://"
|
||||||
|
f"{os.environ['OLLAMA_HOST']}:{os.environ['OLLAMA_PORT']}"
|
||||||
|
)
|
||||||
|
self.ollama_client = Client(host=ollama_url)
|
||||||
|
self.ollama_model = os.environ["EDITOR_MODEL"]
|
||||||
|
|
||||||
|
# ---- Validate required env vars early ----
|
||||||
|
try:
|
||||||
|
_ = json.loads(os.environ["CONTENT_CREATOR_MODELS"])
|
||||||
|
except (KeyError, json.JSONDecodeError) as exc:
|
||||||
|
raise Exception(
|
||||||
|
f"CONTENT_CREATOR_MODELS env var is missing or invalid: {exc}"
|
||||||
|
)
|
||||||
|
|
||||||
|
try:
|
||||||
|
_ = int(os.environ["CHROMA_PORT"])
|
||||||
|
except (KeyError, ValueError) as exc:
|
||||||
|
raise Exception(f"CHROMA_PORT is not an integer: {exc}")
|
||||||
|
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
# Public API
|
||||||
|
# ------------------------------------------------------------------
|
||||||
|
|
||||||
def save_to_file(self, filename: str) -> None:
|
def save_to_file(self, filename: str) -> None:
|
||||||
|
"""Run the full CrewAI blog-generation flow and write the result
|
||||||
|
to *filename*.
|
||||||
|
|
||||||
|
After this call ``self.response`` contains the final markdown text.
|
||||||
|
"""
|
||||||
|
self.response = self.generate_markdown()
|
||||||
with open(filename, "w") as f:
|
with open(filename, "w") as f:
|
||||||
f.write(self.generate_markdown())
|
f.write(self.response)
|
||||||
|
|
||||||
|
def generate_markdown(self) -> str:
|
||||||
|
"""Execute the CrewAI Flow and return the final markdown document.
|
||||||
|
|
||||||
|
The Flow:
|
||||||
|
1. **Research crew** – a critical researcher with web search
|
||||||
|
investigates the topic and produces verified findings.
|
||||||
|
2. **Writing crew** – four creative journalists write draft
|
||||||
|
blog articles in parallel.
|
||||||
|
3. **Editor crew** – a critical editor loads the journalist drafts
|
||||||
|
into the vector DB, queries for relevant context, and produces
|
||||||
|
the polished final document.
|
||||||
|
"""
|
||||||
|
inputs = {
|
||||||
|
"title": self.title,
|
||||||
|
"inner_title": self.inner_title,
|
||||||
|
"content": self.content,
|
||||||
|
}
|
||||||
|
if self.date is not None:
|
||||||
|
inputs["date"] = self.date
|
||||||
|
if self.category is not None:
|
||||||
|
inputs["category"] = self.category
|
||||||
|
|
||||||
|
flow = BlogFlow()
|
||||||
|
result = flow.kickoff(inputs=inputs)
|
||||||
|
return str(result)
|
||||||
|
|
||||||
|
def generate_system_message(self, prompt_system: str, prompt_human: str) -> str:
|
||||||
|
"""Send a system/human message pair to the editor model and return
|
||||||
|
the assistant's response.
|
||||||
|
|
||||||
|
This is a lightweight helper used by ``main.py`` for generating
|
||||||
|
commit messages and notification text – it does **not** invoke the
|
||||||
|
full CrewAI Flow.
|
||||||
|
"""
|
||||||
|
|
||||||
|
def _generate() -> str:
|
||||||
|
response = self.ollama_client.chat(
|
||||||
|
model=self.ollama_model,
|
||||||
|
messages=[
|
||||||
|
{"role": "system", "content": prompt_system},
|
||||||
|
{"role": "user", "content": prompt_human},
|
||||||
|
],
|
||||||
|
)
|
||||||
|
return response["message"]["content"]
|
||||||
|
|
||||||
|
# Retry mechanism with 30-minute timeout (same as the original)
|
||||||
|
timeout_seconds = 30 * 60
|
||||||
|
max_retries = 3
|
||||||
|
|
||||||
|
for attempt in range(max_retries):
|
||||||
|
try:
|
||||||
|
with ThreadPoolExecutor(max_workers=1) as executor:
|
||||||
|
future = executor.submit(_generate)
|
||||||
|
result = future.result(timeout=timeout_seconds)
|
||||||
|
return result
|
||||||
|
except TimeoutError:
|
||||||
|
print(
|
||||||
|
f"AI call timed out after {timeout_seconds} seconds "
|
||||||
|
f"on attempt {attempt + 1}"
|
||||||
|
)
|
||||||
|
if attempt < max_retries - 1:
|
||||||
|
print("Retrying...")
|
||||||
|
time.sleep(5)
|
||||||
|
continue
|
||||||
|
else:
|
||||||
|
raise Exception(
|
||||||
|
f"AI call failed to complete after {max_retries} "
|
||||||
|
f"attempts with {timeout_seconds} second timeouts"
|
||||||
|
)
|
||||||
|
except Exception as exc:
|
||||||
|
if attempt < max_retries - 1:
|
||||||
|
print(
|
||||||
|
f"Attempt {attempt + 1} failed with error: {exc}. Retrying..."
|
||||||
|
)
|
||||||
|
time.sleep(5)
|
||||||
|
continue
|
||||||
|
else:
|
||||||
|
raise Exception(
|
||||||
|
f"Failed to generate system message after "
|
||||||
|
f"{max_retries} attempts: {exc}"
|
||||||
|
)
|
||||||
|
|
||||||
|
# Should never reach here, but satisfy type-checkers
|
||||||
|
raise RuntimeError("Unexpected exit from generate_system_message")
|
||||||
|
|||||||
@@ -0,0 +1,4 @@
|
|||||||
|
# Tools package for the blog generation CrewAI flow.
|
||||||
|
from ai_generators.tools.ollama_web_search_tool import OllamaWebSearchTool
|
||||||
|
|
||||||
|
__all__ = ["OllamaWebSearchTool"]
|
||||||
@@ -0,0 +1,124 @@
|
|||||||
|
"""
|
||||||
|
Custom CrewAI tool that wraps Ollama's native web search API.
|
||||||
|
|
||||||
|
This tool allows CrewAI agents to perform web searches using an Ollama
|
||||||
|
subscription instead of third-party services like Serper or EXA.
|
||||||
|
|
||||||
|
Requires:
|
||||||
|
- Ollama Python library: pip install ollama
|
||||||
|
- OLLAMA_API_KEY environment variable set with your Ollama API key
|
||||||
|
"""
|
||||||
|
|
||||||
|
import os
|
||||||
|
|
||||||
|
import ollama
|
||||||
|
from crewai.tools import BaseTool
|
||||||
|
from pydantic import BaseModel, Field
|
||||||
|
|
||||||
|
|
||||||
|
class OllamaWebSearchInput(BaseModel):
|
||||||
|
"""Input schema for OllamaWebSearchTool."""
|
||||||
|
|
||||||
|
query: str = Field(
|
||||||
|
...,
|
||||||
|
description="The web search query string. Be specific and include relevant keywords.",
|
||||||
|
)
|
||||||
|
max_results: int = Field(
|
||||||
|
default=5,
|
||||||
|
ge=1,
|
||||||
|
le=10,
|
||||||
|
description="Maximum number of search results to return (1-10, default 5).",
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class OllamaWebSearchTool(BaseTool):
|
||||||
|
"""
|
||||||
|
Web search tool using Ollama's native web search API.
|
||||||
|
|
||||||
|
This tool performs live web searches and returns relevant results with
|
||||||
|
titles, URLs, and content snippets. It's ideal for research tasks that
|
||||||
|
require current, up-to-date information from the internet.
|
||||||
|
|
||||||
|
The tool requires an Ollama subscription and the OLLAMA_API_KEY environment
|
||||||
|
variable to be set.
|
||||||
|
|
||||||
|
Example usage:
|
||||||
|
from ai_generators.tools.ollama_web_search_tool import OllamaWebSearchTool
|
||||||
|
|
||||||
|
researcher = Agent(
|
||||||
|
role="Researcher",
|
||||||
|
goal="Research topics thoroughly",
|
||||||
|
tools=[OllamaWebSearchTool()],
|
||||||
|
)
|
||||||
|
"""
|
||||||
|
|
||||||
|
name: str = "ollama_web_search"
|
||||||
|
description: str = (
|
||||||
|
"Search the web for current information using Ollama's web search API. "
|
||||||
|
"Use this tool when you need to find up-to-date information, verify claims, "
|
||||||
|
"find supporting or contradicting evidence, or research topics that require "
|
||||||
|
"current data. Returns search results with titles, URLs, and content snippets."
|
||||||
|
)
|
||||||
|
args_schema: type[BaseModel] = OllamaWebSearchInput
|
||||||
|
|
||||||
|
def _run(self, query: str, max_results: int = 5) -> str:
|
||||||
|
"""
|
||||||
|
Execute a web search and return formatted results.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
query: The search query string
|
||||||
|
max_results: Maximum number of results to return (1-10)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Formatted string with search results, each containing title, URL, and content
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# Ensure API key is set
|
||||||
|
if not os.environ.get("OLLAMA_API_KEY"):
|
||||||
|
return "Error: OLLAMA_API_KEY environment variable is not set. Please set your Ollama API key."
|
||||||
|
|
||||||
|
# Perform the web search
|
||||||
|
response = ollama.web_search(query=query, max_results=max_results)
|
||||||
|
|
||||||
|
# Extract and format results
|
||||||
|
results = response.get("results", [])
|
||||||
|
|
||||||
|
if not results:
|
||||||
|
return f"No search results found for query: '{query}'"
|
||||||
|
|
||||||
|
formatted_results = []
|
||||||
|
for i, result in enumerate(results, 1):
|
||||||
|
title = result.get("title", "No title")
|
||||||
|
url = result.get("url", "No URL")
|
||||||
|
content = result.get("content", "No content available")
|
||||||
|
|
||||||
|
formatted_results.append(
|
||||||
|
f"Result {i}:\nTitle: {title}\nURL: {url}\nContent: {content}\n"
|
||||||
|
)
|
||||||
|
|
||||||
|
return "\n".join(formatted_results)
|
||||||
|
|
||||||
|
except Exception as exc:
|
||||||
|
return f"Error performing web search: {exc}"
|
||||||
|
|
||||||
|
def _handle_exception(self, exc: Exception) -> str:
|
||||||
|
"""Handle exceptions gracefully and return a user-friendly error message."""
|
||||||
|
error_message = str(exc)
|
||||||
|
|
||||||
|
# Check for common error types
|
||||||
|
if "authentication" in error_message.lower() or "401" in error_message:
|
||||||
|
return (
|
||||||
|
"Authentication error: Your OLLAMA_API_KEY may be invalid or expired. "
|
||||||
|
"Please check your API key and ensure it's set correctly in the environment."
|
||||||
|
)
|
||||||
|
elif "rate limit" in error_message.lower() or "429" in error_message:
|
||||||
|
return "Rate limit exceeded: Too many search requests. Please wait a moment and try again."
|
||||||
|
elif (
|
||||||
|
"network" in error_message.lower() or "connection" in error_message.lower()
|
||||||
|
):
|
||||||
|
return (
|
||||||
|
"Network error: Unable to connect to Ollama's web search service. "
|
||||||
|
"Please check your internet connection and try again."
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
return f"Search failed: {error_message}"
|
||||||
+64
-6
@@ -1,5 +1,13 @@
|
|||||||
import ai_generators.ollama_md_generator as omg
|
import ai_generators.ollama_md_generator as omg
|
||||||
import trilium.notes as tn
|
import trilium.notes as tn
|
||||||
|
import repo_management.repo_manager as git_repo
|
||||||
|
from notifications.n8n import N8NWebhookJwt
|
||||||
|
import string,os
|
||||||
|
from datetime import datetime
|
||||||
|
from dotenv import load_dotenv
|
||||||
|
load_dotenv()
|
||||||
|
print(os.environ["CONTENT_CREATOR_MODELS"])
|
||||||
|
|
||||||
|
|
||||||
tril = tn.TrilumNotes()
|
tril = tn.TrilumNotes()
|
||||||
|
|
||||||
@@ -7,16 +15,66 @@ tril.get_new_notes()
|
|||||||
tril_notes = tril.get_notes_content()
|
tril_notes = tril.get_notes_content()
|
||||||
|
|
||||||
|
|
||||||
def convert_to_lowercase_with_underscores(string):
|
def convert_to_lowercase_with_underscores(s):
|
||||||
return string.lower().replace(" ", "_")
|
allowed = set(string.ascii_letters + string.digits + ' ')
|
||||||
|
filtered_string = ''.join(c for c in s if c in allowed)
|
||||||
|
return filtered_string.lower().replace(" ", "_")
|
||||||
|
|
||||||
|
|
||||||
for note in tril_notes:
|
for note in tril_notes:
|
||||||
print(tril_notes[note]['title'])
|
print(tril_notes[note]['title'])
|
||||||
# print(tril_notes[note]['content'])
|
# print(tril_notes[note]['content'])
|
||||||
print("Generating Document")
|
print("Generating Document")
|
||||||
ai_gen = omg.OllamaGenerator(tril_notes[note]['title'],
|
|
||||||
tril_notes[note]['content'],
|
|
||||||
"deepseek-r1:7b")
|
|
||||||
os_friendly_title = convert_to_lowercase_with_underscores(tril_notes[note]['title'])
|
os_friendly_title = convert_to_lowercase_with_underscores(tril_notes[note]['title'])
|
||||||
ai_gen.save_to_file(f"/blog_creator/generated_files/{os_friendly_title}.md")
|
ai_gen = omg.OllamaGenerator(os_friendly_title,
|
||||||
|
tril_notes[note]['content'],
|
||||||
|
tril_notes[note]['title'])
|
||||||
|
blog_path = f"generated_files/{os_friendly_title}.md"
|
||||||
|
ai_gen.save_to_file(blog_path)
|
||||||
|
|
||||||
|
|
||||||
|
# Generate commit messages and push to repo
|
||||||
|
print("Generating Commit Message")
|
||||||
|
git_sytem_prompt = "You are a blog creator commiting a piece of content to a central git repo"
|
||||||
|
git_human_prompt = f"Generate a 5 word git commit message describing {ai_gen.response}. ONLY OUTPUT THE RESPONSE"
|
||||||
|
commit_message = ai_gen.generate_system_message(git_sytem_prompt, git_human_prompt)
|
||||||
|
git_user = os.environ["GIT_USER"]
|
||||||
|
git_pass = os.environ["GIT_PASS"]
|
||||||
|
repo_manager = git_repo.GitRepository("blog/", git_user, git_pass)
|
||||||
|
print("Pushing to Repo")
|
||||||
|
repo_manager.create_copy_commit_push(blog_path, os_friendly_title, commit_message)
|
||||||
|
|
||||||
|
# Generate notification for Matrix
|
||||||
|
print("Generating Notification Message")
|
||||||
|
git_branch_url = f'https://git.aridgwayweb.com/armistace/blog/src/branch/{os_friendly_title}/src/content/{os_friendly_title}.md'
|
||||||
|
n8n_system_prompt = f"You are a blog creator notifiying the final editor of the final creation of blog available at {git_branch_url}"
|
||||||
|
n8n_prompt_human = f"""
|
||||||
|
Generate an informal 100 word
|
||||||
|
summary describing {ai_gen.response}.
|
||||||
|
Don't address it or use names. ONLY OUTPUT THE RESPONSE.
|
||||||
|
ONLY OUTPUT IN PLAINTEXT STRIP ALL MARKDOWN
|
||||||
|
"""
|
||||||
|
notification_message = ai_gen.generate_system_message(n8n_system_prompt, n8n_prompt_human)
|
||||||
|
secret_key = os.environ['N8N_SECRET']
|
||||||
|
webhook_url = os.environ['N8N_WEBHOOK_URL']
|
||||||
|
notification_string = f"""
|
||||||
|
<h2>{tril_notes[note]['title']}</h2>
|
||||||
|
<h3>Summary</h3>
|
||||||
|
<p>{notification_message}</p>
|
||||||
|
<h3>Branch</h3>
|
||||||
|
<p>{os_friendly_title}</p>
|
||||||
|
<p><a href="{git_branch_url}">Link to Branch</a></p>
|
||||||
|
"""
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"message": f"{notification_string}",
|
||||||
|
"timestamp": datetime.now().isoformat()
|
||||||
|
}
|
||||||
|
|
||||||
|
webhook_client = N8NWebhookJwt(secret_key, webhook_url)
|
||||||
|
|
||||||
|
print("Notifying")
|
||||||
|
n8n_result = webhook_client.send_webhook(payload)
|
||||||
|
|
||||||
|
print(f"N8N response: {n8n_result['status']}")
|
||||||
|
|||||||
@@ -0,0 +1,45 @@
|
|||||||
|
from datetime import datetime, timedelta
|
||||||
|
import jwt
|
||||||
|
import requests
|
||||||
|
from typing import Dict, Optional
|
||||||
|
|
||||||
|
class N8NWebhookJwt:
|
||||||
|
def __init__(self, secret_key: str, webhook_url: str):
|
||||||
|
self.secret_key = secret_key
|
||||||
|
self.webhook_url = webhook_url
|
||||||
|
self.token_expiration = datetime.now() + timedelta(hours=1)
|
||||||
|
|
||||||
|
def _generate_jwt_token(self, payload: Dict) -> str:
|
||||||
|
"""Generate JWT token with the given payload."""
|
||||||
|
# Include expiration time (optional)
|
||||||
|
payload["exp"] = self.token_expiration.timestamp()
|
||||||
|
encoded_jwt = jwt.encode(
|
||||||
|
payload,
|
||||||
|
self.secret_key,
|
||||||
|
algorithm="HS256",
|
||||||
|
)
|
||||||
|
return encoded_jwt #jwt.decode(encoded_jwt, self.secret_key, algorithms=['HS256'])
|
||||||
|
|
||||||
|
def send_webhook(self, payload: Dict) -> Dict:
|
||||||
|
"""Send a webhook request with JWT authentication."""
|
||||||
|
# Generate JWT token
|
||||||
|
token = self._generate_jwt_token(payload)
|
||||||
|
|
||||||
|
# Set headers with JWT token
|
||||||
|
headers = {
|
||||||
|
"Authorization": f"Bearer {token}",
|
||||||
|
"Content-Type": "application/json"
|
||||||
|
}
|
||||||
|
|
||||||
|
# Send POST request
|
||||||
|
response = requests.post(
|
||||||
|
self.webhook_url,
|
||||||
|
json=payload,
|
||||||
|
headers=headers
|
||||||
|
)
|
||||||
|
|
||||||
|
# Handle response
|
||||||
|
if response.status_code == 200:
|
||||||
|
return {"status": "success", "response": response.json()}
|
||||||
|
else:
|
||||||
|
return {"status": "error", "response": response.status_code, "message": response.text}
|
||||||
@@ -1,48 +0,0 @@
|
|||||||
import os
|
|
||||||
import sys
|
|
||||||
from git import Repo
|
|
||||||
|
|
||||||
# Set these variables accordingly
|
|
||||||
REPO_OWNER = "your_repo_owner"
|
|
||||||
REPO_NAME = "your_repo_name"
|
|
||||||
|
|
||||||
def clone_repo(repo_url, branch="main"):
|
|
||||||
Repo.clone_from(repo_url, ".", branch=branch)
|
|
||||||
|
|
||||||
def create_markdown_file(file_name, content):
|
|
||||||
with open(f"{file_name}.md", "w") as f:
|
|
||||||
f.write(content)
|
|
||||||
|
|
||||||
def commit_and_push(file_name, message):
|
|
||||||
repo = Repo(".")
|
|
||||||
repo.index.add([f"{file_name}.md"])
|
|
||||||
repo.index.commit(message)
|
|
||||||
repo.remote().push()
|
|
||||||
|
|
||||||
def create_new_branch(branch_name):
|
|
||||||
repo = Repo(".")
|
|
||||||
repo.create_head(branch_name).checkout()
|
|
||||||
repo.head.reference.set_tracking_url(f"https://your_git_server/{REPO_OWNER}/{REPO_NAME}.git/{branch_name}")
|
|
||||||
repo.remote().push()
|
|
||||||
|
|
||||||
if __name__ == "__main__":
|
|
||||||
if len(sys.argv) < 3:
|
|
||||||
print("Usage: python push_markdown.py <repo_url> <markdown_file_name>")
|
|
||||||
sys.exit(1)
|
|
||||||
|
|
||||||
repo_url = sys.argv[1]
|
|
||||||
file_name = sys.argv[2]
|
|
||||||
|
|
||||||
# Clone the repository
|
|
||||||
clone_repo(repo_url)
|
|
||||||
|
|
||||||
# Create a new Markdown file with content
|
|
||||||
create_markdown_file(file_name, "Hello, World!\n")
|
|
||||||
|
|
||||||
# Commit and push changes to the main branch
|
|
||||||
commit_and_push(file_name, f"Add {file_name}.md")
|
|
||||||
|
|
||||||
# Create a new branch named after the Markdown file
|
|
||||||
create_new_branch(file_name)
|
|
||||||
|
|
||||||
print(f"Successfully created '{file_name}' branch with '{file_name}.md'.")
|
|
||||||
@@ -1,35 +1,108 @@
|
|||||||
import os
|
import os
|
||||||
from git import Git
|
import shutil
|
||||||
from git.repo import BaseRepository
|
from urllib.parse import quote
|
||||||
from git.exc import InvalidGitRepositoryError
|
|
||||||
from git.remote import RemoteAction
|
|
||||||
|
|
||||||
# Set the path to your blog repo here
|
from git import Repo
|
||||||
blog_repo = "/path/to/your/blog/repo"
|
from git.exc import GitCommandError
|
||||||
|
|
||||||
# Checkout a new branch and create a new file for our blog post
|
|
||||||
branch_name = "new-post"
|
class GitRepository:
|
||||||
|
# This is designed to be transitory it will desctruvtively create the repo at repo_path
|
||||||
|
# if you have uncommited changes you can kiss them goodbye!
|
||||||
|
# Don't use the repo created by this function for dev -> its a tool!
|
||||||
|
# It is expected that when used you will add, commit, push, delete
|
||||||
|
def __init__(self, repo_path, username=None, password=None):
|
||||||
|
git_protocol = os.environ["GIT_PROTOCOL"]
|
||||||
|
git_remote = os.environ["GIT_REMOTE"]
|
||||||
|
# if username is not set we don't need parse to the url
|
||||||
|
if username == None or password == None:
|
||||||
|
remote = f"{git_protocol}://{git_remote}"
|
||||||
|
else:
|
||||||
|
# of course if it is we need to parse and escape it so that it
|
||||||
|
# can generate a url
|
||||||
|
git_user = quote(username)
|
||||||
|
git_password = quote(password)
|
||||||
|
remote = f"{git_protocol}://{git_user}:{git_password}@{git_remote}"
|
||||||
|
|
||||||
|
if os.path.exists(repo_path):
|
||||||
|
shutil.rmtree(repo_path)
|
||||||
|
self.repo_path = repo_path
|
||||||
|
print("Cloning Repo")
|
||||||
|
Repo.clone_from(remote, repo_path)
|
||||||
|
self.repo = Repo(repo_path)
|
||||||
|
self.username = username
|
||||||
|
self.password = password
|
||||||
|
|
||||||
|
def clone(self, remote_url, destination_path):
|
||||||
|
"""Clone a Git repository with authentication"""
|
||||||
try:
|
try:
|
||||||
repo = Git(blog_repo)
|
self.repo.clone(remote_url, destination_path)
|
||||||
repo.checkout("-b", branch_name, "origin/main")
|
return True
|
||||||
with open("my-blog-post.md", "w") as f:
|
except GitCommandError as e:
|
||||||
f.write(content)
|
print(f"Cloning failed: {e}")
|
||||||
except InvalidGitRepositoryError:
|
return False
|
||||||
# Handle repository errors gracefully
|
|
||||||
pass
|
|
||||||
|
|
||||||
# Add and commit the changes to Git
|
def fetch(self, remote_name="origin", ref_name="main"):
|
||||||
repo.add("my-blog-post.md")
|
"""Fetch updates from a remote repository with authentication"""
|
||||||
repo.commit("-m", "Added new blog post about DevOps best practices.")
|
|
||||||
|
|
||||||
# Push the changes to Git and create a PR
|
|
||||||
repo.remote().push("refs/heads/{0}:refs/for/main".format(branch_name), "--set-upstream")
|
|
||||||
base_branch = "origin/main"
|
|
||||||
target_branch = "main"
|
|
||||||
pr_title = "DevOps best practices"
|
|
||||||
try:
|
try:
|
||||||
repo.create_head("{0}-{1}", base=base_branch, message="{}".format(pr_title))
|
self.repo.remotes[remote_name].fetch(ref_name=ref_name)
|
||||||
except RemoteAction.GitExitStatus as e:
|
return True
|
||||||
# Handle Git exit status errors gracefully
|
except GitCommandError as e:
|
||||||
pass
|
print(f"Fetching failed: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def pull(self, remote_name="origin", ref_name="main"):
|
||||||
|
"""Pull updates from a remote repository with authentication"""
|
||||||
|
print("Pulling Latest Updates (if any)")
|
||||||
|
try:
|
||||||
|
self.repo.remotes[remote_name].pull(ref_name)
|
||||||
|
return True
|
||||||
|
except GitCommandError as e:
|
||||||
|
print(f"Pulling failed: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def get_branches(self):
|
||||||
|
"""List all branches in the repository"""
|
||||||
|
return [branch.name for branch in self.repo.branches]
|
||||||
|
|
||||||
|
def add_and_commit(self, message=None):
|
||||||
|
"""Add and commit changes to the repository."""
|
||||||
|
try:
|
||||||
|
print("Commiting latest draft")
|
||||||
|
# Add all changes
|
||||||
|
self.repo.git.add(all=True)
|
||||||
|
# Commit with the provided message or a default
|
||||||
|
if message is None:
|
||||||
|
commit_message = "Added and committed new content"
|
||||||
|
else:
|
||||||
|
commit_message = message
|
||||||
|
self.repo.git.commit(message=commit_message)
|
||||||
|
return True
|
||||||
|
except GitCommandError as e:
|
||||||
|
print(f"Commit failed: {e}")
|
||||||
|
return False
|
||||||
|
|
||||||
|
def create_copy_commit_push(self, file_path, title, commit_message):
|
||||||
|
# Check if branch exists remotely
|
||||||
|
remote_branches = [
|
||||||
|
ref.name.split("/")[-1] for ref in self.repo.remotes.origin.refs
|
||||||
|
]
|
||||||
|
|
||||||
|
if title in remote_branches:
|
||||||
|
# Branch exists remotely, checkout and pull
|
||||||
|
self.repo.git.checkout(title)
|
||||||
|
self.pull(ref_name=title)
|
||||||
|
else:
|
||||||
|
# New branch, create from main
|
||||||
|
self.repo.git.checkout("-b", title, "origin/main")
|
||||||
|
|
||||||
|
# Ensure destination directory exists
|
||||||
|
dest_dir = f"{self.repo_path}src/content/"
|
||||||
|
os.makedirs(dest_dir, exist_ok=True)
|
||||||
|
|
||||||
|
# Copy file
|
||||||
|
shutil.copy(f"{file_path}", dest_dir)
|
||||||
|
|
||||||
|
# Commit and push
|
||||||
|
self.add_and_commit(commit_message)
|
||||||
|
self.repo.git.push("--set-upstream", "origin", title)
|
||||||
|
|||||||
@@ -18,9 +18,13 @@ class TrilumNotes:
|
|||||||
print("Please run get_token and set your token")
|
print("Please run get_token and set your token")
|
||||||
else:
|
else:
|
||||||
self.ea = ETAPI(self.server_url, self.token)
|
self.ea = ETAPI(self.server_url, self.token)
|
||||||
|
self.new_notes = None
|
||||||
|
self.note_content = None
|
||||||
|
|
||||||
def get_token(self):
|
def get_token(self):
|
||||||
ea = ETAPI(self.server_url)
|
ea = ETAPI(self.server_url)
|
||||||
|
if self.tril_pass == None:
|
||||||
|
raise ValueError("Trillium password can not be none")
|
||||||
token = ea.login(self.tril_pass)
|
token = ea.login(self.tril_pass)
|
||||||
print(token)
|
print(token)
|
||||||
print("I would recomend you update the env file with this tootsweet!")
|
print("I would recomend you update the env file with this tootsweet!")
|
||||||
@@ -40,10 +44,11 @@ class TrilumNotes:
|
|||||||
|
|
||||||
def get_notes_content(self):
|
def get_notes_content(self):
|
||||||
content_dict = {}
|
content_dict = {}
|
||||||
|
if self.new_notes is None:
|
||||||
|
raise ValueError("How did you do this? new_notes is None!")
|
||||||
for note in self.new_notes['results']:
|
for note in self.new_notes['results']:
|
||||||
content_dict[note['noteId']] = {"title" : f"{note['title']}",
|
content_dict[note['noteId']] = {"title" : f"{note['title']}",
|
||||||
"content" : f"{self._get_content(note['noteId'])}"
|
"content" : f"{self._get_content(note['noteId'])}"
|
||||||
}
|
}
|
||||||
self.note_content = content_dict
|
self.note_content = content_dict
|
||||||
return content_dict
|
return content_dict
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user