commit 86fb6a763071a1c888065cc54e7d9d6e33221146 Author: Victor Vargas Date: Sun Jun 28 16:13:21 2026 -0700 chore: initial scaffold with design docs diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..c3baf73 --- /dev/null +++ b/.gitignore @@ -0,0 +1,22 @@ +# Go +*.exe +*.test +*.out +*.prof +vendor/ +coverage.out +coverage.html + +# ChromaDB +chroma/ +*.db +*.db-shm +*.db-wal + +# Editor / OS +.vscode/ +.idea/ +*.swp +*.swo +*~ +.DS_Store \ No newline at end of file diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..0d368bf --- /dev/null +++ b/LICENSE @@ -0,0 +1,21 @@ +MIT License + +Copyright (c) 2026 Victor Hugo Vargas Servín + +Permission is hereby granted, free of charge, to any person obtaining a copy +of this software and associated documentation files (the "Software"), to deal +in the Software without restriction, including without limitation the rights +to use, copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the Software, and to permit persons to whom the Software is +furnished to do so, subject to the following conditions: + +The above copyright notice and this permission notice shall be included in all +copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, +FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE +AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER +LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, +OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE +SOFTWARE. \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..4a8467b --- /dev/null +++ b/README.md @@ -0,0 +1,106 @@ +# Chat-Bot — Portfolio Bot HTTP + +> 🤖 **Chatbot HTTP que presenta tu portfolio y responde preguntas sobre tus proyectos.** + +Este es un chatbot basado en [`go-llm-agent`](https://github.com/VictorVargas/go-llm-agent) que se integra con un sitio Astro/React para responder preguntas sobre Victor Hugo Vargas y sus proyectos, usando **RAG sobre archivos markdown**. + +## ✨ Features + +- 🌐 **HTTP server** con streaming SSE (Server-Sent Events) +- 🧠 **RAG sobre markdown** — indexa automáticamente los `.md` en `data/projects/` +- 🎭 **Persona customizable** — responde como "asistente de Victor" +- ⚡ **Self-hosted** con Ollama o llama.cpp (no requiere API key de cloud) +- 🔌 **Integrable** con Astro/React via proxy HTTP +- 🛡️ **Rate limiting** y logging estructurado +- 📦 **Portable** — se puede adaptar a otros contextos (clientes, productos, etc.) + +## 🚀 Quick start + +```bash +# 1. Instalar +git clone https://github.com/VictorVargas/chat-bot.git +cd chat-bot + +# 2. Resolver dependencias (crea go.sum con hashes) +go mod tidy + +# 3. Configurar provider (ejemplo: Ollama) +# Asegúrate de tener Ollama corriendo: ollama serve +# Modelo descargado: ollama pull qwen2.5:1.5b + +# 4. Cargar tus proyectos en data/projects/ +echo "# Mi Proyecto Cool\nDescripción..." > data/projects/mi-proyecto.md + +# 5. Build +go build -o bin/chat-bot ./cmd/chat-bot + +# 6. Run +./bin/chat-bot serve +# → Sirve en http://localhost:7331 +``` + +## 📁 Estructura + +``` +chat-bot/ +├── cmd/chat-bot/ # Entry point (CLI) +├── internal/ +│ ├── server/ # HTTP handlers + SSE +│ ├── portfolio/ # Data loader (markdown → RAG) +│ ├── persona/ # Persona override +│ └── streaming/ # SSE helpers +├── data/projects/ # ← TUS PROYECTOS EN MARKDOWN +│ ├── rony-tui.md +│ ├── go-llm-agent.md +│ └── ... +├── configs/ +│ └── portfolio-bot.yaml # Provider config +├── docs/ +│ └── architecture.md # ← Especificación técnica completa +└── go.mod # require go-llm-agent +``` + +## 🎯 Uso desde Astro + +Ver [`docs/architecture.md`](./docs/architecture.md) §5 — patrón recomendado de proxy. + +```typescript +// portfolio/src/pages/api/chat.ts +export const POST: APIRoute = async ({ request }) => { + const body = await request.json(); + const resp = await fetch('http://localhost:7331/api/chat', { + method: 'POST', + body: JSON.stringify(body), + }); + return new Response(resp.body, { + headers: { 'Content-Type': 'text/event-stream' }, + }); +}; +``` + +## 🔄 Adaptar a otro cliente + +Este bot está diseñado para ser **atómico** y reusable. Para adaptarlo (ej. chatbot para un concesionario): + +1. Fork/clone este repo +2. Reemplaza `data/projects/` con `data/inventory/` (u otro dominio) +3. Actualiza `configs/portfolio-bot.yaml` con la nueva persona +4. Deploy + +La librería `go-llm-agent` no cambia. + +## 📚 Documentación + +- [**Architecture doc**](./docs/architecture.md) — Especificación técnica completa +- [Library: `go-llm-agent`](https://github.com/VictorVargas/go-llm-agent) — Core reutilizable +- [Harness](https://github.com/VictorVargas/harness) — El otro proyecto que usa la misma librería + +## 📄 Licencia + +MIT — ver [`LICENSE`](./LICENSE). + +## 🔗 Proyectos del workspace + +- [`go-llm-agent`](https://github.com/VictorVargas/go-llm-agent) — Librería core +- [`harness`](https://github.com/VictorVargas/harness) — AI agent harness (TUI) +- [`portfolio`](https://github.com/VictorVargas/portfolio) — Astro + React site (integra este bot) \ No newline at end of file diff --git a/configs/portfolio-bot.yaml b/configs/portfolio-bot.yaml new file mode 100644 index 0000000..319d001 --- /dev/null +++ b/configs/portfolio-bot.yaml @@ -0,0 +1,82 @@ +# Configuración del Portfolio Bot +# Documentación: https://github.com/VictorVargas/go-llm-agent/pkg/llm + +server: + host: "0.0.0.0" + port: 7331 + read_timeout_ms: 30000 + cors_origins: + - "http://localhost:4321" # Astro dev server + - "https://victorvargas.dev" # Producción (cuando exista) + rate_limit: + requests_per_minute: 30 # Por IP + burst: 5 + +# Providers LLM (al menos uno configurado) +providers: + # === Ollama (recomendado para desarrollo) === + - name: ollama-local + type: ollama + model: qwen2.5:1.5b # Modelo pequeño para Q&A + endpoint: http://localhost:11434 + default: true + + # === llama.cpp directo (GGUF) === + - name: llamacpp-local + type: llamacpp + model_path: ${RONY_MODELS_PATH}/qwen2.5-1.5b-instruct-q5_k_m.gguf + context_size: 4096 + n_gpu_layers: 999 + + # === Anthropic (si quieres calidad > privacidad) === + - name: anthropic-api + type: anthropic + model: claude-haiku-4 # Modelo barato + api_key_env: ANTHROPIC_API_KEY + +# RAG: cómo se indexan los proyectos +rag: + enabled: true + data_path: ./data/projects # Directorio con .md + chunk_size: 500 # caracteres por chunk + chunk_overlap: 50 + embedding_provider: ollama # o llamacpp + embedding_model: nomic-embed-text + vector_db_path: ./chroma # Persistencia local + top_k: 5 # Documentos a recuperar por query + rerank: false # Phase 2 + +# Persona: quién es el bot +persona: + name: "Asistente de Victor Hugo Vargas" + tone: "Profesional, conocedor, amable" + language: "Español" + constraints: + - "Solo responder sobre Victor y sus proyectos" + - "Si no sabes, decir 'No tengo esa información'" + - "Ser conciso pero informativo" + - "Usar formato markdown para listas y código" + intro: "¡Hola! Soy el asistente virtual de Victor Hugo Vargas. Pregúntame sobre sus proyectos, skills o experiencia." + +# System prompt base (concatenado con el contenido RAG) +system_prompt: | + Eres el asistente virtual de Victor Hugo Vargas, un ingeniero de software mexicano. + + Tu trabajo es responder preguntas sobre: + - Los proyectos de Victor (ver archivos en data/projects/) + - Su experiencia y skills técnicas + - Su enfoque de trabajo + + Responde en español, con tono profesional pero accesible. + Si te preguntan algo que no está en tu contexto, dilo honestamente. + + Formato recomendado: + - Usa markdown para listas, código, y énfasis + - Sé conciso (máximo 2-3 párrafos por respuesta) + - Incluye links a repos cuando sea relevante + +# Logging +logging: + level: info # debug | info | warn | error + format: json # json | text + output: stderr \ No newline at end of file diff --git a/data/projects/README.md b/data/projects/README.md new file mode 100644 index 0000000..82ec3ab --- /dev/null +++ b/data/projects/README.md @@ -0,0 +1,49 @@ +# Proyectos del Portfolio + +Coloca aquí un archivo `.md` por cada proyecto que quieras que el bot pueda responder. + +## Convención de nombres + +- Un archivo por proyecto: `nombre-del-proyecto.md` +- Nombre en kebab-case (minúsculas con guiones) +- Ejemplo: `rony-tui.md`, `go-llm-agent.md`, `portfolio-astro.md` + +## Frontmatter (opcional pero recomendado) + +```markdown +--- +title: "Rony TUI" +date: 2026-06 +status: "active" # active | archived | wip +tags: ["go", "ai", "cli"] +repo: "https://github.com/VictorVargas/harness" +demo: "https://..." # opcional +--- + +# Rony TUI + +AI agent harness para desarrollo de software... +``` + +## Cómo se procesan + +1. El bot escanea este directorio al arrancar +2. Cada `.md` se divide en chunks de ~500 caracteres +3. Cada chunk se convierte a embedding con Ollama +4. Los embeddings se guardan en ChromaDB +5. Cuando alguien pregunta, se buscan los top-5 chunks más relevantes +6. Esos chunks se inyectan al contexto del LLM + +## Re-indexar + +Si modificas los `.md`, ejecuta: + +```bash +./bin/chat-bot reindex +``` + +Esto reconstruye ChromaDB desde cero. + +## Ejemplo de proyecto + +Ver [`example-project.md`](./example-project.md) para una plantilla. \ No newline at end of file diff --git a/data/projects/example-project.md b/data/projects/example-project.md new file mode 100644 index 0000000..0759a38 --- /dev/null +++ b/data/projects/example-project.md @@ -0,0 +1,39 @@ +--- +title: "Proyecto de Ejemplo" +date: 2026-06 +status: "active" +tags: ["ejemplo", "plantilla"] +repo: "" +demo: "" +--- + +# Proyecto de Ejemplo + +Este es un template. Reemplaza con la información real de tu proyecto. + +## Descripción + +Qué hace el proyecto, en 1-2 párrafos. Evita jerga innecesaria. + +## Stack técnico + +- **Lenguaje:** Go 1.26 +- **Framework:** Ninguno (stdlib) +- **Base de datos:** SQLite +- **Deployment:** Fly.io + +## Features principales + +1. Feature uno — descripción breve +2. Feature dos — descripción breve +3. Feature tres — descripción breve + +## Aprendizajes + +Qué aprendiste, qué challenges tuviste, qué harías diferente. + +## Links + +- Repo: github.com/VictorVargas/proyecto +- Demo: proyecto.example.com +- Docs: docs.proyecto.example.com \ No newline at end of file diff --git a/docs/architecture.md b/docs/architecture.md new file mode 100644 index 0000000..4b19878 --- /dev/null +++ b/docs/architecture.md @@ -0,0 +1,1039 @@ +# 📋 Chat-Bot — Technical Design Document + +**Versión:** 1.0 +**Autor:** Victor Hugo Vargas +**Fecha:** 2026-06-28 +**Estado:** Especificación completa para implementación +**Path:** `chat-bot/docs/architecture.md` + +> 📚 **Workspace:** Este proyecto es parte del workspace `Rony/`. Ver [`../README.md`](../../README.md). +> +> 🔑 **Depende de:** [`go-llm-agent`](https://github.com/VictorVargas/go-llm-agent) — librería core que provee agent loop, LLM clients, RAG, persona system. + +--- + +## 🎯 1. Visión del Proyecto + +### 1.1 ¿Qué es Chat-Bot? + +Un **chatbot HTTP** que responde preguntas sobre Victor Hugo Vargas y sus proyectos. Usa **RAG (Retrieval-Augmented Generation)** sobre archivos markdown que describen cada proyecto, y un LLM local (o cloud) para generar respuestas. + +### 1.2 Caso de uso primario + +Victor tiene un portfolio web (Astro + React). En el sitio hay un widget de chat donde visitantes pueden preguntar: +- "¿Qué proyectos ha hecho Victor?" +- "¿Cuál es su experiencia con Go?" +- "¿Cómo funciona Rony TUI?" +- "¿Victor ha trabajado con PostgreSQL?" + +El bot responde con información precisa extraída de los archivos markdown de proyectos + bio + skills. + +### 1.3 Casos de uso secundarios (futuro) + +- **Adaptación a clientes:** El mismo bot, con otra data y otra persona, sirve para concesionarios, restaurantes, etc. +- **Standalone CLI:** `./chat-bot ask "¿qué sabes de X?"` para uso desde terminal. +- **Slack/Discord bot:** Wrapper que consume el HTTP API. + +### 1.4 Filosofía + +- **Self-hosted por defecto** — funciona 100% local con Ollama + modelos 1-3B +- **Cloud opcional** — si se necesita más calidad, swap a Anthropic API +- **Portable** — fácil de fork/customizar para otros contextos +- **Streaming** — respuestas token-por-token con SSE (no espera a respuesta completa) +- **Reutiliza `go-llm-agent`** — no reinventar el agent loop + +--- + +## 🏗️ 2. Arquitectura + +### 2.1 Vista general + +``` +┌─────────────────────────────────────────────────────────────────┐ +│ Browser (Astro site) │ +│ ↓ HTTP POST /api/chat │ +│ Astro SSR (proxy) ←────────── Sirve portfolio + proxy chat │ +│ ↓ HTTP POST /api/chat │ +│ Chat-Bot HTTP server (:7331) │ +│ ↓ │ +│ Agent loop (go-llm-agent) │ +│ ↓ │ +│ RAG retrieval → ChromaDB sobre data/projects/*.md │ +│ ↓ │ +│ LLM (Ollama local / Anthropic cloud) │ +└─────────────────────────────────────────────────────────────────┘ +``` + +### 2.2 Componentes principales + +| Componente | Path | Responsabilidad | +|---|---|---| +| **HTTP server** | `internal/server/` | Gin/chi handlers, SSE streaming | +| **Agent runner** | `internal/agent/` | Wrapper sobre `go-llm-agent` con config específica | +| **Portfolio loader** | `internal/portfolio/` | Lee `data/projects/*.md`, indexa en ChromaDB | +| **Persona** | `internal/persona/` | Carga persona desde `configs/portfolio-bot.yaml` | +| **CLI** | `cmd/chat-bot/` | Comandos: `serve`, `reindex`, `ask`, `version` | + +### 2.3 Stack tecnológico + +| Capa | Tecnología | Razón | +|---|---|---| +| **Lenguaje** | Go 1.26+ | Mismo que `harness`, aprovechar `os.Root`, `iter.Seq` | +| **HTTP router** | `net/http` + `chi` | Stdlib + chi para middleware (CORS, logging) | +| **SSE** | `net/http` Flusher | Stdlib es suficiente, no necesita librería externa | +| **Config** | `gopkg.in/yaml.v3` | Mismo que harness | +| **RAG backend** | ChromaDB embedded via `chroma-go` | Self-hosted, simple API | +| **Embeddings** | Ollama (nomic-embed-text) | Local, gratis, buena calidad | +| **LLM** | Ollama (qwen2.5:1.5b) o llama.cpp | Self-hosted por defecto | +| **Tests** | stdlib + testify | Consistencia con el resto | + +--- + +## 🔌 3. HTTP API + +### 3.1 Endpoints + +#### `POST /api/chat` — Chat con streaming SSE + +**Request:** +```json +{ + "messages": [ + {"role": "user", "content": "¿Qué proyectos tiene Victor?"} + ], + "stream": true +} +``` + +**Response (SSE):** +``` +data: {"type":"start","conversation_id":"abc123"} + +data: {"type":"chunk","content":"Victor"} +data: {"type":"chunk","content":" tiene"} +data: {"type":"chunk","content":" varios"} +data: {"type":"chunk","content":" proyectos"} + +data: {"type":"sources","documents":["rony-tui.md","go-llm-agent.md"]} + +data: {"type":"done","usage":{"input_tokens":245,"output_tokens":38}} +``` + +**Sin streaming** (`"stream": false`): +```json +{ + "content": "Victor tiene varios proyectos...", + "sources": ["rony-tui.md", "go-llm-agent.md"], + "usage": {"input_tokens": 245, "output_tokens": 38} +} +``` + +#### `POST /api/reindex` — Re-indexar portfolio + +Útil cuando se modifican archivos en `data/projects/`. + +**Request:** vacío +**Response:** +```json +{ + "indexed_files": 12, + "total_chunks": 87, + "duration_ms": 4321 +} +``` + +#### `GET /api/health` — Health check + +```json +{ + "status": "ok", + "version": "1.0.0", + "providers": ["ollama-local"], + "rag": { + "documents": 12, + "chunks": 87, + "last_index": "2026-06-28T10:23:45Z" + } +} +``` + +#### `GET /api/info` — Metadata del bot + +```json +{ + "name": "Asistente de Victor Hugo Vargas", + "model": "qwen2.5:1.5b", + "persona": "...", + "topics": ["proyectos", "experiencia", "skills técnicas"] +} +``` + +### 3.2 SSE Implementation + +```go +// internal/server/chat.go +package server + +import ( + "encoding/json" + "fmt" + "net/http" + "github.com/VictorVargas/go-llm-agent/pkg/agent" +) + +func (s *Server) handleChat(w http.ResponseWriter, r *http.Request) { + // Headers SSE + w.Header().Set("Content-Type", "text/event-stream") + w.Header().Set("Cache-Control", "no-cache") + w.Header().Set("Connection", "keep-alive") + w.Header().Set("X-Accel-Buffering", "no") + + flusher, ok := w.(http.Flusher) + if !ok { + http.Error(w, "SSE no soportado", http.StatusInternalServerError) + return + } + + // Parse request + var req ChatRequest + if err := json.NewDecoder(r.Body).Decode(&req); err != nil { + writeError(w, flusher, "invalid request", err) + return + } + + // Start event + writeSSE(w, flusher, "start", map[string]string{ + "conversation_id": generateConvID(), + }) + + // Run agent con streaming + sources := []string{} + for chunk, err := range s.agent.RunStream(r.Context(), req.Messages) { + if err != nil { + writeSSE(w, flusher, "error", map[string]string{"message": err.Error()}) + return + } + if chunk.Type == "source" { + sources = append(sources, chunk.Source) + } + writeSSE(w, flusher, chunk.Type, chunk.Data) + } + + // Done event + writeSSE(w, flusher, "done", map[string]any{ + "usage": map[string]int{ + "input_tokens": 245, + "output_tokens": 38, + }, + }) +} + +func writeSSE(w http.ResponseWriter, flusher http.Flusher, eventType string, data any) { + payload, _ := json.Marshal(data) + fmt.Fprintf(w, "data: {\"type\":%q,\"data\":%s}\n\n", eventType, payload) + flusher.Flush() +} +``` + +### 3.3 Middleware + +```go +// internal/server/middleware.go +package server + +func (s *Server) loggingMiddleware(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + start := time.Now() + // Wrap response writer para capturar status + rw := &statusRecorder{ResponseWriter: w, status: 200} + next.ServeHTTP(rw, r) + + slog.Info("http.request", + "method", r.Method, + "path", r.URL.Path, + "status", rw.status, + "duration_ms", time.Since(start).Milliseconds(), + "ip", r.RemoteAddr, + ) + }) +} + +func (s *Server) corsMiddleware(next http.Handler) http.Handler { + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + origin := r.Header.Get("Origin") + for _, allowed := range s.config.Server.CORSOrigins { + if origin == allowed { + w.Header().Set("Access-Control-Allow-Origin", origin) + w.Header().Set("Access-Control-Allow-Methods", "POST, GET, OPTIONS") + w.Header().Set("Access-Control-Allow-Headers", "Content-Type") + break + } + } + if r.Method == "OPTIONS" { + w.WriteHeader(204) + return + } + next.ServeHTTP(w, r) + }) +} + +func (s *Server) rateLimitMiddleware(next http.Handler) http.Handler { + limiter := rate.NewLimiter(rate.Every(time.Minute/time.Duration(s.config.Server.RateLimit.RequestsPerMinute)), s.config.Server.RateLimit.Burst) + return http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + if !limiter.Allow() { + http.Error(w, "rate limit exceeded", http.StatusTooManyRequests) + return + } + next.ServeHTTP(w, r) + }) +} +``` + +--- + +## 🧠 4. RAG (Retrieval-Augmented Generation) + +### 4.1 Pipeline de indexación + +``` +data/projects/*.md + ↓ (read all files) +Raw markdown content + ↓ (split into chunks, ~500 chars, 50 overlap) +Chunks [] + ↓ (embed each chunk via Ollama nomic-embed-text) +Vectors [][]float32 + ↓ (store in ChromaDB collection "portfolio") +Indexed corpus +``` + +**Cuándo se ejecuta:** +- Al arrancar el bot (si `--reindex-on-start` flag) +- Manualmente: `./chat-bot reindex` +- Vía HTTP: `POST /api/reindex` + +### 4.2 Pipeline de retrieval + +``` +User query "¿qué proyectos tiene Victor?" + ↓ (embed query) +Query vector + ↓ (cosine similarity search en ChromaDB, top_k=5) +Top 5 chunks relevantes + ↓ (format as context block) +System prompt += chunks relevantes + ↓ (send to LLM) +LLM generates answer +``` + +### 4.3 Implementación + +```go +// internal/portfolio/indexer.go +package portfolio + +import ( + "context" + "os" + "path/filepath" + "strings" + "github.com/VictorVargas/go-llm-agent/pkg/rag" +) + +type Indexer struct { + dataPath string + memory rag.Memory + embedder rag.Embedder + chunkSize int + chunkOverlap int +} + +func (i *Indexer) IndexAll(ctx context.Context) (int, error) { + files, err := filepath.Glob(filepath.Join(i.dataPath, "*.md")) + if err != nil { + return 0, err + } + + totalChunks := 0 + for _, file := range files { + chunks, err := i.indexFile(ctx, file) + if err != nil { + slog.Warn("failed to index file", "file", file, "err", err) + continue + } + totalChunks += chunks + } + + return totalChunks, nil +} + +func (i *Indexer) indexFile(ctx context.Context, path string) (int, error) { + content, err := os.ReadFile(path) + if err != nil { + return 0, err + } + + projectID := strings.TrimSuffix(filepath.Base(path), ".md") + chunks := splitIntoChunks(string(content), i.chunkSize, i.chunkOverlap) + + for idx, chunk := range chunks { + embedding, err := i.embedder.Embed(ctx, chunk) + if err != nil { + return idx, err + } + + fragment := rag.Fragment{ + ID: fmt.Sprintf("%s-chunk-%d", projectID, idx), + Content: chunk, + Vector: embedding, + ProjectID: projectID, + Metadata: map[string]string{ + "source_file": path, + "chunk_index": fmt.Sprint(idx), + }, + } + + if err := i.memory.Add(ctx, fragment); err != nil { + return idx, err + } + } + + return len(chunks), nil +} + +func splitIntoChunks(text string, size, overlap int) []string { + // Implementación simple: split por tamaño con overlap + // Versión production usa tokenizer-aware chunking + var chunks []string + for i := 0; i < len(text); i += size - overlap { + end := i + size + if end > len(text) { + end = len(text) + } + chunks = append(chunks, text[i:end]) + } + return chunks +} +``` + +### 4.4 Retrieval en el agent loop + +```go +// internal/agent/runner.go +package agent + +func (r *Runner) buildSystemPrompt(ctx context.Context, query string) (string, error) { + // 1. Base persona prompt + basePrompt := r.persona.SystemPrompt + + // 2. Retrieve relevant chunks + fragments, err := r.memory.Search(ctx, query, r.config.RAG.TopK) + if err != nil { + return "", err + } + + // 3. Format as context + var contextBlock strings.Builder + contextBlock.WriteString(basePrompt) + contextBlock.WriteString("\n\n## Contexto relevante\n\n") + for idx, frag := range fragments { + contextBlock.WriteString(fmt.Sprintf("### Fuente: %s\n%s\n\n", + frag.Metadata["source_file"], frag.Content)) + } + + return contextBlock.String(), nil +} + +func (r *Runner) RunStream(ctx context.Context, messages []llm.Message) iter.Seq2[Chunk, error] { + return func(yield func(Chunk, error) bool) { + // Build prompt with RAG context + lastUserMsg := getLastUserMessage(messages) + systemPrompt, err := r.buildSystemPrompt(ctx, lastUserMsg) + if err != nil { + yield(Chunk{}, err) + return + } + + // Inject system prompt + messages = prependSystem(messages, systemPrompt) + + // Run agent loop + for chunk, err := range r.loop.RunStream(ctx, messages) { + if !yield(chunk, err) { + return + } + } + } +} +``` + +--- + +## 🌐 5. Integración con Astro (Portfolio) + +### 5.1 Patrón recomendado: Astro proxy + +``` +[Browser] ←→ [Astro SSR :4321] ←→ [Chat-Bot :7331] +``` + +**Por qué proxy y no llamada directa del browser al chat-bot:** +- ✅ Single domain (no CORS) +- ✅ Astro maneja auth/sesión si se necesita +- ✅ Puede haber rate limiting centralizado en Astro +- ✅ El chat-bot queda en red privada (no expuesto a internet directamente) + +### 5.2 Astro: API route del proxy + +```typescript +// portfolio/src/pages/api/chat.ts +import type { APIRoute } from 'astro'; + +const CHAT_BOT_URL = process.env.CHAT_BOT_URL || 'http://localhost:7331'; + +export const POST: APIRoute = async ({ request }) => { + const body = await request.json(); + + const resp = await fetch(`${CHAT_BOT_URL}/api/chat`, { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify(body), + }); + + if (!resp.ok) { + return new Response('Chat bot error', { status: resp.status }); + } + + // Stream SSE de vuelta al browser + return new Response(resp.body, { + status: 200, + headers: { + 'Content-Type': 'text/event-stream', + 'Cache-Control': 'no-cache', + 'Connection': 'keep-alive', + }, + }); +}; +``` + +### 5.3 React: Componente del chat + +```tsx +// portfolio/src/components/Chat.tsx +import { useState, useRef } from 'react'; + +interface Message { + role: 'user' | 'assistant'; + content: string; +} + +export default function Chat() { + const [messages, setMessages] = useState([]); + const [input, setInput] = useState(''); + const [streaming, setStreaming] = useState(false); + const abortRef = useRef(null); + + const send = async () => { + if (!input.trim() || streaming) return; + + const userMsg: Message = { role: 'user', content: input }; + setMessages(prev => [...prev, userMsg]); + setInput(''); + setStreaming(true); + + // Placeholder para streaming + const assistantMsg: Message = { role: 'assistant', content: '' }; + setMessages(prev => [...prev, assistantMsg]); + + abortRef.current = new AbortController(); + + try { + const resp = await fetch('/api/chat', { + method: 'POST', + headers: { 'Content-Type': 'application/json' }, + body: JSON.stringify({ + messages: [...messages, userMsg], + stream: true, + }), + signal: abortRef.current.signal, + }); + + const reader = resp.body!.getReader(); + const decoder = new TextDecoder(); + let buffer = ''; + + while (true) { + const { done, value } = await reader.read(); + if (done) break; + + buffer += decoder.decode(value, { stream: true }); + const lines = buffer.split('\n\n'); + buffer = lines.pop() || ''; + + for (const line of lines) { + if (!line.startsWith('data: ')) continue; + const event = JSON.parse(line.slice(6)); + + if (event.type === 'chunk') { + setMessages(prev => { + const updated = [...prev]; + updated[updated.length - 1].content += event.data.content; + return updated; + }); + } + } + } + } catch (err) { + if ((err as Error).name !== 'AbortError') { + console.error(err); + } + } finally { + setStreaming(false); + abortRef.current = null; + } + }; + + const stop = () => abortRef.current?.abort(); + + return ( +
+
+ {messages.map((m, i) => ( +
+ {m.content || (streaming && i === messages.length - 1 ? '...' : '')} +
+ ))} +
+
+ setInput(e.target.value)} + onKeyDown={e => e.key === 'Enter' && send()} + placeholder="Pregunta sobre Victor..." + disabled={streaming} + /> + {streaming ? ( + + ) : ( + + )} +
+
+ ); +} +``` + +--- + +## 🤖 6. Self-hosting con Ollama + +### 6.1 Setup + +```bash +# 1. Instalar Ollama +curl -fsSL https://ollama.com/install.sh | sh + +# 2. Descargar modelo de chat +ollama pull qwen2.5:1.5b + +# 3. Descargar modelo de embeddings +ollama pull nomic-embed-text + +# 4. Verificar +ollama list +``` + +### 6.2 Configuración por defecto + +`configs/portfolio-bot.yaml` ya viene con Ollama como default. Solo necesitas: + +```bash +# Asegurar que Ollama está corriendo +ollama serve + +# Arrancar el bot +./bin/chat-bot serve +``` + +### 6.3 Alternativa: llama.cpp directo + +Para más control o si Ollama no funciona en tu setup: + +```yaml +providers: + - name: llamacpp-local + type: llamacpp + model_path: ${RONY_MODELS_PATH}/qwen2.5-1.5b-instruct-q5_k_m.gguf + context_size: 4096 + n_gpu_layers: 999 # offload todo a GPU + default: true +``` + +El adapter `llamacpp` se importa desde `go-llm-agent/pkg/llm/providers/llamacpp` y se compila contra `llama.cpp` vía CGO o binario externo. + +--- + +## 📦 7. CLI del bot + +### 7.1 Comandos + +```bash +# Arrancar servidor HTTP +chat-bot serve [--port 7331] [--host 0.0.0.0] [--reindex-on-start] + +# Re-indexar portfolio (lee data/projects/*.md → ChromaDB) +chat-bot reindex + +# Pregunta única (sin servidor, útil para tests) +chat-bot ask "¿Qué proyectos tiene Victor?" [--no-rag] + +# Validar config +chat-bot config validate + +# Health check (útil para monitoring) +chat-bot health + +# Versión +chat-bot version +``` + +### 7.2 Implementación con Cobra + +```go +// cmd/chat-bot/main.go +package main + +import ( + "github.com/spf13/cobra" +) + +func main() { + root := &cobra.Command{ + Use: "chat-bot", + Short: "Portfolio chatbot HTTP server", + } + + root.AddCommand(serveCmd()) + root.AddCommand(reindexCmd()) + root.AddCommand(askCmd()) + root.AddCommand(configCmd()) + root.AddCommand(healthCmd()) + root.AddCommand(versionCmd()) + + if err := root.Execute(); err != nil { + os.Exit(1) + } +} + +func serveCmd() *cobra.Command { + var port int + var host string + var reindexOnStart bool + + cmd := &cobra.Command{ + Use: "serve", + Short: "Start HTTP server", + RunE: func(cmd *cobra.Command, args []string) error { + return server.Serve(server.Config{ + Port: port, + Host: host, + ReindexOnStart: reindexOnStart, + }) + }, + } + + cmd.Flags().IntVar(&port, "port", 7331, "HTTP port") + cmd.Flags().StringVar(&host, "host", "0.0.0.0", "HTTP host") + cmd.Flags().BoolVar(&reindexOnStart, "reindex-on-start", false, "Re-index RAG before serving") + + return cmd +} +``` + +--- + +## 🚀 8. Deployment + +### 8.1 Recomendación: Self-hosted en VPS + +```bash +# 1. Instalar dependencias +sudo apt install golang-go ollama +ollama pull qwen2.5:1.5b +ollama pull nomic-embed-text + +# 2. Build +go build -o /usr/local/bin/chat-bot ./cmd/chat-bot + +# 3. systemd service +cat > /etc/systemd/system/chat-bot.service <