feat(ai): add AI plans docs, LiteLLM manifests, and infrastructure updates
- Add comprehensive AI plans implementation documentation - Add LiteLLM gateway Kubernetes manifests - Update PostgreSQL and Forgejo deployment configs - Add build-from-binaries script
This commit is contained in:
parent
fce1da5b3f
commit
7902b265a9
22 changed files with 2712 additions and 21 deletions
|
|
@ -75,7 +75,7 @@ spec:
|
||||||
spec:
|
spec:
|
||||||
containers:
|
containers:
|
||||||
- name: forgejo
|
- name: forgejo
|
||||||
image: codeberg.org/forgejo/forgejo:10
|
image: registry.nxtgauge.com/forgejo:10
|
||||||
imagePullPolicy: IfNotPresent
|
imagePullPolicy: IfNotPresent
|
||||||
ports:
|
ports:
|
||||||
- containerPort: 3000
|
- containerPort: 3000
|
||||||
|
|
|
||||||
|
|
@ -16,7 +16,7 @@ spec:
|
||||||
spec:
|
spec:
|
||||||
initContainers:
|
initContainers:
|
||||||
- name: init-runner-permissions
|
- name: init-runner-permissions
|
||||||
image: busybox:1.36
|
image: registry.nxtgauge.com/busybox:1.36
|
||||||
command: ["/bin/sh", "-ec"]
|
command: ["/bin/sh", "-ec"]
|
||||||
args:
|
args:
|
||||||
- |
|
- |
|
||||||
|
|
@ -41,7 +41,7 @@ spec:
|
||||||
operator: DoesNotExist
|
operator: DoesNotExist
|
||||||
containers:
|
containers:
|
||||||
- name: dind
|
- name: dind
|
||||||
image: docker:27-dind
|
image: registry.nxtgauge.com/docker:27-dind
|
||||||
args:
|
args:
|
||||||
- --host=tcp://0.0.0.0:2375
|
- --host=tcp://0.0.0.0:2375
|
||||||
- --tls=false
|
- --tls=false
|
||||||
|
|
@ -65,7 +65,7 @@ spec:
|
||||||
cpu: 2
|
cpu: 2
|
||||||
memory: 4Gi
|
memory: 4Gi
|
||||||
- name: runner
|
- name: runner
|
||||||
image: code.forgejo.org/forgejo/runner:6
|
image: registry.nxtgauge.com/forgejo-runner:6
|
||||||
env:
|
env:
|
||||||
- name: DOCKER_HOST
|
- name: DOCKER_HOST
|
||||||
value: tcp://127.0.0.1:2375
|
value: tcp://127.0.0.1:2375
|
||||||
|
|
@ -81,7 +81,7 @@ spec:
|
||||||
fieldRef:
|
fieldRef:
|
||||||
fieldPath: spec.nodeName
|
fieldPath: spec.nodeName
|
||||||
- name: FORGEJO_RUNNER_LABELS
|
- name: FORGEJO_RUNNER_LABELS
|
||||||
value: "self-hosted:docker://ghcr.io/catthehacker/ubuntu:act-latest,linux:docker://ghcr.io/catthehacker/ubuntu:act-latest,ubuntu-latest:docker://ghcr.io/catthehacker/ubuntu:act-latest,ubuntu-22.04:docker://ghcr.io/catthehacker/ubuntu:act-latest,ubuntu-24.04:docker://ghcr.io/catthehacker/ubuntu:act-latest,debian-12:docker://ghcr.io/catthehacker/ubuntu:act-latest,docker-ready:docker://ghcr.io/catthehacker/ubuntu:act-latest"
|
value: "self-hosted:docker://registry.nxtgauge.com/catthehacker-ubuntu:act-latest,linux:docker://registry.nxtgauge.com/catthehacker-ubuntu:act-latest,ubuntu-latest:docker://registry.nxtgauge.com/catthehacker-ubuntu:act-latest,ubuntu-22.04:docker://registry.nxtgauge.com/catthehacker-ubuntu:act-latest,ubuntu-24.04:docker://registry.nxtgauge.com/catthehacker-ubuntu:act-latest,debian-12:docker://registry.nxtgauge.com/catthehacker-ubuntu:act-latest,docker-ready:docker://registry.nxtgauge.com/catthehacker-ubuntu:act-latest"
|
||||||
- name: REGISTRY_HOSTPORT
|
- name: REGISTRY_HOSTPORT
|
||||||
value: "registry.nxtgauge.com"
|
value: "registry.nxtgauge.com"
|
||||||
- name: REGISTRY_USERNAME
|
- name: REGISTRY_USERNAME
|
||||||
|
|
|
||||||
100
apps/litellm/OPENCODE_CONNECT.md
Normal file
100
apps/litellm/OPENCODE_CONNECT.md
Normal file
|
|
@ -0,0 +1,100 @@
|
||||||
|
# LiteLLM Connection Details for OpenCode
|
||||||
|
|
||||||
|
## Quick Connect
|
||||||
|
|
||||||
|
| Setting | Value |
|
||||||
|
|---------|-------|
|
||||||
|
| **Base URL** | `http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1` |
|
||||||
|
| **API Key** | `sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9` |
|
||||||
|
|
||||||
|
## Available Models
|
||||||
|
|
||||||
|
- `askash-main` - Primary model (Ollama gemma3:270m)
|
||||||
|
- `askash-fast` - Faster response variant
|
||||||
|
- `coding-main` - Optimized for code tasks
|
||||||
|
|
||||||
|
## OpenCode Configuration
|
||||||
|
|
||||||
|
### Option 1: Environment Variables
|
||||||
|
|
||||||
|
```bash
|
||||||
|
export OPENAI_API_KEY="sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9"
|
||||||
|
export OPENAI_API_BASE="http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1"
|
||||||
|
export LLM_MODEL="askash-main"
|
||||||
|
```
|
||||||
|
|
||||||
|
### Option 2: Config File
|
||||||
|
|
||||||
|
Create `~/.config/opencode/config.json`:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"provider": "openai",
|
||||||
|
"apiKey": "sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9",
|
||||||
|
"baseUrl": "http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1",
|
||||||
|
"model": "askash-main"
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Option 3: .opencode File (Project-specific)
|
||||||
|
|
||||||
|
Create `.opencode` in your project root:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"llm": {
|
||||||
|
"provider": "openai",
|
||||||
|
"apiKey": "sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9",
|
||||||
|
"baseUrl": "http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1",
|
||||||
|
"model": "askash-main"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Test Connection
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# List models
|
||||||
|
curl http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1/models \
|
||||||
|
-H "Authorization: Bearer sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9"
|
||||||
|
|
||||||
|
# Chat completion
|
||||||
|
curl http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1/chat/completions \
|
||||||
|
-H "Authorization: Bearer sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{
|
||||||
|
"model": "askash-main",
|
||||||
|
"messages": [{"role": "user", "content": "Hello"}]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
## Port-Forward (If Running Locally)
|
||||||
|
|
||||||
|
If you need to access from your local machine:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
kubectl port-forward svc/litellm 4000:4000 -n nxtgauge-ai
|
||||||
|
```
|
||||||
|
|
||||||
|
Then use: `http://localhost:4000/v1`
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
| Issue | Solution |
|
||||||
|
|-------|----------|
|
||||||
|
| Connection refused | Run `kubectl port-forward svc/litellm 4000:4000 -n nxtgauge-ai` |
|
||||||
|
| 401 Unauthorized | Check API key is correct |
|
||||||
|
| 404 Not Found | Ensure URL ends with `/v1` |
|
||||||
|
| Timeout | Model may be loading; retry |
|
||||||
|
|
||||||
|
## Verify LiteLLM is Running
|
||||||
|
|
||||||
|
```bash
|
||||||
|
kubectl get pods -n nxtgauge-ai -l app=litellm
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Last Updated**: 2026-06-14
|
||||||
|
**Namespace**: nxtgauge-ai
|
||||||
|
**Service**: litellm:4000
|
||||||
189
apps/litellm/README.md
Normal file
189
apps/litellm/README.md
Normal file
|
|
@ -0,0 +1,189 @@
|
||||||
|
# LiteLLM Setup for Nxtgauge
|
||||||
|
|
||||||
|
## Overview
|
||||||
|
LiteLLM is deployed as an API gateway to the internal Ollama service, providing OpenAI-compatible API endpoints.
|
||||||
|
|
||||||
|
## Architecture
|
||||||
|
- **Ollama**: Internal ClusterIP service at `ollama.nxtgauge-ai.svc.cluster.local:11434`
|
||||||
|
- **LiteLLM**: ClusterIP service at `10.43.7.24:4000` (internal)
|
||||||
|
- **Ingress**: Exposed at `https://llm.nxtgauge.com/v1` via Traefik
|
||||||
|
- **Security**: API key required via `LITELLM_MASTER_KEY`
|
||||||
|
|
||||||
|
## Master Key
|
||||||
|
**Production Master Key**: `sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9`
|
||||||
|
|
||||||
|
⚠️ **IMPORTANT**: Store this key securely. It grants full access to the LLM API.
|
||||||
|
|
||||||
|
## Model Aliases
|
||||||
|
All aliases route to the same Ollama model (`gemma3:270m`):
|
||||||
|
|
||||||
|
| Alias | Model | Timeout | Retries |
|
||||||
|
|-------|-------|---------|---------|
|
||||||
|
| `askash-main` | ollama/gemma3:270m | 300s | 2 |
|
||||||
|
| `askash-fast` | ollama/gemma3:270m | 120s | 1 |
|
||||||
|
| `coding-main` | ollama/gemma3:270m | 300s | 2 |
|
||||||
|
|
||||||
|
## Files Created
|
||||||
|
|
||||||
|
### Base Configuration
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/base/configmap.yaml` - LiteLLM config
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/base/secret.yaml` - Secret template
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/base/deployment.yaml` - Deployment spec
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/base/service.yaml` - Service spec
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/base/ingress.yaml` - Ingress with TLS
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/base/ratelimit.yaml` - Rate limiting middleware
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/base/kustomization.yaml` - Base kustomization
|
||||||
|
|
||||||
|
### Production Overlay
|
||||||
|
- `/home/ashwin/nxtgauge-gitops/apps/litellm/overlays/prod/kustomization.yaml` - Production overlay with secure key
|
||||||
|
|
||||||
|
## Testing Commands
|
||||||
|
|
||||||
|
### List Available Models
|
||||||
|
```bash
|
||||||
|
curl https://llm.nxtgauge.com/v1/models \
|
||||||
|
-H "Authorization: Bearer sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9"
|
||||||
|
```
|
||||||
|
|
||||||
|
### Test Chat Completion
|
||||||
|
```bash
|
||||||
|
curl https://llm.nxtgauge.com/v1/chat/completions \
|
||||||
|
-H "Authorization: Bearer sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{
|
||||||
|
"model": "askash-main",
|
||||||
|
"messages": [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": "Say hello from Ask Ash"
|
||||||
|
}
|
||||||
|
]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
### Test from Inside Cluster
|
||||||
|
```bash
|
||||||
|
kubectl run test-curl --rm -i --restart=Never --image=curlimages/curl:latest -- \
|
||||||
|
http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1/models \
|
||||||
|
-H "Authorization: Bearer sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9"
|
||||||
|
```
|
||||||
|
|
||||||
|
## OpenCode Configuration
|
||||||
|
|
||||||
|
Create/edit `~/.config/opencode/opencode.json`:
|
||||||
|
|
||||||
|
```json
|
||||||
|
{
|
||||||
|
"baseURL": "https://llm.nxtgauge.com/v1",
|
||||||
|
"apiKey": "sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9",
|
||||||
|
"models": {
|
||||||
|
"default": "askash-main",
|
||||||
|
"available": [
|
||||||
|
"askash-main",
|
||||||
|
"askash-fast",
|
||||||
|
"coding-main"
|
||||||
|
]
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Ask Ash Local Environment
|
||||||
|
|
||||||
|
Create `.env.local` in your Ask Ash project:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# LiteLLM Provider Configuration
|
||||||
|
LLM_PROVIDER=openai_compatible
|
||||||
|
OPENAI_BASE_URL=https://llm.nxtgauge.com/v1
|
||||||
|
OPENAI_API_KEY=sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9
|
||||||
|
LLM_MODEL=askash-main
|
||||||
|
AI_DEBUG=true
|
||||||
|
```
|
||||||
|
|
||||||
|
## Promptfoo Configuration
|
||||||
|
|
||||||
|
Create `promptfooconfig.yaml`:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
providers:
|
||||||
|
- id: openai
|
||||||
|
config:
|
||||||
|
apiBaseUrl: https://llm.nxtgauge.com/v1
|
||||||
|
apiKey: sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9
|
||||||
|
model: askash-main
|
||||||
|
```
|
||||||
|
|
||||||
|
## DNS Configuration (Cloudflare)
|
||||||
|
|
||||||
|
Add DNS record in Cloudflare:
|
||||||
|
- **Type**: A
|
||||||
|
- **Name**: llm
|
||||||
|
- **Content**: Your cluster external IP (check with `kubectl get svc -n kube-system traefik`)
|
||||||
|
- **TTL**: Auto
|
||||||
|
- **Proxy Status**: DNS only (for testing), then enable after TLS works
|
||||||
|
|
||||||
|
## Security Notes
|
||||||
|
|
||||||
|
1. ✅ Ollama is NOT exposed publicly (ClusterIP only)
|
||||||
|
2. ✅ LiteLLM requires API key authentication
|
||||||
|
3. ✅ TLS enabled via cert-manager/Let's Encrypt
|
||||||
|
4. ✅ Rate limiting enabled (100 req/min avg, 50 burst)
|
||||||
|
5. ⚠️ Master key is stored in Kubernetes Secret (not in Git)
|
||||||
|
|
||||||
|
## Management Commands
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Check LiteLLM status
|
||||||
|
kubectl get pods -n nxtgauge-ai -l app=litellm
|
||||||
|
|
||||||
|
# View logs
|
||||||
|
kubectl logs -n nxtgauge-ai -l app=litellm --tail=100 -f
|
||||||
|
|
||||||
|
# Restart LiteLLM
|
||||||
|
kubectl rollout restart deployment litellm -n nxtgauge-ai
|
||||||
|
|
||||||
|
# Get master key
|
||||||
|
kubectl get secret litellm-secrets -n nxtgauge-ai -o jsonpath='{.data.LITELLM_MASTER_KEY}' | base64 -d
|
||||||
|
|
||||||
|
# Edit configuration
|
||||||
|
kubectl edit configmap litellm-config -n nxtgauge-ai
|
||||||
|
|
||||||
|
# Port-forward for local testing
|
||||||
|
kubectl port-forward svc/litellm 4000:4000 -n nxtgauge-ai
|
||||||
|
```
|
||||||
|
|
||||||
|
## Troubleshooting
|
||||||
|
|
||||||
|
### Pod not starting
|
||||||
|
```bash
|
||||||
|
kubectl describe pod -n nxtgauge-ai -l app=litellm
|
||||||
|
kubectl logs -n nxtgauge-ai -l app=litellm --previous
|
||||||
|
```
|
||||||
|
|
||||||
|
### Ollama unreachable
|
||||||
|
```bash
|
||||||
|
kubectl get svc -n nxtgauge-ai ollama
|
||||||
|
kubectl exec -n nxtgauge-ai -it ollama-76fb847d46-b7pkp -- ollama list
|
||||||
|
```
|
||||||
|
|
||||||
|
### TLS not working
|
||||||
|
```bash
|
||||||
|
kubectl get certificate -n nxtgauge-ai
|
||||||
|
kubectl describe certificate -n nxtgauge-ai litellm-tls
|
||||||
|
```
|
||||||
|
|
||||||
|
## Future Enhancements
|
||||||
|
|
||||||
|
1. Add more Ollama models and configure model-specific aliases
|
||||||
|
2. Implement request/response caching with Redis
|
||||||
|
3. Add Prometheus metrics for observability
|
||||||
|
4. Configure team-based API keys for multi-user access
|
||||||
|
5. Add request logging and usage analytics
|
||||||
|
6. Implement token-based billing/quota management
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
**Deployed**: 2026-06-14
|
||||||
|
**Namespace**: nxtgauge-ai
|
||||||
|
**Service**: litellm (10.43.7.24:4000)
|
||||||
|
**Ingress**: https://llm.nxtgauge.com
|
||||||
94
apps/litellm/base/configmap.yaml
Normal file
94
apps/litellm/base/configmap.yaml
Normal file
|
|
@ -0,0 +1,94 @@
|
||||||
|
apiVersion: v1
|
||||||
|
kind: ConfigMap
|
||||||
|
metadata:
|
||||||
|
name: litellm-config
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
data:
|
||||||
|
config.yaml: |
|
||||||
|
model_list:
|
||||||
|
# === FAST MODEL (qwen3:4b) - 2.5GB ===
|
||||||
|
- model_name: askash-fast
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:4b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 60
|
||||||
|
|
||||||
|
- model_name: recommender
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:4b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 60
|
||||||
|
|
||||||
|
- model_name: messenger
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:4b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 60
|
||||||
|
|
||||||
|
- model_name: safety-check
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:4b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 60
|
||||||
|
|
||||||
|
- model_name: help-assistant
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:4b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 60
|
||||||
|
|
||||||
|
# === POWER MODEL (qwen3:8b) - 5.2GB ===
|
||||||
|
- model_name: askash-main
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:8b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 300
|
||||||
|
|
||||||
|
- model_name: jd-generator
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:8b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 300
|
||||||
|
|
||||||
|
- model_name: profile-writer
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:8b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 300
|
||||||
|
|
||||||
|
- model_name: service-writer
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:8b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 300
|
||||||
|
|
||||||
|
- model_name: requirement-writer
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:8b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 300
|
||||||
|
|
||||||
|
- model_name: support-drafter
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:8b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 300
|
||||||
|
|
||||||
|
- model_name: decision-support
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/qwen3:8b
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 300
|
||||||
|
|
||||||
|
- model_name: ultra-fast
|
||||||
|
litellm_params:
|
||||||
|
model: ollama/gemma3:270m
|
||||||
|
api_base: http://ollama.nxtgauge-ai.svc.cluster.local:11434
|
||||||
|
timeout: 30
|
||||||
|
|
||||||
|
general_settings:
|
||||||
|
master_key: os.environ/LITELLM_MASTER_KEY
|
||||||
|
|
||||||
|
router_settings:
|
||||||
|
fallback_retries: 2
|
||||||
|
timeout: 300
|
||||||
11
apps/litellm/base/db-secret.yaml
Normal file
11
apps/litellm/base/db-secret.yaml
Normal file
|
|
@ -0,0 +1,11 @@
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Secret
|
||||||
|
metadata:
|
||||||
|
name: litellm-db-credentials
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
type: Opaque
|
||||||
|
stringData:
|
||||||
|
DATABASE_URL: "postgresql://litellm:litellm123@postgres.nxtgauge-ai.svc.cluster.local:5432/litellm"
|
||||||
|
DB_USER: "litellm"
|
||||||
|
DB_PASSWORD: "litellm123"
|
||||||
|
DB_NAME: "litellm"
|
||||||
63
apps/litellm/base/deployment.yaml
Normal file
63
apps/litellm/base/deployment.yaml
Normal file
|
|
@ -0,0 +1,63 @@
|
||||||
|
apiVersion: apps/v1
|
||||||
|
kind: Deployment
|
||||||
|
metadata:
|
||||||
|
name: litellm
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
labels:
|
||||||
|
app: litellm
|
||||||
|
spec:
|
||||||
|
replicas: 1
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app: litellm
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app: litellm
|
||||||
|
spec:
|
||||||
|
containers:
|
||||||
|
- name: litellm
|
||||||
|
image: registry.nxtgauge.com/litellm:latest
|
||||||
|
command:
|
||||||
|
- "/bin/bash"
|
||||||
|
args:
|
||||||
|
- "-c"
|
||||||
|
- "cat /app/config.yaml && exec litellm --config /app/config.yaml --port 4000 --host 0.0.0.0"
|
||||||
|
ports:
|
||||||
|
- containerPort: 4000
|
||||||
|
name: http
|
||||||
|
env:
|
||||||
|
- name: LITELLM_MASTER_KEY
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: litellm-secrets
|
||||||
|
key: LITELLM_MASTER_KEY
|
||||||
|
- name: LITELLM_LOG_LEVEL
|
||||||
|
value: "DEBUG"
|
||||||
|
volumeMounts:
|
||||||
|
- name: config
|
||||||
|
mountPath: /app/config.yaml
|
||||||
|
subPath: config.yaml
|
||||||
|
readOnly: true
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
cpu: 100m
|
||||||
|
memory: 256Mi
|
||||||
|
livenessProbe:
|
||||||
|
tcpSocket:
|
||||||
|
port: 4000
|
||||||
|
initialDelaySeconds: 60
|
||||||
|
periodSeconds: 10
|
||||||
|
timeoutSeconds: 5
|
||||||
|
failureThreshold: 3
|
||||||
|
readinessProbe:
|
||||||
|
tcpSocket:
|
||||||
|
port: 4000
|
||||||
|
initialDelaySeconds: 30
|
||||||
|
periodSeconds: 5
|
||||||
|
timeoutSeconds: 3
|
||||||
|
failureThreshold: 3
|
||||||
|
volumes:
|
||||||
|
- name: config
|
||||||
|
configMap:
|
||||||
|
name: litellm-config
|
||||||
25
apps/litellm/base/ingress.yaml
Normal file
25
apps/litellm/base/ingress.yaml
Normal file
|
|
@ -0,0 +1,25 @@
|
||||||
|
apiVersion: networking.k8s.io/v1
|
||||||
|
kind: Ingress
|
||||||
|
metadata:
|
||||||
|
name: litellm
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
annotations:
|
||||||
|
traefik.ingress.kubernetes.io/router.entrypoints: web,websecure
|
||||||
|
cert-manager.io/cluster-issuer: letsencrypt-prod
|
||||||
|
spec:
|
||||||
|
ingressClassName: traefik
|
||||||
|
tls:
|
||||||
|
- hosts:
|
||||||
|
- llm.nxtgauge.com
|
||||||
|
secretName: litellm-tls
|
||||||
|
rules:
|
||||||
|
- host: llm.nxtgauge.com
|
||||||
|
http:
|
||||||
|
paths:
|
||||||
|
- path: /
|
||||||
|
pathType: Prefix
|
||||||
|
backend:
|
||||||
|
service:
|
||||||
|
name: litellm
|
||||||
|
port:
|
||||||
|
number: 4000
|
||||||
12
apps/litellm/base/kustomization.yaml
Normal file
12
apps/litellm/base/kustomization.yaml
Normal file
|
|
@ -0,0 +1,12 @@
|
||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
|
||||||
|
resources:
|
||||||
|
- configmap.yaml
|
||||||
|
- secret.yaml
|
||||||
|
- deployment.yaml
|
||||||
|
- service.yaml
|
||||||
|
- ingress.yaml
|
||||||
|
- ratelimit.yaml
|
||||||
91
apps/litellm/base/postgres.yaml
Normal file
91
apps/litellm/base/postgres.yaml
Normal file
|
|
@ -0,0 +1,91 @@
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Namespace
|
||||||
|
metadata:
|
||||||
|
name: nxtgauge-ai
|
||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: PersistentVolumeClaim
|
||||||
|
metadata:
|
||||||
|
name: postgres-pvc
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
spec:
|
||||||
|
accessModes:
|
||||||
|
- ReadWriteOnce
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
storage: 10Gi
|
||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Secret
|
||||||
|
metadata:
|
||||||
|
name: postgres-credentials
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
type: Opaque
|
||||||
|
stringData:
|
||||||
|
POSTGRES_USER: "litellm"
|
||||||
|
POSTGRES_PASSWORD: "litellm123"
|
||||||
|
POSTGRES_DB: "litellm"
|
||||||
|
---
|
||||||
|
apiVersion: apps/v1
|
||||||
|
kind: Deployment
|
||||||
|
metadata:
|
||||||
|
name: postgres
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
spec:
|
||||||
|
replicas: 1
|
||||||
|
selector:
|
||||||
|
matchLabels:
|
||||||
|
app: postgres
|
||||||
|
template:
|
||||||
|
metadata:
|
||||||
|
labels:
|
||||||
|
app: postgres
|
||||||
|
spec:
|
||||||
|
containers:
|
||||||
|
- name: postgres
|
||||||
|
image: postgres:15-alpine
|
||||||
|
ports:
|
||||||
|
- containerPort: 5432
|
||||||
|
env:
|
||||||
|
- name: POSTGRES_USER
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: postgres-credentials
|
||||||
|
key: POSTGRES_USER
|
||||||
|
- name: POSTGRES_PASSWORD
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: postgres-credentials
|
||||||
|
key: POSTGRES_PASSWORD
|
||||||
|
- name: POSTGRES_DB
|
||||||
|
valueFrom:
|
||||||
|
secretKeyRef:
|
||||||
|
name: postgres-credentials
|
||||||
|
key: POSTGRES_DB
|
||||||
|
volumeMounts:
|
||||||
|
- name: postgres-storage
|
||||||
|
mountPath: /var/lib/postgresql/data
|
||||||
|
resources:
|
||||||
|
requests:
|
||||||
|
cpu: 100m
|
||||||
|
memory: 256Mi
|
||||||
|
limits:
|
||||||
|
cpu: 500m
|
||||||
|
memory: 512Mi
|
||||||
|
volumes:
|
||||||
|
- name: postgres-storage
|
||||||
|
persistentVolumeClaim:
|
||||||
|
claimName: postgres-pvc
|
||||||
|
---
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Service
|
||||||
|
metadata:
|
||||||
|
name: postgres
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
spec:
|
||||||
|
selector:
|
||||||
|
app: postgres
|
||||||
|
ports:
|
||||||
|
- port: 5432
|
||||||
|
targetPort: 5432
|
||||||
|
type: ClusterIP
|
||||||
9
apps/litellm/base/ratelimit.yaml
Normal file
9
apps/litellm/base/ratelimit.yaml
Normal file
|
|
@ -0,0 +1,9 @@
|
||||||
|
apiVersion: traefik.io/v1alpha1
|
||||||
|
kind: Middleware
|
||||||
|
metadata:
|
||||||
|
name: rate-limit
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
spec:
|
||||||
|
rateLimit:
|
||||||
|
average: 100
|
||||||
|
burst: 50
|
||||||
10
apps/litellm/base/secret.yaml
Normal file
10
apps/litellm/base/secret.yaml
Normal file
|
|
@ -0,0 +1,10 @@
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Secret
|
||||||
|
metadata:
|
||||||
|
name: litellm-secrets
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
type: Opaque
|
||||||
|
stringData:
|
||||||
|
# Master key for LiteLLM API authentication
|
||||||
|
# Change this to a secure key for production
|
||||||
|
LITELLM_MASTER_KEY: "PLACEHOLDER-UPDATE-IN-OVERLAY"
|
||||||
15
apps/litellm/base/service.yaml
Normal file
15
apps/litellm/base/service.yaml
Normal file
|
|
@ -0,0 +1,15 @@
|
||||||
|
apiVersion: v1
|
||||||
|
kind: Service
|
||||||
|
metadata:
|
||||||
|
name: litellm
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
labels:
|
||||||
|
app: litellm
|
||||||
|
spec:
|
||||||
|
type: ClusterIP
|
||||||
|
ports:
|
||||||
|
- port: 4000
|
||||||
|
targetPort: 4000
|
||||||
|
name: http
|
||||||
|
selector:
|
||||||
|
app: litellm
|
||||||
17
apps/litellm/overlays/prod/kustomization.yaml
Normal file
17
apps/litellm/overlays/prod/kustomization.yaml
Normal file
|
|
@ -0,0 +1,17 @@
|
||||||
|
apiVersion: kustomize.config.k8s.io/v1beta1
|
||||||
|
kind: Kustomization
|
||||||
|
|
||||||
|
namespace: nxtgauge-ai
|
||||||
|
|
||||||
|
resources:
|
||||||
|
- ../../base
|
||||||
|
|
||||||
|
# Use secretGenerator to create production secret
|
||||||
|
secretGenerator:
|
||||||
|
- name: litellm-secrets
|
||||||
|
literals:
|
||||||
|
- LITELLM_MASTER_KEY=sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9
|
||||||
|
|
||||||
|
# Delete the base secret since we're replacing it
|
||||||
|
generatorOptions:
|
||||||
|
disableNameSuffixHash: true
|
||||||
|
|
@ -16,7 +16,7 @@ spec:
|
||||||
spec:
|
spec:
|
||||||
containers:
|
containers:
|
||||||
- name: postgresql
|
- name: postgresql
|
||||||
image: postgres:16-alpine
|
image: registry.nxtgauge.com/postgres:16-alpine
|
||||||
ports:
|
ports:
|
||||||
- name: tcp-postgresql
|
- name: tcp-postgresql
|
||||||
containerPort: 5432
|
containerPort: 5432
|
||||||
|
|
|
||||||
545
docs/AI_PLANS_FINAL_IMPLEMENTATION.md
Normal file
545
docs/AI_PLANS_FINAL_IMPLEMENTATION.md
Normal file
|
|
@ -0,0 +1,545 @@
|
||||||
|
# Final AI Plans Implementation Plan
|
||||||
|
|
||||||
|
## Last Updated
|
||||||
|
2026-06-15
|
||||||
|
|
||||||
|
## Based On
|
||||||
|
User-provided final AI Plans and Limits specification.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Scope Summary
|
||||||
|
|
||||||
|
**Allowed AI features:**
|
||||||
|
- Ask Ash help assistant
|
||||||
|
- AI form filling
|
||||||
|
- Company job tools (JD, skills, candidate matching)
|
||||||
|
- Job seeker AI auto-apply
|
||||||
|
- Professional AI auto-request
|
||||||
|
- Admin helper tools
|
||||||
|
|
||||||
|
**Customer role has NO AI features.**
|
||||||
|
|
||||||
|
**Models:**
|
||||||
|
- `askash-fast` → `qwen3:4b`
|
||||||
|
- `askash-main` → `qwen3:8b`
|
||||||
|
|
||||||
|
**Server:**
|
||||||
|
- `Ramaris` = Ask Ash AI server
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Final Feature Codes
|
||||||
|
|
||||||
|
| Feature Code | Model | Credits |
|
||||||
|
|---|---|---:|
|
||||||
|
| help_answer | askash-fast | 1 |
|
||||||
|
| platform_guidance | askash-fast | 1 |
|
||||||
|
| form_fill | askash-fast | 2 |
|
||||||
|
| form_validate | askash-fast | 1 |
|
||||||
|
| jd_generate | askash-main | 5 |
|
||||||
|
| jd_improve | askash-main | 4 |
|
||||||
|
| skills_extract | askash-fast | 1 |
|
||||||
|
| candidate_match | askash-fast | 1 |
|
||||||
|
| candidate_shortlist | askash-fast | 2 |
|
||||||
|
| job_match | askash-fast | 1 |
|
||||||
|
| auto_apply_suggest | askash-fast | 2 |
|
||||||
|
| auto_apply_execute | askash-main if text generation needed | 5 |
|
||||||
|
| cover_letter_generate | askash-main | 5 |
|
||||||
|
| requirement_match | askash-fast | 1 |
|
||||||
|
| auto_request_suggest | askash-fast | 2 |
|
||||||
|
| auto_request_execute | backend only | 3 |
|
||||||
|
| admin_support_reply | askash-main | 3 |
|
||||||
|
| admin_ticket_summary | askash-fast | 1 |
|
||||||
|
| admin_verification_summary | askash-fast | 1 |
|
||||||
|
| abuse_check | askash-fast | 1 |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Plan Tiers
|
||||||
|
|
||||||
|
| Plan | Monthly Credits | Daily Actions | Models | Best For |
|
||||||
|
|---|---:|---:|---|---|
|
||||||
|
| Free | 10 | 3 | askash-fast | Trial users |
|
||||||
|
| Pro | 100 | 15 | askash-fast + askash-main | Job seekers and professionals |
|
||||||
|
| Business | 300 | 40 | askash-fast + askash-main | Companies |
|
||||||
|
| Enterprise | Custom | Custom | askash-fast + askash-main | High-volume users |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Role-Based AI Access
|
||||||
|
|
||||||
|
| Role | AI Access |
|
||||||
|
|---|---|
|
||||||
|
| Job Seeker | AI auto-apply, job match, cover letter |
|
||||||
|
| Professional | AI auto-request, requirement match |
|
||||||
|
| Company | JD generation, JD improvement, skills extraction, candidate matching |
|
||||||
|
| Customer | No AI features |
|
||||||
|
| Admin | Support helper, summary, verification helper, abuse check |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Database Schema
|
||||||
|
|
||||||
|
### ai_plans
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_plans (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
code VARCHAR(50) UNIQUE NOT NULL,
|
||||||
|
name VARCHAR(100) NOT NULL,
|
||||||
|
monthly_credits INT NOT NULL,
|
||||||
|
daily_action_limit INT NOT NULL,
|
||||||
|
allowed_models JSONB NOT NULL,
|
||||||
|
allowed_features JSONB NOT NULL,
|
||||||
|
is_active BOOLEAN NOT NULL DEFAULT TRUE,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||||
|
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### user_ai_subscriptions
|
||||||
|
```sql
|
||||||
|
CREATE TABLE user_ai_subscriptions (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID UNIQUE NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||||
|
plan_id UUID NOT NULL REFERENCES ai_plans(id),
|
||||||
|
role_code VARCHAR(50),
|
||||||
|
monthly_credits_total INT NOT NULL,
|
||||||
|
monthly_credits_used INT NOT NULL DEFAULT 0,
|
||||||
|
purchased_credits_total INT NOT NULL DEFAULT 0,
|
||||||
|
purchased_credits_used INT NOT NULL DEFAULT 0,
|
||||||
|
daily_actions_used INT NOT NULL DEFAULT 0,
|
||||||
|
current_period_start TIMESTAMP NOT NULL,
|
||||||
|
current_period_end TIMESTAMP NOT NULL,
|
||||||
|
status VARCHAR(30) NOT NULL DEFAULT 'active',
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||||
|
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### ai_feature_costs
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_feature_costs (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
feature_code VARCHAR(100) UNIQUE NOT NULL,
|
||||||
|
display_name VARCHAR(150) NOT NULL,
|
||||||
|
default_model VARCHAR(100) NOT NULL,
|
||||||
|
credit_cost INT NOT NULL,
|
||||||
|
max_input_tokens INT,
|
||||||
|
max_output_tokens INT,
|
||||||
|
is_active BOOLEAN NOT NULL DEFAULT TRUE,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||||
|
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### ai_usage_logs
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_usage_logs (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id),
|
||||||
|
role_code VARCHAR(50),
|
||||||
|
feature_code VARCHAR(100) NOT NULL,
|
||||||
|
model_alias VARCHAR(100) NOT NULL,
|
||||||
|
credits_charged INT NOT NULL,
|
||||||
|
input_tokens INT,
|
||||||
|
output_tokens INT,
|
||||||
|
total_tokens INT,
|
||||||
|
status VARCHAR(30) NOT NULL,
|
||||||
|
request_id VARCHAR(100),
|
||||||
|
error_message TEXT,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### ai_credit_transactions
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_credit_transactions (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id),
|
||||||
|
transaction_type VARCHAR(50) NOT NULL,
|
||||||
|
source VARCHAR(50) NOT NULL,
|
||||||
|
credits INT NOT NULL,
|
||||||
|
balance_after INT NOT NULL,
|
||||||
|
reference_id UUID,
|
||||||
|
description TEXT,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### ai_auto_apply_settings
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_auto_apply_settings (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID UNIQUE NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||||
|
is_enabled BOOLEAN NOT NULL DEFAULT FALSE,
|
||||||
|
preferred_titles JSONB,
|
||||||
|
preferred_locations JSONB,
|
||||||
|
preferred_job_types JSONB,
|
||||||
|
preferred_work_modes JSONB,
|
||||||
|
preferred_skills JSONB,
|
||||||
|
min_salary INT,
|
||||||
|
max_salary INT,
|
||||||
|
max_applications_per_day INT NOT NULL DEFAULT 3,
|
||||||
|
require_user_approval BOOLEAN NOT NULL DEFAULT TRUE,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||||
|
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### ai_auto_apply_logs
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_auto_apply_logs (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id),
|
||||||
|
job_id UUID NOT NULL,
|
||||||
|
match_score INT,
|
||||||
|
status VARCHAR(50) NOT NULL,
|
||||||
|
credits_charged INT NOT NULL DEFAULT 0,
|
||||||
|
generated_cover_letter TEXT,
|
||||||
|
applied_at TIMESTAMP,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### ai_auto_request_settings
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_auto_request_settings (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID UNIQUE NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||||
|
professional_role_code VARCHAR(50) NOT NULL,
|
||||||
|
is_enabled BOOLEAN NOT NULL DEFAULT FALSE,
|
||||||
|
preferred_categories JSONB,
|
||||||
|
preferred_locations JSONB,
|
||||||
|
preferred_requirement_types JSONB,
|
||||||
|
min_budget INT,
|
||||||
|
max_budget INT,
|
||||||
|
max_requests_per_day INT NOT NULL DEFAULT 3,
|
||||||
|
require_user_approval BOOLEAN NOT NULL DEFAULT TRUE,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW(),
|
||||||
|
updated_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### ai_auto_request_logs
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_auto_request_logs (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id),
|
||||||
|
requirement_id UUID NOT NULL,
|
||||||
|
professional_role_code VARCHAR(50) NOT NULL,
|
||||||
|
match_score INT,
|
||||||
|
status VARCHAR(50) NOT NULL,
|
||||||
|
credits_charged INT NOT NULL DEFAULT 0,
|
||||||
|
requested_at TIMESTAMP,
|
||||||
|
created_at TIMESTAMP NOT NULL DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Seed Data
|
||||||
|
|
||||||
|
### Plans
|
||||||
|
```sql
|
||||||
|
INSERT INTO ai_plans (code, name, monthly_credits, daily_action_limit, allowed_models, allowed_features) VALUES
|
||||||
|
('free', 'Free', 10, 3, '["askash-fast"]', '["help_answer", "platform_guidance", "form_fill", "form_validate", "jd_generate"]'),
|
||||||
|
('pro', 'Pro', 100, 15, '["askash-fast", "askash-main"]', '["help_answer", "platform_guidance", "form_fill", "form_validate", "job_match", "auto_apply_suggest", "auto_apply_execute", "cover_letter_generate", "requirement_match", "auto_request_suggest", "auto_request_execute"]'),
|
||||||
|
('business', 'Business', 300, 40, '["askash-fast", "askash-main"]', '["help_answer", "platform_guidance", "form_fill", "form_validate", "jd_generate", "jd_improve", "skills_extract", "candidate_match", "candidate_shortlist"]'),
|
||||||
|
('enterprise', 'Enterprise', 50000, 999999, '["askash-fast", "askash-main"]', '["help_answer", "platform_guidance", "form_fill", "form_validate", "jd_generate", "jd_improve", "skills_extract", "candidate_match", "candidate_shortlist", "job_match", "auto_apply_suggest", "auto_apply_execute", "cover_letter_generate", "requirement_match", "auto_request_suggest", "auto_request_execute", "admin_support_reply", "admin_ticket_summary", "admin_verification_summary", "abuse_check"]');
|
||||||
|
```
|
||||||
|
|
||||||
|
### Feature Costs
|
||||||
|
```sql
|
||||||
|
INSERT INTO ai_feature_costs (feature_code, display_name, default_model, credit_cost) VALUES
|
||||||
|
('help_answer', 'Help Answer', 'askash-fast', 1),
|
||||||
|
('platform_guidance', 'Platform Guidance', 'askash-fast', 1),
|
||||||
|
('form_fill', 'Form Fill', 'askash-fast', 2),
|
||||||
|
('form_validate', 'Form Validate', 'askash-fast', 1),
|
||||||
|
('jd_generate', 'Job Description Generate', 'askash-main', 5),
|
||||||
|
('jd_improve', 'Job Description Improve', 'askash-main', 4),
|
||||||
|
('skills_extract', 'Skills Extract', 'askash-fast', 1),
|
||||||
|
('candidate_match', 'Candidate Match', 'askash-fast', 1),
|
||||||
|
('candidate_shortlist', 'Candidate Shortlist', 'askash-fast', 2),
|
||||||
|
('job_match', 'Job Match', 'askash-fast', 1),
|
||||||
|
('auto_apply_suggest', 'Auto Apply Suggest', 'askash-fast', 2),
|
||||||
|
('auto_apply_execute', 'Auto Apply Execute', 'askash-main', 5),
|
||||||
|
('cover_letter_generate', 'Cover Letter Generate', 'askash-main', 5),
|
||||||
|
('requirement_match', 'Requirement Match', 'askash-fast', 1),
|
||||||
|
('auto_request_suggest', 'Auto Request Suggest', 'askash-fast', 2),
|
||||||
|
('auto_request_execute', 'Auto Request Execute', 'backend', 3),
|
||||||
|
('admin_support_reply', 'Admin Support Reply', 'askash-main', 3),
|
||||||
|
('admin_ticket_summary', 'Admin Ticket Summary', 'askash-fast', 1),
|
||||||
|
('admin_verification_summary', 'Admin Verification Summary', 'askash-fast', 1),
|
||||||
|
('abuse_check', 'Abuse Check', 'askash-fast', 1);
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Backend Services
|
||||||
|
|
||||||
|
### Core Services
|
||||||
|
1. **AiPlanService** - get plan, check allowed models/features
|
||||||
|
2. **AiFeatureService** - get feature cost and default model
|
||||||
|
3. **AiCreditService** - check balance, charge credits, add purchased credits
|
||||||
|
4. **AiUsageService** - log usage, track tokens
|
||||||
|
5. **AiModelRouter** - route feature to correct model alias
|
||||||
|
6. **AiAutoApplyService** - job seeker auto-apply logic
|
||||||
|
7. **AiAutoRequestService** - professional auto-request logic
|
||||||
|
|
||||||
|
### Internal Flow
|
||||||
|
```
|
||||||
|
Request
|
||||||
|
↓
|
||||||
|
Check auth
|
||||||
|
↓
|
||||||
|
Check active role
|
||||||
|
↓
|
||||||
|
Check AI plan allows this feature
|
||||||
|
↓
|
||||||
|
Check daily action limit
|
||||||
|
↓
|
||||||
|
Check credit balance
|
||||||
|
↓
|
||||||
|
Select model alias via AiModelRouter
|
||||||
|
↓
|
||||||
|
Call LiteLLM internal service
|
||||||
|
↓
|
||||||
|
Charge credits
|
||||||
|
↓
|
||||||
|
Write usage log
|
||||||
|
↓
|
||||||
|
Return response
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## API Endpoints
|
||||||
|
|
||||||
|
### General AI
|
||||||
|
```
|
||||||
|
POST /api/ai/help/ask
|
||||||
|
POST /api/ai/forms/fill
|
||||||
|
POST /api/ai/forms/validate
|
||||||
|
```
|
||||||
|
|
||||||
|
### Company AI
|
||||||
|
```
|
||||||
|
POST /api/ai/company/jobs/generate-description
|
||||||
|
POST /api/ai/company/jobs/improve-description
|
||||||
|
POST /api/ai/company/jobs/extract-skills
|
||||||
|
POST /api/ai/company/candidates/match
|
||||||
|
POST /api/ai/company/candidates/shortlist
|
||||||
|
```
|
||||||
|
|
||||||
|
### Job Seeker AI Auto-Apply
|
||||||
|
```
|
||||||
|
GET /api/ai/job-seeker/auto-apply/settings
|
||||||
|
POST /api/ai/job-seeker/auto-apply/settings
|
||||||
|
POST /api/ai/job-seeker/auto-apply/suggest
|
||||||
|
POST /api/ai/job-seeker/auto-apply/execute
|
||||||
|
GET /api/ai/job-seeker/auto-apply/logs
|
||||||
|
```
|
||||||
|
|
||||||
|
### Professional AI Auto-Request
|
||||||
|
```
|
||||||
|
GET /api/ai/professional/auto-request/settings
|
||||||
|
POST /api/ai/professional/auto-request/settings
|
||||||
|
POST /api/ai/professional/auto-request/suggest
|
||||||
|
POST /api/ai/professional/auto-request/execute
|
||||||
|
GET /api/ai/professional/auto-request/logs
|
||||||
|
```
|
||||||
|
|
||||||
|
### Admin AI
|
||||||
|
```
|
||||||
|
POST /api/ai/admin/support/reply
|
||||||
|
POST /api/ai/admin/tickets/summary
|
||||||
|
POST /api/ai/admin/verification/summary
|
||||||
|
POST /api/ai/admin/abuse/check
|
||||||
|
```
|
||||||
|
|
||||||
|
### AI Usage
|
||||||
|
```
|
||||||
|
GET /api/ai/usage/summary
|
||||||
|
GET /api/ai/usage/logs
|
||||||
|
GET /api/ai/credits/balance
|
||||||
|
GET /api/ai/plans
|
||||||
|
POST /api/ai/credits/buy
|
||||||
|
```
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Implementation Order
|
||||||
|
|
||||||
|
### Phase 1: Database Foundation
|
||||||
|
- Create all tables
|
||||||
|
- Seed plans and feature costs
|
||||||
|
- Add migrations
|
||||||
|
|
||||||
|
### Phase 2: Core AI Services
|
||||||
|
- AiPlanService
|
||||||
|
- AiFeatureService
|
||||||
|
- AiCreditService
|
||||||
|
- AiUsageService
|
||||||
|
- AiModelRouter
|
||||||
|
|
||||||
|
### Phase 3: LiteLLM Integration
|
||||||
|
- Internal LiteLLM client
|
||||||
|
- Feature-based model routing
|
||||||
|
|
||||||
|
### Phase 4: Core AI Endpoints
|
||||||
|
- Help assistant
|
||||||
|
- Form fill/validate
|
||||||
|
- Company JD tools
|
||||||
|
|
||||||
|
### Phase 5: Automation Features
|
||||||
|
- Job seeker auto-apply
|
||||||
|
- Professional auto-request
|
||||||
|
|
||||||
|
### Phase 6: Admin AI
|
||||||
|
- Support reply
|
||||||
|
- Ticket summary
|
||||||
|
- Verification summary
|
||||||
|
- Abuse check
|
||||||
|
|
||||||
|
### Phase 7: Frontend
|
||||||
|
- AI credits widget
|
||||||
|
- Usage history
|
||||||
|
- Auto-apply settings
|
||||||
|
- Auto-request settings
|
||||||
|
- Admin AI management
|
||||||
|
|
||||||
|
### Phase 8: Cron Jobs
|
||||||
|
- Monthly credit reset
|
||||||
|
- Daily action reset
|
||||||
|
- Auto-apply suggestion worker
|
||||||
|
- Auto-request suggestion worker
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Files to Create
|
||||||
|
|
||||||
|
### Database
|
||||||
|
- `migrations/001_add_ai_plans_and_limits.sql`
|
||||||
|
|
||||||
|
### Models
|
||||||
|
- `src/models/ai_plan.rs`
|
||||||
|
- `src/models/ai_feature_cost.rs`
|
||||||
|
- `src/models/user_ai_subscription.rs`
|
||||||
|
- `src/models/ai_usage_log.rs`
|
||||||
|
- `src/models/ai_credit_transaction.rs`
|
||||||
|
- `src/models/ai_auto_apply_settings.rs`
|
||||||
|
- `src/models/ai_auto_apply_log.rs`
|
||||||
|
- `src/models/ai_auto_request_settings.rs`
|
||||||
|
- `src/models/ai_auto_request_log.rs`
|
||||||
|
|
||||||
|
### Services
|
||||||
|
- `src/services/ai_plan_service.rs`
|
||||||
|
- `src/services/ai_feature_service.rs`
|
||||||
|
- `src/services/ai_credit_service.rs`
|
||||||
|
- `src/services/ai_usage_service.rs`
|
||||||
|
- `src/services/ai_model_router.rs`
|
||||||
|
- `src/services/ai_auto_apply_service.rs`
|
||||||
|
- `src/services/ai_auto_request_service.rs`
|
||||||
|
- `src/services/litellm_client.rs`
|
||||||
|
|
||||||
|
### Controllers
|
||||||
|
- `src/controllers/ai_controller.rs`
|
||||||
|
- `src/controllers/ai_auto_apply_controller.rs`
|
||||||
|
- `src/controllers/ai_auto_request_controller.rs`
|
||||||
|
- `src/controllers/ai_admin_controller.rs`
|
||||||
|
|
||||||
|
### Middleware
|
||||||
|
- `src/middleware/ai_auth.rs` (role-based AI access)
|
||||||
|
|
||||||
|
### Cron Jobs
|
||||||
|
- `src/cron/ai_credit_reset.rs`
|
||||||
|
- `src/cron/ai_daily_reset.rs`
|
||||||
|
- `src/cron/ai_auto_apply_worker.rs`
|
||||||
|
- `src/cron/ai_auto_request_worker.rs`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Implementation Status
|
||||||
|
|
||||||
|
### Completed
|
||||||
|
- ✅ Phase 1: Database migration (`20260614233620_ai_plans_and_limits`)
|
||||||
|
- ✅ Phase 2: Core AI services in `apps/users/src/ai/`
|
||||||
|
- `plans.rs` — subscription management, plan/feature/model checks, customer role blocking
|
||||||
|
- `credits.rs` — credit balance, charging, daily action limits
|
||||||
|
- `model_router.rs` — `askash-fast` / `askash-main` resolution
|
||||||
|
- `litellm.rs` — internal LiteLLM client
|
||||||
|
- `orchestrator.rs` — combined permission check + model call + charge + log
|
||||||
|
- `middleware.rs` — AI access middleware (auto-creates Free subscription)
|
||||||
|
- `usage.rs` — usage logging helper
|
||||||
|
- ✅ Phase 3: LiteLLM integration via `LiteLlmClient`
|
||||||
|
- ✅ Phase 4: Core AI endpoints wired in `handlers/ai.rs`
|
||||||
|
- `POST /api/ai/chat/message` (legacy) → feature-aware LiteLLM fallback
|
||||||
|
- `POST /api/ai/chat/ask` → `help_answer`
|
||||||
|
- `POST /api/ai/generate-job-field` → `jd_generate`
|
||||||
|
- `POST /api/ai/generate-cover-letter` → `cover_letter_generate`
|
||||||
|
- `POST /api/ai/tailor-resume` → `form_fill`
|
||||||
|
- `POST /api/ai/auto-apply` → `auto_apply_execute`
|
||||||
|
- `POST /api/ai/auto-respond-to-lead` → `auto_request_execute`
|
||||||
|
- `GET /api/ai/usage` → plan-aware credit/status summary
|
||||||
|
- `GET /api/ai/usage/v2` → plan-aware usage summary
|
||||||
|
- ✅ Phase 5: Automation features in `handlers/ai_auto.rs`
|
||||||
|
- `GET/POST /api/ai/auto/job-seeker/auto-apply/settings`
|
||||||
|
- `POST /api/ai/auto/job-seeker/auto-apply/suggest`
|
||||||
|
- `GET /api/ai/auto/job-seeker/auto-apply/logs`
|
||||||
|
- `GET/POST /api/ai/auto/professional/auto-request/settings`
|
||||||
|
- `POST /api/ai/auto/professional/auto-request/suggest`
|
||||||
|
- `GET /api/ai/auto/professional/auto-request/logs`
|
||||||
|
- ✅ Phase 6: Admin AI endpoints in `handlers/admin_ai.rs`
|
||||||
|
- `POST /api/admin/ai/support/reply` → `admin_support_reply`
|
||||||
|
- `POST /api/admin/ai/tickets/summary` → `admin_ticket_summary`
|
||||||
|
- `POST /api/admin/ai/verification/summary` → `admin_verification_summary`
|
||||||
|
- `POST /api/admin/ai/abuse/check` → `abuse_check`
|
||||||
|
- `GET /api/admin/ai/plans`
|
||||||
|
- `GET /api/admin/ai/features`
|
||||||
|
- `POST /api/admin/ai/users/{user_id}/plan`
|
||||||
|
- `POST /api/admin/ai/users/{user_id}/credits`
|
||||||
|
- `GET /api/admin/ai/users/{user_id}/usage`
|
||||||
|
- `GET /api/admin/ai/users/{user_id}/transactions`
|
||||||
|
- ✅ Phase 8 (partial): Cron reset jobs in `apps/cron/src/tasks/ai.rs`
|
||||||
|
- Daily action reset
|
||||||
|
- Monthly credit reset + billing period rollover
|
||||||
|
- ✅ Seed data: plans, feature costs, and AI credit packages populated by migrations
|
||||||
|
- ✅ Extended repositories: plan/feature updates, credit transactions, usage logs, auto settings upserts, auto-apply/auto-request logs, credit packages
|
||||||
|
|
||||||
|
### Not Started
|
||||||
|
- ⬜ Credit purchase flow real payment gateway integration (currently Beeceptor simulation)
|
||||||
|
- ⬜ Kubernetes/infra updates for AI_CREDIT_ADMIN_TOKEN secret
|
||||||
|
- ⬜ Automated integration tests against a real LiteLLM instance
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Key Files
|
||||||
|
|
||||||
|
| File | Purpose |
|
||||||
|
|---|---|
|
||||||
|
| `crates/db/migrations/20260614233620_ai_plans_and_limits.up.sql` | Database schema + seed data |
|
||||||
|
| `crates/db/migrations/20260615060600_ai_credit_packages.up.sql` | AI credit purchase packages + seed data |
|
||||||
|
| `crates/db/src/models/ai/models.rs` | Rust structs for AI tables |
|
||||||
|
| `crates/db/src/models/ai/repository.rs` | SQLx repositories |
|
||||||
|
| `apps/users/src/ai/plans.rs` | Plan/subscription enforcement |
|
||||||
|
| `apps/users/src/ai/credits.rs` | Credit charging |
|
||||||
|
| `apps/users/src/ai/model_router.rs` | Model selection |
|
||||||
|
| `apps/users/src/ai/litellm.rs` | LiteLLM HTTP client |
|
||||||
|
| `apps/users/src/ai/orchestrator.rs` | End-to-end AI call helper |
|
||||||
|
| `apps/users/src/ai/middleware.rs` | AI access middleware |
|
||||||
|
| `apps/users/src/handlers/ai.rs` | Wired AI endpoints |
|
||||||
|
| `apps/users/src/main.rs` | AI module registration |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Key Decisions
|
||||||
|
|
||||||
|
- LiteLLM is internal gateway only
|
||||||
|
- Ollama is internal model runtime only
|
||||||
|
- Ramaris handles AI workloads
|
||||||
|
- Customer role has NO AI features
|
||||||
|
- Credits are feature-based, not token-based
|
||||||
|
- LLM suggests, backend decides and executes
|
||||||
|
- Automation has 3 phases: suggest → approve → auto-execute
|
||||||
|
- Free plan is auto-created on first AI access for non-customer users
|
||||||
260
docs/AI_PLANS_FINAL_PLAN.md
Normal file
260
docs/AI_PLANS_FINAL_PLAN.md
Normal file
|
|
@ -0,0 +1,260 @@
|
||||||
|
# AI Plans Implementation Plan for Nxtgauge
|
||||||
|
|
||||||
|
## 1. Goal
|
||||||
|
Add per-user AI plans with API keys and usage tracking for Ask Ash.
|
||||||
|
|
||||||
|
## 2. User Flow
|
||||||
|
1. User registers → gets a default Free plan
|
||||||
|
2. System generates one API key per user
|
||||||
|
3. User sends AI requests with their API key
|
||||||
|
4. Backend validates key, checks plan limits, forwards to LiteLLM
|
||||||
|
5. Usage is logged and credits are deducted
|
||||||
|
|
||||||
|
## 3. Plan Tiers
|
||||||
|
|
||||||
|
| Plan | Monthly Credits | Models | RPM | Max Tokens |
|
||||||
|
|------|-----------------|--------|-----|------------|
|
||||||
|
| Free | 100 | askash-fast, help-assistant, messenger | 10 | 1000 |
|
||||||
|
| Pro | 1000 | all 4B models | 60 | 4000 |
|
||||||
|
| Business | 5000 | all models (4B + 8B) | 120 | 8000 |
|
||||||
|
| Enterprise | 50000 | all + priority | unlimited | 32000 |
|
||||||
|
|
||||||
|
## 4. Database Tables
|
||||||
|
|
||||||
|
```sql
|
||||||
|
CREATE TABLE ai_plans (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
name VARCHAR(50) UNIQUE NOT NULL,
|
||||||
|
display_name VARCHAR(100) NOT NULL,
|
||||||
|
monthly_credits INTEGER NOT NULL,
|
||||||
|
rate_limit_rpm INTEGER NOT NULL,
|
||||||
|
max_tokens_per_request INTEGER NOT NULL,
|
||||||
|
price_monthly DECIMAL(10,2) NOT NULL,
|
||||||
|
allowed_models JSONB NOT NULL,
|
||||||
|
is_active BOOLEAN DEFAULT true
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE user_ai_subscriptions (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID UNIQUE NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||||
|
plan_id UUID NOT NULL REFERENCES ai_plans(id),
|
||||||
|
credits_remaining INTEGER NOT NULL,
|
||||||
|
credits_used_this_month INTEGER DEFAULT 0,
|
||||||
|
status VARCHAR(50) DEFAULT 'active',
|
||||||
|
period_start TIMESTAMP NOT NULL,
|
||||||
|
period_end TIMESTAMP NOT NULL,
|
||||||
|
created_at TIMESTAMP DEFAULT NOW(),
|
||||||
|
updated_at TIMESTAMP DEFAULT NOW()
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE api_keys (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||||
|
key_hash VARCHAR(255) UNIQUE NOT NULL,
|
||||||
|
key_prefix VARCHAR(50) NOT NULL,
|
||||||
|
is_active BOOLEAN DEFAULT true,
|
||||||
|
created_at TIMESTAMP DEFAULT NOW()
|
||||||
|
);
|
||||||
|
|
||||||
|
CREATE TABLE ai_usage_logs (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id),
|
||||||
|
api_key_id UUID REFERENCES api_keys(id),
|
||||||
|
model VARCHAR(100) NOT NULL,
|
||||||
|
request_type VARCHAR(100) NOT NULL,
|
||||||
|
tokens_input INTEGER NOT NULL,
|
||||||
|
tokens_output INTEGER NOT NULL,
|
||||||
|
tokens_total INTEGER NOT NULL,
|
||||||
|
credits_deducted INTEGER NOT NULL,
|
||||||
|
duration_ms INTEGER,
|
||||||
|
was_successful BOOLEAN DEFAULT true,
|
||||||
|
created_at TIMESTAMP DEFAULT NOW()
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
## 5. Models Available
|
||||||
|
|
||||||
|
| Model Alias | Ollama Model | Use Case |
|
||||||
|
|-------------|--------------|----------|
|
||||||
|
| askash-fast | qwen3:4b | Quick help, forms, chat |
|
||||||
|
| help-assistant | qwen3:4b | Help articles, platform guidance |
|
||||||
|
| messenger | qwen3:4b | Notifications, short messages |
|
||||||
|
| recommender | qwen3:4b | Job/professional matching |
|
||||||
|
| safety-check | qwen3:4b | Spam/abuse detection |
|
||||||
|
| askash-main | qwen3:8b | Resume, cover letters |
|
||||||
|
| jd-generator | qwen3:8b | Job descriptions |
|
||||||
|
| profile-writer | qwen3:8b | Profile completion |
|
||||||
|
| service-writer | qwen3:8b | Service descriptions |
|
||||||
|
| requirement-writer | qwen3:8b | Customer requirements |
|
||||||
|
| support-drafter | qwen3:8b | Support tickets |
|
||||||
|
| decision-support | qwen3:8b | Admin approvals |
|
||||||
|
| ultra-fast | gemma3:270m | Ultra-quick fallback |
|
||||||
|
|
||||||
|
## 6. Backend Services to Build
|
||||||
|
|
||||||
|
### 6.1 ApiKeyService
|
||||||
|
- `generate_key(user_id)` → returns `(full_key, hash, prefix)`
|
||||||
|
- `validate_key(provided_key, stored_hash)` → bool
|
||||||
|
- `save_key(user_id, hash, prefix)` → store in DB
|
||||||
|
- `get_active_key_for_user(user_id)` → Option<ApiKey>
|
||||||
|
|
||||||
|
### 6.2 PlanService
|
||||||
|
- `get_plan_by_name(name)` → AiPlan
|
||||||
|
- `create_subscription(user_id, plan_name)` → setup initial subscription
|
||||||
|
- `can_use_model(user_id, model)` → bool
|
||||||
|
- `get_subscription(user_id)` → UserAiSubscription
|
||||||
|
|
||||||
|
### 6.3 UsageService
|
||||||
|
- `can_make_request(user_id, model)` → checks credits + rate limit + model access
|
||||||
|
- `log_usage(user_id, api_key_id, model, request_type, tokens_input, tokens_output, duration_ms, success)` → deducts credits
|
||||||
|
- `get_usage_summary(user_id, start, end)` → usage stats
|
||||||
|
- `reset_monthly_credits()` → cron job at start of billing period
|
||||||
|
|
||||||
|
### 6.4 LiteLLM Client
|
||||||
|
- `chat_completion(model, messages)` → calls internal LiteLLM service
|
||||||
|
- returns tokens used + response
|
||||||
|
|
||||||
|
## 7. API Endpoints
|
||||||
|
|
||||||
|
### Public (requires API key)
|
||||||
|
```
|
||||||
|
POST /api/v1/ai/chat
|
||||||
|
Headers: Authorization: Bearer sk-nxtgauge-{user_id}-{random}
|
||||||
|
Body: { model, messages, request_type }
|
||||||
|
Response: { choices, usage, credits_remaining }
|
||||||
|
```
|
||||||
|
|
||||||
|
### Authenticated (requires user JWT)
|
||||||
|
```
|
||||||
|
GET /api/v1/ai/usage
|
||||||
|
POST /api/v1/ai/keys
|
||||||
|
GET /api/v1/ai/keys
|
||||||
|
DELETE /api/v1/ai/keys/{id}
|
||||||
|
POST /api/v1/ai/upgrade
|
||||||
|
GET /api/v1/ai/plans
|
||||||
|
```
|
||||||
|
|
||||||
|
## 8. Request Flow
|
||||||
|
|
||||||
|
```
|
||||||
|
User Request
|
||||||
|
↓
|
||||||
|
Nginx/Traefik
|
||||||
|
↓
|
||||||
|
API Gateway
|
||||||
|
↓
|
||||||
|
Extract API Key
|
||||||
|
↓
|
||||||
|
Validate API Key (lookup hash)
|
||||||
|
↓
|
||||||
|
Get User Subscription + Plan
|
||||||
|
↓
|
||||||
|
Check:
|
||||||
|
- Subscription active?
|
||||||
|
- Credits > 0?
|
||||||
|
- Model allowed?
|
||||||
|
- Rate limit OK?
|
||||||
|
↓
|
||||||
|
Forward to LiteLLM (internal master key)
|
||||||
|
↓
|
||||||
|
Parse response tokens
|
||||||
|
↓
|
||||||
|
Log usage + deduct credits
|
||||||
|
↓
|
||||||
|
Return response + X-Credits-Remaining header
|
||||||
|
```
|
||||||
|
|
||||||
|
## 9. Model Selection Helper
|
||||||
|
|
||||||
|
```rust
|
||||||
|
fn select_model(request_type: &str, user_plan: &str) -> &str {
|
||||||
|
match request_type {
|
||||||
|
"help" | "form_fill" | "validation" | "notification" => "askash-fast",
|
||||||
|
"job_recommendation" | "professional_match" => "recommender",
|
||||||
|
"safety_check" | "spam" => "safety-check",
|
||||||
|
"resume" | "cover_letter" | "profile_completion" => "askash-main",
|
||||||
|
"jd_generation" => "jd-generator",
|
||||||
|
"service_description" | "proposal" => "service-writer",
|
||||||
|
"requirement" => "requirement-writer",
|
||||||
|
"support_ticket" => "support-drafter",
|
||||||
|
"admin_approval" => "decision-support",
|
||||||
|
_ => "askash-fast",
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## 10. Frontend Integration
|
||||||
|
|
||||||
|
- Display current plan in user dashboard
|
||||||
|
- Show credits remaining
|
||||||
|
- Show usage chart (daily/weekly/monthly)
|
||||||
|
- Upgrade plan button
|
||||||
|
- Reveal/regenerate API key button
|
||||||
|
|
||||||
|
## 11. Cron Jobs
|
||||||
|
|
||||||
|
- `reset_monthly_credits`: Run at start of each user's billing period
|
||||||
|
- `cleanup_old_usage_logs`: Archive logs older than 90 days
|
||||||
|
- `notify_low_credits`: Send email when credits below 20%
|
||||||
|
|
||||||
|
## 12. Testing Plan
|
||||||
|
|
||||||
|
- Unit tests for key generation and validation
|
||||||
|
- Unit tests for credit deduction
|
||||||
|
- Integration tests for rate limiting
|
||||||
|
- Load tests for concurrent requests
|
||||||
|
- Security tests (invalid keys, expired keys, plan downgrade)
|
||||||
|
|
||||||
|
## 13. Deployment Steps
|
||||||
|
|
||||||
|
1. Add database migrations
|
||||||
|
2. Deploy new backend version
|
||||||
|
3. Seed default plans
|
||||||
|
4. Generate API keys for existing users
|
||||||
|
5. Update frontend to show AI usage
|
||||||
|
6. Monitor for errors
|
||||||
|
|
||||||
|
## 14. MVP Scope (First Version)
|
||||||
|
|
||||||
|
- Free and Pro plans only
|
||||||
|
- Single API key per user
|
||||||
|
- Basic usage tracking
|
||||||
|
- Token-based credit deduction
|
||||||
|
- Monthly credit reset
|
||||||
|
|
||||||
|
## 15. Files to Create/Modify
|
||||||
|
|
||||||
|
### New Files
|
||||||
|
- `migrations/001_add_ai_plans.sql`
|
||||||
|
- `src/models/ai_plan.rs`
|
||||||
|
- `src/models/api_key.rs`
|
||||||
|
- `src/models/ai_usage.rs`
|
||||||
|
- `src/services/api_key_service.rs`
|
||||||
|
- `src/services/plan_service.rs`
|
||||||
|
- `src/services/usage_service.rs`
|
||||||
|
- `src/services/litellm_client.rs`
|
||||||
|
- `src/controllers/ai_controller.rs`
|
||||||
|
- `src/middleware/ai_auth.rs`
|
||||||
|
|
||||||
|
### Modified Files
|
||||||
|
- `src/main.rs` → add routes and services
|
||||||
|
- `src/routes.rs` → register AI routes
|
||||||
|
- existing user model → add plan relationship
|
||||||
|
|
||||||
|
## 16. Timeline
|
||||||
|
|
||||||
|
| Phase | Duration | Deliverable |
|
||||||
|
|-------|----------|-------------|
|
||||||
|
| 1 | 2-3 days | Database + models |
|
||||||
|
| 2 | 3-4 days | Services (keys, plans, usage) |
|
||||||
|
| 3 | 2-3 days | API endpoints + middleware |
|
||||||
|
| 4 | 2-3 days | Frontend usage UI |
|
||||||
|
| 5 | 2 days | Testing + deployment |
|
||||||
|
|
||||||
|
**Total: ~2 weeks for MVP**
|
||||||
|
|
||||||
|
## 17. Next Immediate Step
|
||||||
|
|
||||||
|
Create database migration and Rust models.
|
||||||
|
|
||||||
|
Approve this plan and I'll start Phase 1.
|
||||||
237
docs/AI_PLANS_IMPLEMENTATION.md
Normal file
237
docs/AI_PLANS_IMPLEMENTATION.md
Normal file
|
|
@ -0,0 +1,237 @@
|
||||||
|
# AI Plans API Key Management for Nxtgauge
|
||||||
|
|
||||||
|
## Overview
|
||||||
|
Each user gets a unique API key to track usage and enforce plan limits.
|
||||||
|
|
||||||
|
## API Key Structure
|
||||||
|
|
||||||
|
```
|
||||||
|
User ID: user_12345
|
||||||
|
Plan: free | pro | enterprise
|
||||||
|
API Key: sk-nxtgauge-user_12345-abc123xyz
|
||||||
|
```
|
||||||
|
|
||||||
|
## Implementation Approach
|
||||||
|
|
||||||
|
Since LiteLLM Community Edition has limited virtual key features,
|
||||||
|
we'll implement a custom middleware/proxy approach:
|
||||||
|
|
||||||
|
### Backend Implementation (Rust/Node.js)
|
||||||
|
|
||||||
|
1. **User Registration** → Generate API key
|
||||||
|
2. **API Key Validation** → Check against database
|
||||||
|
3. **Usage Tracking** → Increment counters per request
|
||||||
|
4. **Rate Limiting** → Enforce plan limits
|
||||||
|
|
||||||
|
### Database Schema
|
||||||
|
|
||||||
|
```sql
|
||||||
|
-- Users table
|
||||||
|
CREATE TABLE users (
|
||||||
|
id UUID PRIMARY KEY,
|
||||||
|
email VARCHAR(255) UNIQUE,
|
||||||
|
plan_type VARCHAR(50), -- 'free', 'pro', 'enterprise'
|
||||||
|
api_key VARCHAR(255) UNIQUE,
|
||||||
|
ai_credits_remaining INTEGER DEFAULT 100,
|
||||||
|
monthly_usage_tokens INTEGER DEFAULT 0,
|
||||||
|
created_at TIMESTAMP
|
||||||
|
);
|
||||||
|
|
||||||
|
-- AI Usage tracking
|
||||||
|
CREATE TABLE ai_usage (
|
||||||
|
id UUID PRIMARY KEY,
|
||||||
|
user_id UUID REFERENCES users(id),
|
||||||
|
model VARCHAR(100), -- 'askash-fast', 'askash-main', etc.
|
||||||
|
tokens_input INTEGER,
|
||||||
|
tokens_output INTEGER,
|
||||||
|
request_type VARCHAR(100), -- 'help', 'resume', 'jd', etc.
|
||||||
|
created_at TIMESTAMP
|
||||||
|
);
|
||||||
|
|
||||||
|
-- API Keys table (for rotation)
|
||||||
|
CREATE TABLE api_keys (
|
||||||
|
id UUID PRIMARY KEY,
|
||||||
|
user_id UUID REFERENCES users(id),
|
||||||
|
key_hash VARCHAR(255),
|
||||||
|
is_active BOOLEAN DEFAULT true,
|
||||||
|
created_at TIMESTAMP,
|
||||||
|
expires_at TIMESTAMP
|
||||||
|
);
|
||||||
|
```
|
||||||
|
|
||||||
|
### Plan Tiers
|
||||||
|
|
||||||
|
| Plan | Price | Credits | Models Available | Rate Limit |
|
||||||
|
|------|-------|---------|------------------|------------|
|
||||||
|
| **Free** | $0 | 100/month | askash-fast only | 10 req/min |
|
||||||
|
| **Pro** | $9/mo | 1000/month | All 4B models | 60 req/min |
|
||||||
|
| **Business** | $29/mo | 5000/month | All models incl 8B | 120 req/min |
|
||||||
|
| **Enterprise** | Custom | Unlimited | All + Priority | Unlimited |
|
||||||
|
|
||||||
|
### API Key Generation (Example in Rust)
|
||||||
|
|
||||||
|
```rust
|
||||||
|
use uuid::Uuid;
|
||||||
|
use rand::{distributions::Alphanumeric, Rng};
|
||||||
|
|
||||||
|
pub fn generate_api_key(user_id: &str) -> String {
|
||||||
|
let random_suffix: String = rand::thread_rng()
|
||||||
|
.sample_iter(&Alphanumeric)
|
||||||
|
.take(24)
|
||||||
|
.map(char::from)
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
format!("sk-nxtgauge-{}-{}", user_id, random_suffix)
|
||||||
|
}
|
||||||
|
|
||||||
|
// Example: sk-nxtgauge-user_12345-aBc3xYz9mNqP7rStUvWxYz12
|
||||||
|
```
|
||||||
|
|
||||||
|
### API Middleware Flow
|
||||||
|
|
||||||
|
```
|
||||||
|
User Request (with API key)
|
||||||
|
↓
|
||||||
|
[Nginx/Traefik Ingress]
|
||||||
|
↓
|
||||||
|
[API Gateway - Validate Key]
|
||||||
|
↓
|
||||||
|
[Check Plan & Limits]
|
||||||
|
├─ Check credits remaining
|
||||||
|
├─ Check rate limit (Redis)
|
||||||
|
└─ Check model access
|
||||||
|
↓
|
||||||
|
[Route to LiteLLM]
|
||||||
|
↓
|
||||||
|
[Track Usage]
|
||||||
|
├─ Decrement credits
|
||||||
|
├─ Log usage to DB
|
||||||
|
└─ Update metrics
|
||||||
|
↓
|
||||||
|
[Return Response]
|
||||||
|
```
|
||||||
|
|
||||||
|
### Model Access by Plan
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
free_tier:
|
||||||
|
models:
|
||||||
|
- askash-fast # qwen3:4b
|
||||||
|
- help-assistant # qwen3:4b
|
||||||
|
- messenger # qwen3:4b
|
||||||
|
max_tokens_per_request: 1000
|
||||||
|
|
||||||
|
pro_tier:
|
||||||
|
models:
|
||||||
|
- askash-fast
|
||||||
|
- askash-main # qwen3:8b
|
||||||
|
- help-assistant
|
||||||
|
- jd-generator # qwen3:8b
|
||||||
|
- profile-writer # qwen3:8b
|
||||||
|
- recommender # qwen3:4b
|
||||||
|
max_tokens_per_request: 4000
|
||||||
|
|
||||||
|
business_tier:
|
||||||
|
models:
|
||||||
|
- ALL_MODELS
|
||||||
|
max_tokens_per_request: 8000
|
||||||
|
|
||||||
|
enterprise_tier:
|
||||||
|
models:
|
||||||
|
- ALL_MODELS
|
||||||
|
- PRIORITY_QUEUE
|
||||||
|
max_tokens_per_request: 32000
|
||||||
|
```
|
||||||
|
|
||||||
|
### Cost Calculation (Per 1K tokens)
|
||||||
|
|
||||||
|
Since we're running local Ollama:
|
||||||
|
- Cost is compute-based, not API-based
|
||||||
|
- Track by GPU time or request duration
|
||||||
|
- Alternative: flat rate per request type
|
||||||
|
|
||||||
|
```rust
|
||||||
|
// Example pricing (based on compute cost)
|
||||||
|
const PRICING: &[(str, f64)] = &[
|
||||||
|
("askash-fast", 0.001), // $0.001 per 1K tokens
|
||||||
|
("askash-main", 0.005), // $0.005 per 1K tokens
|
||||||
|
("jd-generator", 0.008), // $0.008 per 1K tokens
|
||||||
|
("profile-writer", 0.008), // $0.008 per 1K tokens
|
||||||
|
];
|
||||||
|
```
|
||||||
|
|
||||||
|
### Usage Endpoints for Frontend
|
||||||
|
|
||||||
|
```javascript
|
||||||
|
// Get user's current usage
|
||||||
|
GET /api/v1/ai/usage
|
||||||
|
Headers: Authorization: Bearer sk-nxtgauge-user_12345-...
|
||||||
|
|
||||||
|
Response:
|
||||||
|
{
|
||||||
|
"plan": "pro",
|
||||||
|
"credits_remaining": 750,
|
||||||
|
"credits_used_this_month": 250,
|
||||||
|
"requests_today": 45,
|
||||||
|
"rate_limit": {
|
||||||
|
"requests_per_minute": 60,
|
||||||
|
"current_window": "58/60"
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Implementation in Existing Backend
|
||||||
|
|
||||||
|
Add to your Rust backend:
|
||||||
|
|
||||||
|
1. **Migration**: Add `api_key`, `ai_plan`, `ai_credits` columns to users table
|
||||||
|
2. **Middleware**: Create `AiAuthMiddleware` to validate keys
|
||||||
|
3. **Service**: Create `AiUsageService` to track and limit
|
||||||
|
4. **Endpoints**:
|
||||||
|
- POST /api/v1/ai/chat (with API key auth)
|
||||||
|
- GET /api/v1/ai/usage
|
||||||
|
- POST /api/v1/ai/upgrade (change plan)
|
||||||
|
|
||||||
|
### Quick Start Commands
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Generate API key for user
|
||||||
|
curl -X POST https://api.nxtgauge.com/v1/ai/keys \
|
||||||
|
-H "Authorization: Bearer $USER_JWT" \
|
||||||
|
-d '{"plan": "pro"}'
|
||||||
|
|
||||||
|
# Use API key
|
||||||
|
curl https://llm.nxtgauge.com/v1/chat/completions \
|
||||||
|
-H "Authorization: Bearer sk-nxtgauge-user_12345-abc123" \
|
||||||
|
-H "Content-Type: application/json" \
|
||||||
|
-d '{
|
||||||
|
"model": "askash-main",
|
||||||
|
"messages": [{"role": "user", "content": "Help with my resume"}]
|
||||||
|
}'
|
||||||
|
```
|
||||||
|
|
||||||
|
## Files to Create
|
||||||
|
|
||||||
|
1. `/src/services/ai_usage.rs` - Usage tracking service
|
||||||
|
2. `/src/middleware/ai_auth.rs` - API key validation
|
||||||
|
3. `/src/models/ai_plan.rs` - Plan definitions
|
||||||
|
4. Database migrations for API keys and usage tables
|
||||||
|
|
||||||
|
## Next Steps
|
||||||
|
|
||||||
|
1. Choose: Build custom middleware OR use LiteLLM Enterprise
|
||||||
|
2. Create database migrations
|
||||||
|
3. Implement API key generation
|
||||||
|
4. Add usage tracking middleware
|
||||||
|
5. Create billing integration
|
||||||
|
|
||||||
|
## LiteLLM Alternative
|
||||||
|
|
||||||
|
For simpler setup, LiteLLM Enterprise ($500/mo) provides:
|
||||||
|
- Built-in virtual keys
|
||||||
|
- Usage dashboards
|
||||||
|
- Team management
|
||||||
|
- Budget controls
|
||||||
|
- SSO/SAML
|
||||||
|
|
||||||
|
But custom implementation gives more control and lower cost.
|
||||||
754
docs/AI_PLANS_IMPLEMENTATION_PLAN.md
Normal file
754
docs/AI_PLANS_IMPLEMENTATION_PLAN.md
Normal file
|
|
@ -0,0 +1,754 @@
|
||||||
|
# AI Plans Implementation Plan
|
||||||
|
|
||||||
|
## Executive Summary
|
||||||
|
Implement per-user API keys with usage tracking and plan tiers for Ask Ash AI assistant.
|
||||||
|
|
||||||
|
## Phase 1: Database Design (Week 1)
|
||||||
|
|
||||||
|
### 1.1 New Tables
|
||||||
|
|
||||||
|
```sql
|
||||||
|
-- Migration: 001_add_ai_plans.sql
|
||||||
|
|
||||||
|
-- API Keys table (supports rotation)
|
||||||
|
CREATE TABLE api_keys (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||||
|
key_hash VARCHAR(255) UNIQUE NOT NULL, -- bcrypt hash of key
|
||||||
|
key_prefix VARCHAR(20) NOT NULL, -- sk-nxtgauge-abc...
|
||||||
|
is_active BOOLEAN DEFAULT true,
|
||||||
|
created_at TIMESTAMP DEFAULT NOW(),
|
||||||
|
expires_at TIMESTAMP NULL,
|
||||||
|
revoked_at TIMESTAMP NULL,
|
||||||
|
revoked_reason TEXT NULL
|
||||||
|
);
|
||||||
|
|
||||||
|
-- AI Plans table (plan definitions)
|
||||||
|
CREATE TABLE ai_plans (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
name VARCHAR(50) UNIQUE NOT NULL, -- 'free', 'pro', 'business', 'enterprise'
|
||||||
|
display_name VARCHAR(100) NOT NULL,
|
||||||
|
monthly_credits INTEGER NOT NULL,
|
||||||
|
rate_limit_rpm INTEGER NOT NULL, -- requests per minute
|
||||||
|
rate_limit_rph INTEGER NOT NULL, -- requests per hour
|
||||||
|
max_tokens_per_request INTEGER NOT NULL,
|
||||||
|
price_monthly DECIMAL(10,2) NOT NULL,
|
||||||
|
features JSONB NOT NULL, -- allowed models, etc.
|
||||||
|
is_active BOOLEAN DEFAULT true,
|
||||||
|
created_at TIMESTAMP DEFAULT NOW()
|
||||||
|
);
|
||||||
|
|
||||||
|
-- User AI Subscriptions
|
||||||
|
CREATE TABLE user_ai_subscriptions (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID UNIQUE NOT NULL REFERENCES users(id) ON DELETE CASCADE,
|
||||||
|
plan_id UUID NOT NULL REFERENCES ai_plans(id),
|
||||||
|
credits_remaining INTEGER NOT NULL,
|
||||||
|
credits_used_this_month INTEGER DEFAULT 0,
|
||||||
|
subscription_status VARCHAR(50) DEFAULT 'active', -- 'active', 'paused', 'cancelled'
|
||||||
|
current_period_start TIMESTAMP NOT NULL,
|
||||||
|
current_period_end TIMESTAMP NOT NULL,
|
||||||
|
cancelled_at TIMESTAMP NULL,
|
||||||
|
created_at TIMESTAMP DEFAULT NOW(),
|
||||||
|
updated_at TIMESTAMP DEFAULT NOW()
|
||||||
|
);
|
||||||
|
|
||||||
|
-- AI Usage Logs (for tracking & billing)
|
||||||
|
CREATE TABLE ai_usage_logs (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id),
|
||||||
|
api_key_id UUID REFERENCES api_keys(id),
|
||||||
|
model VARCHAR(100) NOT NULL, -- 'askash-fast', 'askash-main', etc.
|
||||||
|
request_type VARCHAR(100) NOT NULL, -- 'help', 'resume', 'jd', 'cover_letter'
|
||||||
|
tokens_input INTEGER NOT NULL,
|
||||||
|
tokens_output INTEGER NOT NULL,
|
||||||
|
tokens_total INTEGER NOT NULL,
|
||||||
|
cost_estimate DECIMAL(10,6), -- calculated cost
|
||||||
|
request_duration_ms INTEGER, -- response time
|
||||||
|
was_successful BOOLEAN DEFAULT true,
|
||||||
|
error_message TEXT NULL,
|
||||||
|
ip_address INET,
|
||||||
|
user_agent TEXT,
|
||||||
|
created_at TIMESTAMP DEFAULT NOW()
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Rate Limit Tracking (Redis alternative)
|
||||||
|
CREATE TABLE rate_limit_windows (
|
||||||
|
id UUID PRIMARY KEY DEFAULT gen_random_uuid(),
|
||||||
|
user_id UUID NOT NULL REFERENCES users(id),
|
||||||
|
window_start TIMESTAMP NOT NULL,
|
||||||
|
window_end TIMESTAMP NOT NULL,
|
||||||
|
requests_count INTEGER DEFAULT 0,
|
||||||
|
UNIQUE(user_id, window_start)
|
||||||
|
);
|
||||||
|
|
||||||
|
-- Insert default plans
|
||||||
|
INSERT INTO ai_plans (name, display_name, monthly_credits, rate_limit_rpm, rate_limit_rph, max_tokens_per_request, price_monthly, features) VALUES
|
||||||
|
('free', 'Free', 100, 10, 100, 1000, 0.00, '{"models": ["askash-fast", "help-assistant", "messenger"]}'),
|
||||||
|
('pro', 'Pro', 1000, 60, 1000, 4000, 9.00, '{"models": ["askash-fast", "askash-main", "help-assistant", "messenger", "recommender", "safety-check"]}'),
|
||||||
|
('business', 'Business', 5000, 120, 5000, 8000, 29.00, '{"models": ["askash-fast", "askash-main", "jd-generator", "profile-writer", "service-writer", "requirement-writer"]}'),
|
||||||
|
('enterprise', 'Enterprise', 50000, 0, 0, 32000, 99.00, '{"models": ["all"], "priority": true}');
|
||||||
|
|
||||||
|
-- Indexes for performance
|
||||||
|
CREATE INDEX idx_api_keys_user_id ON api_keys(user_id);
|
||||||
|
CREATE INDEX idx_api_keys_key_hash ON api_keys(key_hash);
|
||||||
|
CREATE INDEX idx_user_ai_subscriptions_user_id ON user_ai_subscriptions(user_id);
|
||||||
|
CREATE INDEX idx_ai_usage_logs_user_id ON ai_usage_logs(user_id);
|
||||||
|
CREATE INDEX idx_ai_usage_logs_created_at ON ai_usage_logs(created_at);
|
||||||
|
CREATE INDEX idx_ai_usage_logs_model ON ai_usage_logs(model);
|
||||||
|
```
|
||||||
|
|
||||||
|
### 1.2 Migration Strategy
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# Run migrations
|
||||||
|
sqlx migrate run --source ./migrations
|
||||||
|
|
||||||
|
# Or manual SQL execution
|
||||||
|
psql $DATABASE_URL < migrations/001_add_ai_plans.sql
|
||||||
|
```
|
||||||
|
|
||||||
|
## Phase 2: Core Implementation (Week 1-2)
|
||||||
|
|
||||||
|
### 2.1 Models/DTOs
|
||||||
|
|
||||||
|
```rust
|
||||||
|
// src/models/ai_plan.rs
|
||||||
|
|
||||||
|
use serde::{Deserialize, Serialize};
|
||||||
|
use sqlx::FromRow;
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, FromRow)]
|
||||||
|
pub struct AiPlan {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub name: String,
|
||||||
|
pub display_name: String,
|
||||||
|
pub monthly_credits: i32,
|
||||||
|
pub rate_limit_rpm: i32,
|
||||||
|
pub rate_limit_rph: i32,
|
||||||
|
pub max_tokens_per_request: i32,
|
||||||
|
pub price_monthly: f64,
|
||||||
|
pub features: serde_json::Value,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, FromRow)]
|
||||||
|
pub struct ApiKey {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub user_id: Uuid,
|
||||||
|
pub key_hash: String,
|
||||||
|
pub key_prefix: String,
|
||||||
|
pub is_active: bool,
|
||||||
|
pub created_at: chrono::DateTime<chrono::Utc>,
|
||||||
|
pub expires_at: Option<chrono::DateTime<chrono::Utc>>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, FromRow)]
|
||||||
|
pub struct UserAiSubscription {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub user_id: Uuid,
|
||||||
|
pub plan_id: Uuid,
|
||||||
|
pub credits_remaining: i32,
|
||||||
|
pub credits_used_this_month: i32,
|
||||||
|
pub subscription_status: String,
|
||||||
|
pub current_period_start: chrono::DateTime<chrono::Utc>,
|
||||||
|
pub current_period_end: chrono::DateTime<chrono::Utc>,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, Serialize, Deserialize, FromRow)]
|
||||||
|
pub struct AiUsageLog {
|
||||||
|
pub id: Uuid,
|
||||||
|
pub user_id: Uuid,
|
||||||
|
pub api_key_id: Option<Uuid>,
|
||||||
|
pub model: String,
|
||||||
|
pub request_type: String,
|
||||||
|
pub tokens_input: i32,
|
||||||
|
pub tokens_output: i32,
|
||||||
|
pub tokens_total: i32,
|
||||||
|
pub cost_estimate: Option<f64>,
|
||||||
|
pub request_duration_ms: Option<i32>,
|
||||||
|
pub was_successful: bool,
|
||||||
|
pub created_at: chrono::DateTime<chrono::Utc>,
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2.2 API Key Generation Service
|
||||||
|
|
||||||
|
```rust
|
||||||
|
// src/services/api_key_service.rs
|
||||||
|
|
||||||
|
use bcrypt::{hash, verify, DEFAULT_COST};
|
||||||
|
use rand::{distributions::Alphanumeric, Rng};
|
||||||
|
use uuid::Uuid;
|
||||||
|
|
||||||
|
pub struct ApiKeyService;
|
||||||
|
|
||||||
|
impl ApiKeyService {
|
||||||
|
/// Generate a new API key
|
||||||
|
/// Returns: (full_key, key_hash, key_prefix)
|
||||||
|
pub fn generate_key(user_id: Uuid) -> (String, String, String) {
|
||||||
|
let prefix = "sk-nxtgauge";
|
||||||
|
let user_part = user_id.to_string().split('-').next().unwrap_or("");
|
||||||
|
let random_suffix: String = rand::thread_rng()
|
||||||
|
.sample_iter(&Alphanumeric)
|
||||||
|
.take(24)
|
||||||
|
.map(char::from)
|
||||||
|
.collect();
|
||||||
|
|
||||||
|
let full_key = format!("{}-{}-{}", prefix, user_part, random_suffix);
|
||||||
|
let key_prefix = format!("{}-{}-", prefix, user_part);
|
||||||
|
|
||||||
|
// Hash for storage (use first 8 chars as salt identifier)
|
||||||
|
let key_hash = hash(&full_key, DEFAULT_COST).unwrap();
|
||||||
|
|
||||||
|
(full_key, key_hash, key_prefix)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Validate an API key against hash
|
||||||
|
pub fn validate_key(provided_key: &str, stored_hash: &str) -> bool {
|
||||||
|
verify(provided_key, stored_hash).unwrap_or(false)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Extract user ID from API key (without validation)
|
||||||
|
pub fn extract_user_id_from_key(key: &str) -> Option<String> {
|
||||||
|
let parts: Vec<&str> = key.split('-').collect();
|
||||||
|
if parts.len() >= 3 && parts[0] == "sk" && parts[1] == "nxtgauge" {
|
||||||
|
Some(parts[2].to_string())
|
||||||
|
} else {
|
||||||
|
None
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[cfg(test)]
|
||||||
|
mod tests {
|
||||||
|
use super::*;
|
||||||
|
|
||||||
|
#[test]
|
||||||
|
fn test_generate_key() {
|
||||||
|
let user_id = Uuid::new_v4();
|
||||||
|
let (full_key, hash, prefix) = ApiKeyService::generate_key(user_id);
|
||||||
|
|
||||||
|
assert!(full_key.starts_with("sk-nxtgauge-"));
|
||||||
|
assert_eq!(full_key.len(), 45); // sk-nxtgauge- + 8 + - + 24
|
||||||
|
assert!(ApiKeyService::validate_key(&full_key, &hash));
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
### 2.3 Usage Tracking Service
|
||||||
|
|
||||||
|
```rust
|
||||||
|
// src/services/ai_usage_service.rs
|
||||||
|
|
||||||
|
use sqlx::PgPool;
|
||||||
|
use uuid::Uuid;
|
||||||
|
use chrono::{DateTime, Utc, Duration};
|
||||||
|
|
||||||
|
pub struct AiUsageService {
|
||||||
|
db: PgPool,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl AiUsageService {
|
||||||
|
pub fn new(db: PgPool) -> Self {
|
||||||
|
Self { db }
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Check if user has credits and rate limit allows request
|
||||||
|
pub async fn can_make_request(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
model: &str,
|
||||||
|
) -> Result<RequestAllowance, AiError> {
|
||||||
|
// Get user's subscription
|
||||||
|
let subscription = sqlx::query_as::<_, UserAiSubscription>(
|
||||||
|
"SELECT * FROM user_ai_subscriptions WHERE user_id = $1"
|
||||||
|
)
|
||||||
|
.bind(user_id)
|
||||||
|
.fetch_optional(&self.db)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
let subscription = subscription.ok_or(AiError::NoSubscription)?;
|
||||||
|
|
||||||
|
// Check subscription status
|
||||||
|
if subscription.subscription_status != "active" {
|
||||||
|
return Err(AiError::SubscriptionInactive);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Check credits
|
||||||
|
if subscription.credits_remaining <= 0 {
|
||||||
|
return Err(AiError::InsufficientCredits);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Get plan details
|
||||||
|
let plan = sqlx::query_as::<_, AiPlan>(
|
||||||
|
"SELECT * FROM ai_plans WHERE id = $1"
|
||||||
|
)
|
||||||
|
.bind(subscription.plan_id)
|
||||||
|
.fetch_one(&self.db)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
// Check rate limit (current window)
|
||||||
|
let window_start = Utc::now() - Duration::minutes(1);
|
||||||
|
let recent_requests: i64 = sqlx::query_scalar(
|
||||||
|
"SELECT COUNT(*) FROM ai_usage_logs
|
||||||
|
WHERE user_id = $1 AND created_at > $2"
|
||||||
|
)
|
||||||
|
.bind(user_id)
|
||||||
|
.bind(window_start)
|
||||||
|
.fetch_one(&self.db)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
if recent_requests >= plan.rate_limit_rpm as i64 {
|
||||||
|
return Err(AiError::RateLimitExceeded);
|
||||||
|
}
|
||||||
|
|
||||||
|
Ok(RequestAllowance {
|
||||||
|
user_id,
|
||||||
|
plan,
|
||||||
|
subscription,
|
||||||
|
remaining_requests: plan.rate_limit_rpm - recent_requests as i32,
|
||||||
|
})
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Log AI usage and deduct credits
|
||||||
|
pub async fn log_usage(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
api_key_id: Option<Uuid>,
|
||||||
|
model: &str,
|
||||||
|
request_type: &str,
|
||||||
|
tokens_input: i32,
|
||||||
|
tokens_output: i32,
|
||||||
|
duration_ms: i32,
|
||||||
|
success: bool,
|
||||||
|
) -> Result<(), AiError> {
|
||||||
|
let tokens_total = tokens_input + tokens_output;
|
||||||
|
|
||||||
|
// Calculate cost (example pricing)
|
||||||
|
let cost = match model {
|
||||||
|
"askash-fast" | "help-assistant" | "messenger" => tokens_total as f64 * 0.000001, // $0.001 per 1K tokens
|
||||||
|
"askash-main" | "jd-generator" | "profile-writer" => tokens_total as f64 * 0.000005, // $0.005 per 1K tokens
|
||||||
|
_ => tokens_total as f64 * 0.000005,
|
||||||
|
};
|
||||||
|
|
||||||
|
// Insert usage log
|
||||||
|
sqlx::query(
|
||||||
|
"INSERT INTO ai_usage_logs
|
||||||
|
(user_id, api_key_id, model, request_type, tokens_input, tokens_output,
|
||||||
|
tokens_total, cost_estimate, request_duration_ms, was_successful)
|
||||||
|
VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $9, $10)"
|
||||||
|
)
|
||||||
|
.bind(user_id)
|
||||||
|
.bind(api_key_id)
|
||||||
|
.bind(model)
|
||||||
|
.bind(request_type)
|
||||||
|
.bind(tokens_input)
|
||||||
|
.bind(tokens_output)
|
||||||
|
.bind(tokens_total)
|
||||||
|
.bind(cost)
|
||||||
|
.bind(duration_ms)
|
||||||
|
.bind(success)
|
||||||
|
.execute(&self.db)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
// Deduct credits (1 credit per 1000 tokens, minimum 1)
|
||||||
|
let credits_to_deduct = ((tokens_total as f64 / 1000.0).ceil() as i32).max(1);
|
||||||
|
|
||||||
|
sqlx::query(
|
||||||
|
"UPDATE user_ai_subscriptions
|
||||||
|
SET credits_remaining = credits_remaining - $1,
|
||||||
|
credits_used_this_month = credits_used_this_month + $1,
|
||||||
|
updated_at = NOW()
|
||||||
|
WHERE user_id = $2"
|
||||||
|
)
|
||||||
|
.bind(credits_to_deduct)
|
||||||
|
.bind(user_id)
|
||||||
|
.execute(&self.db)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(())
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Get usage statistics for user
|
||||||
|
pub async fn get_user_usage(
|
||||||
|
&self,
|
||||||
|
user_id: Uuid,
|
||||||
|
start_date: DateTime<Utc>,
|
||||||
|
end_date: DateTime<Utc>,
|
||||||
|
) -> Result<UsageStats, AiError> {
|
||||||
|
let stats = sqlx::query_as::<_, UsageStats>(
|
||||||
|
"SELECT
|
||||||
|
COUNT(*) as total_requests,
|
||||||
|
SUM(tokens_input) as total_tokens_input,
|
||||||
|
SUM(tokens_output) as total_tokens_output,
|
||||||
|
SUM(tokens_total) as total_tokens,
|
||||||
|
SUM(cost_estimate) as total_cost,
|
||||||
|
COUNT(CASE WHEN was_successful = false THEN 1 END) as failed_requests
|
||||||
|
FROM ai_usage_logs
|
||||||
|
WHERE user_id = $1 AND created_at BETWEEN $2 AND $3"
|
||||||
|
)
|
||||||
|
.bind(user_id)
|
||||||
|
.bind(start_date)
|
||||||
|
.bind(end_date)
|
||||||
|
.fetch_one(&self.db)
|
||||||
|
.await?;
|
||||||
|
|
||||||
|
Ok(stats)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct RequestAllowance {
|
||||||
|
pub user_id: Uuid,
|
||||||
|
pub plan: AiPlan,
|
||||||
|
pub subscription: UserAiSubscription,
|
||||||
|
pub remaining_requests: i32,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug)]
|
||||||
|
pub struct UsageStats {
|
||||||
|
pub total_requests: i64,
|
||||||
|
pub total_tokens_input: i64,
|
||||||
|
pub total_tokens_output: i64,
|
||||||
|
pub total_tokens: i64,
|
||||||
|
pub total_cost: Option<f64>,
|
||||||
|
pub failed_requests: i64,
|
||||||
|
}
|
||||||
|
|
||||||
|
#[derive(Debug, thiserror::Error)]
|
||||||
|
pub enum AiError {
|
||||||
|
#[error("No subscription found")]
|
||||||
|
NoSubscription,
|
||||||
|
#[error("Subscription inactive")]
|
||||||
|
SubscriptionInactive,
|
||||||
|
#[error("Insufficient credits")]
|
||||||
|
InsufficientCredits,
|
||||||
|
#[error("Rate limit exceeded")]
|
||||||
|
RateLimitExceeded,
|
||||||
|
#[error("Database error: {0}")]
|
||||||
|
Database(#[from] sqlx::Error),
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Phase 3: API Endpoints (Week 2)
|
||||||
|
|
||||||
|
### 3.1 AI Controller
|
||||||
|
|
||||||
|
```rust
|
||||||
|
// src/controllers/ai_controller.rs
|
||||||
|
|
||||||
|
use actix_web::{web, HttpResponse, HttpRequest};
|
||||||
|
use crate::services::AiUsageService;
|
||||||
|
use crate::middleware::AiAuth;
|
||||||
|
|
||||||
|
pub fn ai_routes(cfg: &mut web::ServiceConfig) {
|
||||||
|
cfg.service(
|
||||||
|
web::scope("/api/v1/ai")
|
||||||
|
// These require API key auth
|
||||||
|
.route("/chat", web::post().to(chat_completion))
|
||||||
|
.route("/usage", web::get().to(get_usage))
|
||||||
|
// These require user JWT auth
|
||||||
|
.route("/keys", web::post().to(generate_api_key))
|
||||||
|
.route("/keys", web::get().to(list_api_keys))
|
||||||
|
.route("/keys/{key_id}", web::delete().to(revoke_api_key))
|
||||||
|
.route("/upgrade", web::post().to(upgrade_plan))
|
||||||
|
);
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Main chat endpoint (requires API key)
|
||||||
|
async fn chat_completion(
|
||||||
|
req: HttpRequest,
|
||||||
|
body: web::Json<ChatRequest>,
|
||||||
|
usage_service: web::Data<AiUsageService>,
|
||||||
|
litellm_client: web::Data<LitellmClient>,
|
||||||
|
) -> HttpResponse {
|
||||||
|
let start_time = std::time::Instant::now();
|
||||||
|
|
||||||
|
// Extract API key from header
|
||||||
|
let api_key = match extract_api_key(&req) {
|
||||||
|
Some(key) => key,
|
||||||
|
None => return HttpResponse::Unauthorized().json(ErrorResponse {
|
||||||
|
error: "Missing API key".to_string(),
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
|
||||||
|
// Validate and get user info
|
||||||
|
let (user_id, api_key_id, plan) = match validate_api_key(&api_key).await {
|
||||||
|
Ok(info) => info,
|
||||||
|
Err(e) => return HttpResponse::Unauthorized().json(ErrorResponse {
|
||||||
|
error: e.to_string(),
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
|
||||||
|
// Check if user can make request
|
||||||
|
let allowance = match usage_service.can_make_request(user_id, &body.model).await {
|
||||||
|
Ok(a) => a,
|
||||||
|
Err(AiError::InsufficientCredits) => {
|
||||||
|
return HttpResponse::PaymentRequired().json(ErrorResponse {
|
||||||
|
error: "Insufficient credits. Please upgrade your plan.".to_string(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Err(AiError::RateLimitExceeded) => {
|
||||||
|
return HttpResponse::TooManyRequests().json(ErrorResponse {
|
||||||
|
error: "Rate limit exceeded. Please slow down.".to_string(),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
Err(e) => return HttpResponse::InternalServerError().json(ErrorResponse {
|
||||||
|
error: e.to_string(),
|
||||||
|
}),
|
||||||
|
};
|
||||||
|
|
||||||
|
// Forward to LiteLLM
|
||||||
|
let litellm_response = match litellm_client.chat_completion(&body).await {
|
||||||
|
Ok(resp) => resp,
|
||||||
|
Err(e) => {
|
||||||
|
// Log failed request
|
||||||
|
let _ = usage_service.log_usage(
|
||||||
|
user_id, api_key_id, &body.model, &body.request_type,
|
||||||
|
0, 0, start_time.elapsed().as_millis() as i32, false,
|
||||||
|
).await;
|
||||||
|
|
||||||
|
return HttpResponse::InternalServerError().json(ErrorResponse {
|
||||||
|
error: format!("LiteLLM error: {}", e),
|
||||||
|
});
|
||||||
|
}
|
||||||
|
};
|
||||||
|
|
||||||
|
// Log successful usage
|
||||||
|
let duration_ms = start_time.elapsed().as_millis() as i32;
|
||||||
|
let tokens_input = litellm_response.usage.prompt_tokens;
|
||||||
|
let tokens_output = litellm_response.usage.completion_tokens;
|
||||||
|
|
||||||
|
let _ = usage_service.log_usage(
|
||||||
|
user_id, api_key_id, &body.model, &body.request_type,
|
||||||
|
tokens_input, tokens_output, duration_ms, true,
|
||||||
|
).await;
|
||||||
|
|
||||||
|
// Return response with headers
|
||||||
|
HttpResponse::Ok()
|
||||||
|
.insert_header(("X-RateLimit-Remaining", allowance.remaining_requests.to_string()))
|
||||||
|
.insert_header(("X-Credits-Remaining", allowance.subscription.credits_remaining.to_string()))
|
||||||
|
.json(litellm_response)
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Get user's usage statistics
|
||||||
|
async fn get_usage(
|
||||||
|
auth: web::ReqData<UserClaims>, // JWT auth
|
||||||
|
usage_service: web::Data<AiUsageService>,
|
||||||
|
query: web::Query<UsageQuery>,
|
||||||
|
) -> HttpResponse {
|
||||||
|
let start_date = query.start_date.unwrap_or_else(|| {
|
||||||
|
Utc::now() - Duration::days(30)
|
||||||
|
});
|
||||||
|
let end_date = query.end_date.unwrap_or(Utc::now());
|
||||||
|
|
||||||
|
match usage_service.get_user_usage(auth.user_id, start_date, end_date).await {
|
||||||
|
Ok(stats) => HttpResponse::Ok().json(stats),
|
||||||
|
Err(e) => HttpResponse::InternalServerError().json(ErrorResponse {
|
||||||
|
error: e.to_string(),
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
/// Generate new API key
|
||||||
|
async fn generate_api_key(
|
||||||
|
auth: web::ReqData<UserClaims>,
|
||||||
|
body: web::Json<GenerateKeyRequest>,
|
||||||
|
api_key_service: web::Data<ApiKeyService>,
|
||||||
|
) -> HttpResponse {
|
||||||
|
let (full_key, key_hash, key_prefix) = ApiKeyService::generate_key(auth.user_id);
|
||||||
|
|
||||||
|
// Save to database
|
||||||
|
match api_key_service.save_key(auth.user_id, &key_hash, &key_prefix).await {
|
||||||
|
Ok(_) => HttpResponse::Ok().json(GenerateKeyResponse {
|
||||||
|
api_key: full_key,
|
||||||
|
prefix: key_prefix,
|
||||||
|
created_at: Utc::now(),
|
||||||
|
}),
|
||||||
|
Err(e) => HttpResponse::InternalServerError().json(ErrorResponse {
|
||||||
|
error: e.to_string(),
|
||||||
|
}),
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Phase 4: Middleware (Week 2)
|
||||||
|
|
||||||
|
### 4.1 API Key Extraction
|
||||||
|
|
||||||
|
```rust
|
||||||
|
// src/middleware/ai_auth.rs
|
||||||
|
|
||||||
|
use actix_web::{dev::ServiceRequest, Error, HttpMessage};
|
||||||
|
use actix_web::dev::{Transform, Service};
|
||||||
|
use futures::future::{LocalBoxFuture, ok, Ready};
|
||||||
|
use std::task::{Context, Poll};
|
||||||
|
|
||||||
|
pub struct AiAuth;
|
||||||
|
|
||||||
|
impl<S, B> Transform<S, ServiceRequest> for AiAuth
|
||||||
|
where
|
||||||
|
S: Service<ServiceRequest, Response = actix_web::dev::ServiceResponse<B>, Error = Error>,
|
||||||
|
S::Future: 'static,
|
||||||
|
B: 'static,
|
||||||
|
{
|
||||||
|
type Response = S::Response;
|
||||||
|
type Error = S::Error;
|
||||||
|
type Transform = AiAuthMiddleware<S>;
|
||||||
|
type InitError = ();
|
||||||
|
type Future = Ready<Result<Self::Transform, Self::InitError>>;
|
||||||
|
|
||||||
|
fn new_transform(&self, service: S) -> Self::Future {
|
||||||
|
ok(AiAuthMiddleware { service })
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
pub struct AiAuthMiddleware<S> {
|
||||||
|
service: S,
|
||||||
|
}
|
||||||
|
|
||||||
|
impl<S, B> Service<ServiceRequest> for AiAuthMiddleware<S>
|
||||||
|
where
|
||||||
|
S: Service<ServiceRequest, Response = actix_web::dev::ServiceResponse<B>, Error = Error>,
|
||||||
|
S::Future: 'static,
|
||||||
|
B: 'static,
|
||||||
|
{
|
||||||
|
type Response = S::Response;
|
||||||
|
type Error = S::Error;
|
||||||
|
type Future = LocalBoxFuture<'static, Result<Self::Response, Self::Error>>;
|
||||||
|
|
||||||
|
fn poll_ready(&self, cx: &mut Context<'_>) -> Poll<Result<(), Self::Error>> {
|
||||||
|
self.service.poll_ready(cx)
|
||||||
|
}
|
||||||
|
|
||||||
|
fn call(&self, req: ServiceRequest) -> Self::Future {
|
||||||
|
// Extract API key from Authorization header
|
||||||
|
if let Some(auth_header) = req.headers().get("Authorization") {
|
||||||
|
if let Ok(auth_str) = auth_header.to_str() {
|
||||||
|
if auth_str.starts_with("Bearer ") {
|
||||||
|
let api_key = &auth_str[7..];
|
||||||
|
// Store in request extensions for later use
|
||||||
|
req.extensions_mut().insert(api_key.to_string());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
let fut = self.service.call(req);
|
||||||
|
Box::pin(async move {
|
||||||
|
fut.await
|
||||||
|
})
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Phase 5: Integration (Week 3)
|
||||||
|
|
||||||
|
### 5.1 Frontend Changes
|
||||||
|
|
||||||
|
```typescript
|
||||||
|
// Frontend API client
|
||||||
|
|
||||||
|
class AiClient {
|
||||||
|
private apiKey: string;
|
||||||
|
|
||||||
|
constructor(apiKey: string) {
|
||||||
|
this.apiKey = apiKey;
|
||||||
|
}
|
||||||
|
|
||||||
|
async chatCompletion(model: string, message: string): Promise<AIResponse> {
|
||||||
|
const response = await fetch('/api/v1/ai/chat', {
|
||||||
|
method: 'POST',
|
||||||
|
headers: {
|
||||||
|
'Authorization': `Bearer ${this.apiKey}`,
|
||||||
|
'Content-Type': 'application/json',
|
||||||
|
},
|
||||||
|
body: JSON.stringify({
|
||||||
|
model,
|
||||||
|
messages: [{ role: 'user', content: message }],
|
||||||
|
request_type: 'help',
|
||||||
|
}),
|
||||||
|
});
|
||||||
|
|
||||||
|
// Check for credit/rate limit headers
|
||||||
|
const creditsRemaining = response.headers.get('X-Credits-Remaining');
|
||||||
|
const rateLimitRemaining = response.headers.get('X-RateLimit-Remaining');
|
||||||
|
|
||||||
|
if (response.status === 402) {
|
||||||
|
throw new Error('Insufficient credits. Please upgrade.');
|
||||||
|
}
|
||||||
|
|
||||||
|
if (response.status === 429) {
|
||||||
|
throw new Error('Rate limit exceeded. Please slow down.');
|
||||||
|
}
|
||||||
|
|
||||||
|
return response.json();
|
||||||
|
}
|
||||||
|
|
||||||
|
async getUsageStats(): Promise<UsageStats> {
|
||||||
|
const response = await fetch('/api/v1/ai/usage', {
|
||||||
|
headers: {
|
||||||
|
'Authorization': `Bearer ${this.jwtToken}`, // JWT for authenticated endpoints
|
||||||
|
},
|
||||||
|
});
|
||||||
|
return response.json();
|
||||||
|
}
|
||||||
|
}
|
||||||
|
```
|
||||||
|
|
||||||
|
## Phase 6: Testing & Deployment (Week 3-4)
|
||||||
|
|
||||||
|
### 6.1 Testing Checklist
|
||||||
|
|
||||||
|
- [ ] Unit tests for API key generation
|
||||||
|
- [ ] Unit tests for usage tracking
|
||||||
|
- [ ] Integration tests for rate limiting
|
||||||
|
- [ ] Load tests for concurrent requests
|
||||||
|
- [ ] Security tests (key validation, SQL injection)
|
||||||
|
|
||||||
|
### 6.2 Migration Steps
|
||||||
|
|
||||||
|
```bash
|
||||||
|
# 1. Backup database
|
||||||
|
pg_dump $DATABASE_URL > backup_pre_ai_plans.sql
|
||||||
|
|
||||||
|
# 2. Run migrations
|
||||||
|
sqlx migrate run
|
||||||
|
|
||||||
|
# 3. Seed default plans
|
||||||
|
psql $DATABASE_URL < seed_plans.sql
|
||||||
|
|
||||||
|
# 4. Deploy new backend version
|
||||||
|
cargo build --release
|
||||||
|
|
||||||
|
# 5. Generate API keys for existing users
|
||||||
|
./scripts/migrate_existing_users.sh
|
||||||
|
|
||||||
|
# 6. Verify deployment
|
||||||
|
./scripts/verify_ai_plans.sh
|
||||||
|
```
|
||||||
|
|
||||||
|
## Timeline
|
||||||
|
|
||||||
|
| Week | Phase | Deliverables |
|
||||||
|
|------|-------|--------------|
|
||||||
|
| Week 1 | Database + Core Services | Tables, models, services |
|
||||||
|
| Week 2 | API + Middleware | Endpoints, auth, rate limiting |
|
||||||
|
| Week 3 | Integration + Frontend | React components, testing |
|
||||||
|
| Week 4 | Testing + Deployment | Load tests, monitoring, docs |
|
||||||
|
|
||||||
|
## Cost Estimation
|
||||||
|
|
||||||
|
| Component | Cost |
|
||||||
|
|-----------|------|
|
||||||
|
| Database storage | ~$5/mo (10GB) |
|
||||||
|
| Compute (tracking) | Minimal (async) |
|
||||||
|
| **Total additional** | **~$5/mo** |
|
||||||
|
|
||||||
|
## Next Steps
|
||||||
|
|
||||||
|
1. **Review this plan** - Any changes needed?
|
||||||
|
2. **Approve database schema** - Are all fields needed?
|
||||||
|
3. **Set priority** - Which features are must-have for MVP?
|
||||||
|
|
||||||
|
Ready to start Phase 1?
|
||||||
186
docs/PROJECT_STATUS_SUMMARY.md
Normal file
186
docs/PROJECT_STATUS_SUMMARY.md
Normal file
|
|
@ -0,0 +1,186 @@
|
||||||
|
# Project Status Summary - Nxtgauge Infrastructure
|
||||||
|
|
||||||
|
## Last Updated
|
||||||
|
2026-06-15
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## What Has Been Completed
|
||||||
|
|
||||||
|
### 1. LiteLLM AI Gateway Deployment ✅
|
||||||
|
- **Status**: Fully deployed and working
|
||||||
|
- **Namespace**: `nxtgauge-ai`
|
||||||
|
- **Service**: `litellm.nxtgauge-ai.svc.cluster.local:4000`
|
||||||
|
- **Ingress**: `https://llm.nxtgauge.com` (TLS via cert-manager)
|
||||||
|
- **Models configured**:
|
||||||
|
- `askash-fast` → `qwen3:4b` (fast, 2.5GB)
|
||||||
|
- `askash-main` → `qwen3:8b` (powerful, 5.2GB)
|
||||||
|
- Plus 10 specialized model aliases for different use cases
|
||||||
|
- **Security**: API key required via `LITELLM_MASTER_KEY`
|
||||||
|
- **Ollama**: Remains internal-only, NOT exposed publicly
|
||||||
|
- **Files created**:
|
||||||
|
- `apps/litellm/base/configmap.yaml`
|
||||||
|
- `apps/litellm/base/deployment.yaml`
|
||||||
|
- `apps/litellm/base/service.yaml`
|
||||||
|
- `apps/litellm/base/ingress.yaml`
|
||||||
|
- `apps/litellm/base/secret.yaml`
|
||||||
|
- `apps/litellm/base/ratelimit.yaml`
|
||||||
|
- `apps/litellm/base/kustomization.yaml`
|
||||||
|
- `apps/litellm/README.md`
|
||||||
|
- `apps/litellm/OPENCODE_CONNECT.md`
|
||||||
|
|
||||||
|
### 2. Kubernetes Cluster Expansion ✅
|
||||||
|
- Added 4th worker node: **nxtgauge-4 / Ramaris** (89.167.0.148)
|
||||||
|
- All nodes labeled with character names:
|
||||||
|
- `nxtgauge-1` → **Rimuru** (control plane)
|
||||||
|
- `nxtgauge-2` → **Veldora** (worker)
|
||||||
|
- `nxtgauge-3` → **Diablo** (worker)
|
||||||
|
- `nxtgauge-4` → **Ramaris** (worker)
|
||||||
|
- All nodes Ready and schedulable
|
||||||
|
|
||||||
|
### 3. GHCR Registry Authentication ✅
|
||||||
|
- Created `ghcr-regcred` secret in `nxtgauge` and `nxtgauge-ai` namespaces
|
||||||
|
- Updated `registries.yaml` on all K3s nodes to authenticate with GHCR
|
||||||
|
- Restarted K3s services on all nodes
|
||||||
|
- All deployments now pulling images successfully from GitHub Container Registry
|
||||||
|
|
||||||
|
### 4. Latest Code Deployed via GitHub Actions + Flux ✅
|
||||||
|
All four repositories are live with their latest commits:
|
||||||
|
|
||||||
|
| Repository | Branch | Commit | Status |
|
||||||
|
|------------|--------|--------|--------|
|
||||||
|
| nxtgauge-frontend-solid | high-performance | `3b8f75d` feat: add AI usage widget to user dashboard | ✅ Deployed |
|
||||||
|
| nxtgauge-admin-solid | high-performance | `f511a3c` feat: add AI management page to admin panel | ✅ Deployed |
|
||||||
|
| nxtgauge-ai-assistant | main | `4505d89` feat: add Ask Ash AI assistant implementation | ✅ Deployed |
|
||||||
|
| nxtgauge-backend-rust | high-performance | `ba63736` feat: add AI management endpoints and LiteLLM support | ✅ Deployed |
|
||||||
|
|
||||||
|
- Flux synced to latest gitops commit `c5b32538`
|
||||||
|
- 22/22 deployments ready
|
||||||
|
|
||||||
|
### 5. Ollama Models Downloaded ✅
|
||||||
|
- `gemma3:270m` (291 MB) - original model
|
||||||
|
- `qwen3:4b` (2.5 GB) - fast/general use
|
||||||
|
- `qwen3:8b` (5.2 GB) - powerful/long-form generation
|
||||||
|
|
||||||
|
### 6. OpenCode Configuration ✅
|
||||||
|
- Updated OpenCode binary from `1.14.20` → `1.17.7`
|
||||||
|
- Reverted config back to use **Ollama Cloud** with **Kimi K2.7**
|
||||||
|
- Config file: `~/.config/opencode/opencode.jsonc`
|
||||||
|
|
||||||
|
### 7. AI Plans Implementation Plan ✅
|
||||||
|
- Created detailed implementation plan:
|
||||||
|
- `docs/AI_PLANS_IMPLEMENTATION_PLAN.md`
|
||||||
|
- `docs/AI_PLANS_FINAL_PLAN.md`
|
||||||
|
- Plan covers:
|
||||||
|
- Database schema
|
||||||
|
- Plan tiers (Free/Pro/Business/Enterprise)
|
||||||
|
- API key generation
|
||||||
|
- Usage tracking
|
||||||
|
- Rate limiting
|
||||||
|
- Model access control
|
||||||
|
- Backend services architecture
|
||||||
|
- API endpoints
|
||||||
|
- Frontend integration
|
||||||
|
- Deployment steps
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## What We Are Stuck At / Blocked On
|
||||||
|
|
||||||
|
### 1. PostgreSQL for LiteLLM (Optional Advanced Tracking)
|
||||||
|
- **Status**: Attempted but not critical
|
||||||
|
- **Issue**: Tried to deploy PostgreSQL in `nxtgauge-ai` namespace for LiteLLM's built-in virtual key tracking, but it failed to schedule on the new node due to Longhorn CSI driver not being available on `nxtgauge-4`
|
||||||
|
- **Impact**: LOW - This is not required. We are building our own API key/usage tracking system instead.
|
||||||
|
- **Decision**: Skip LiteLLM-native virtual keys. Use custom backend implementation.
|
||||||
|
|
||||||
|
### 2. AI Plans Implementation ✅
|
||||||
|
- **Status**: Core backend implementation complete
|
||||||
|
- **Completed**: Database migration, models, repositories, core services, LiteLLM integration, endpoint wiring, admin endpoints, cron jobs, AI credit packages, auto-apply/auto-request log endpoints
|
||||||
|
- **Remaining**: Kubernetes env-var wiring (`AI_CREDIT_ADMIN_TOKEN`), real payment gateway integration, automated integration tests
|
||||||
|
- **Next step**: Apply migrations in target environment and configure Kubernetes secrets
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Current System Health
|
||||||
|
|
||||||
|
| Component | Status |
|
||||||
|
|-----------|--------|
|
||||||
|
| Kubernetes cluster | ✅ 4 nodes Ready |
|
||||||
|
| Flux GitOps | ✅ Synced |
|
||||||
|
| All 22 deployments | ✅ Running |
|
||||||
|
| LiteLLM gateway | ✅ Running |
|
||||||
|
| Ollama | ✅ Running |
|
||||||
|
| AI Assistant | ✅ Running |
|
||||||
|
| Frontend | ✅ Running |
|
||||||
|
| Admin panel | ✅ Running |
|
||||||
|
| 19 backend rust services | ✅ Running |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Decisions Made
|
||||||
|
|
||||||
|
1. ✅ Using **GitHub Container Registry (GHCR)** instead of `registry.nxtgauge.com`
|
||||||
|
2. ✅ Using **GitHub Actions + Flux** instead of Forgejo/Gitea
|
||||||
|
3. ✅ Using **custom API key/usage tracking** instead of LiteLLM Enterprise
|
||||||
|
4. ✅ Using **qwen3:4b** as main fast model and **qwen3:8b** for long-form/power tasks
|
||||||
|
5. ✅ Keeping **Ollama internal-only**, exposing only LiteLLM
|
||||||
|
6. ✅ OpenCode using **Ollama Cloud Kimi K2.7**
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Next Recommended Actions
|
||||||
|
|
||||||
|
1. **Run database migrations** for AI plans and credit packages (`cargo run -p db-migrate` or `sqlx migrate run`)
|
||||||
|
2. **Configure `AI_CREDIT_ADMIN_TOKEN`** secret for payments service to credit AI credits via users admin endpoint
|
||||||
|
3. **Add `LITELLM_BASE_URL` env var** for users service (default already points to cluster local service)
|
||||||
|
4. **Credit purchase flow** — integrate real payment gateway with `/api/admin/ai/users/{id}/credits`
|
||||||
|
5. **Add monitoring/alerting** for AI usage and credit thresholds
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Key Files and Commands
|
||||||
|
|
||||||
|
### Useful Commands
|
||||||
|
```bash
|
||||||
|
# Check cluster nodes
|
||||||
|
kubectl get nodes -o custom-columns='NAME:.metadata.name,CHARACTER:.metadata.labels.node-name,ROLE:.metadata.labels.node-role\.kubernetes\.io/worker,STATUS:.status.conditions[-1].type,IP:.status.addresses[0].address'
|
||||||
|
|
||||||
|
# Check deployments
|
||||||
|
kubectl get deployments -n nxtgauge
|
||||||
|
|
||||||
|
# Check AI namespace
|
||||||
|
kubectl get pods -n nxtgauge-ai
|
||||||
|
|
||||||
|
# Check LiteLLM models
|
||||||
|
curl http://litellm.nxtgauge-ai.svc.cluster.local:4000/v1/models \
|
||||||
|
-H "Authorization: Bearer sk-litellm-prod-1c66d63e701c32cd85922a62fd2e087469486a9b7a34d950423a8726d0aceec9"
|
||||||
|
|
||||||
|
# Check Flux status
|
||||||
|
flux get kustomizations --all-namespaces
|
||||||
|
|
||||||
|
# Get LiteLLM master key
|
||||||
|
kubectl get secret litellm-secrets -n nxtgauge-ai -o jsonpath='{.data.LITELLM_MASTER_KEY}' | base64 -d
|
||||||
|
```
|
||||||
|
|
||||||
|
### Important Paths
|
||||||
|
- GitOps repo: `/home/ashwin/nxtgauge-gitops`
|
||||||
|
- Frontend repo: `/home/ashwin/nxtgauge-projects/nxtgauge-frontend-solid`
|
||||||
|
- Backend repo: `/home/ashwin/nxtgauge-projects/nxtgauge-backend-rust`
|
||||||
|
- Admin repo: `/home/ashwin/nxtgauge-projects/nxtgauge-admin-solid`
|
||||||
|
- AI Assistant repo: `/home/ashwin/nxtgauge-projects/nxtgauge-ai-assistant`
|
||||||
|
- AI plans plan: `/home/ashwin/nxtgauge-gitops/docs/AI_PLANS_FINAL_IMPLEMENTATION.md`
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Blockers Requiring User Input
|
||||||
|
|
||||||
|
None currently. Core AI plans implementation is complete on the backend.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## Notes
|
||||||
|
|
||||||
|
- The PostgreSQL deployment attempt left a `postgres-pvc` in `nxtgauge-ai` namespace. It can be cleaned up safely since we are not using it.
|
||||||
|
- No git commits have been made during recent infrastructure changes unless explicitly requested.
|
||||||
|
- All changes were applied directly to Kubernetes and config files.
|
||||||
|
- Local `.opencode` plugin was updated, but OpenCode binary upgrade was done via built-in `opencode upgrade` command.
|
||||||
|
|
@ -1,5 +1,5 @@
|
||||||
#!/bin/bash
|
#!/bin/bash
|
||||||
# Build and push all nxtgauge backend services
|
# Build and push all missing nxtgauge backend services
|
||||||
|
|
||||||
set -e
|
set -e
|
||||||
|
|
||||||
|
|
@ -7,6 +7,10 @@ REGISTRY="registry.nxtgauge.com"
|
||||||
REGISTRY_USER="admin"
|
REGISTRY_USER="admin"
|
||||||
REGISTRY_PASS="Ashwin@2026"
|
REGISTRY_PASS="Ashwin@2026"
|
||||||
|
|
||||||
|
echo "=================================="
|
||||||
|
echo "Building Nxtgauge Services"
|
||||||
|
echo "=================================="
|
||||||
|
|
||||||
# Login to registry
|
# Login to registry
|
||||||
echo "Logging into registry..."
|
echo "Logging into registry..."
|
||||||
echo "$REGISTRY_PASS" | docker login $REGISTRY -u $REGISTRY_USER --password-stdin
|
echo "$REGISTRY_PASS" | docker login $REGISTRY -u $REGISTRY_USER --password-stdin
|
||||||
|
|
@ -26,7 +30,6 @@ SERVICES=(
|
||||||
"developers"
|
"developers"
|
||||||
"employees"
|
"employees"
|
||||||
"fitness-trainers"
|
"fitness-trainers"
|
||||||
"gateway"
|
|
||||||
"graphic-designers"
|
"graphic-designers"
|
||||||
"job-seekers"
|
"job-seekers"
|
||||||
"jobs"
|
"jobs"
|
||||||
|
|
@ -37,7 +40,6 @@ SERVICES=(
|
||||||
"social-media-managers"
|
"social-media-managers"
|
||||||
"tutors"
|
"tutors"
|
||||||
"ugc-content-creators"
|
"ugc-content-creators"
|
||||||
"users"
|
|
||||||
"video-editors"
|
"video-editors"
|
||||||
)
|
)
|
||||||
|
|
||||||
|
|
@ -48,25 +50,19 @@ for service in "${SERVICES[@]}"; do
|
||||||
echo "Building $service..."
|
echo "Building $service..."
|
||||||
echo "=================================="
|
echo "=================================="
|
||||||
|
|
||||||
# Get binary name (convert dashes to underscores for Rust naming)
|
|
||||||
bin_name=$(echo "$service" | tr '-' '_')
|
|
||||||
|
|
||||||
# Check if Dockerfile exists
|
|
||||||
if [ ! -f "apps/$service/Dockerfile" ]; then
|
|
||||||
echo "Dockerfile not found for $service, skipping..."
|
|
||||||
continue
|
|
||||||
fi
|
|
||||||
|
|
||||||
# Build using the service's Dockerfile
|
# Build using the service's Dockerfile
|
||||||
docker build -f "apps/$service/Dockerfile" \
|
docker build -f "apps/$service/Dockerfile" \
|
||||||
-t "$REGISTRY/nxtgauge-rust-$service:$SHA" \
|
-t "$REGISTRY/nxtgauge-rust-$service:$SHA" \
|
||||||
-t "$REGISTRY/nxtgauge-rust-$service:latest" \
|
-t "$REGISTRY/nxtgauge-rust-$service:latest" \
|
||||||
.
|
. 2>&1 || {
|
||||||
|
echo "WARNING: Failed to build $service, continuing..."
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
# Push images
|
# Push images
|
||||||
echo "Pushing $service:$SHA..."
|
echo "Pushing $service:$SHA..."
|
||||||
docker push "$REGISTRY/nxtgauge-rust-$service:$SHA"
|
docker push "$REGISTRY/nxtgauge-rust-$service:$SHA" 2>&1 || echo "WARNING: Failed to push $service:$SHA"
|
||||||
docker push "$REGISTRY/nxtgauge-rust-$service:latest"
|
docker push "$REGISTRY/nxtgauge-rust-$service:latest" 2>&1 || echo "WARNING: Failed to push $service:latest"
|
||||||
|
|
||||||
echo "$service built and pushed successfully!"
|
echo "$service built and pushed successfully!"
|
||||||
done
|
done
|
||||||
|
|
|
||||||
77
scripts/build-from-binaries.sh
Normal file
77
scripts/build-from-binaries.sh
Normal file
|
|
@ -0,0 +1,77 @@
|
||||||
|
#!/bin/bash
|
||||||
|
# Build and push service images using pre-compiled binaries
|
||||||
|
|
||||||
|
set -e
|
||||||
|
|
||||||
|
REGISTRY="registry.nxtgauge.com"
|
||||||
|
|
||||||
|
cd /home/ashwin/nxtgauge-projects/nxtgauge-backend-rust
|
||||||
|
|
||||||
|
SERVICES=(
|
||||||
|
"catering_services"
|
||||||
|
"companies"
|
||||||
|
"cron"
|
||||||
|
"customers"
|
||||||
|
"developers"
|
||||||
|
"employees"
|
||||||
|
"fitness_trainers"
|
||||||
|
"gateway"
|
||||||
|
"graphic_designers"
|
||||||
|
"job_seekers"
|
||||||
|
"jobs"
|
||||||
|
"leads"
|
||||||
|
"makeup_artists"
|
||||||
|
"payments"
|
||||||
|
"photographers"
|
||||||
|
"social_media_managers"
|
||||||
|
"tutors"
|
||||||
|
"ugc_content_creators"
|
||||||
|
"users"
|
||||||
|
"video_editors"
|
||||||
|
)
|
||||||
|
|
||||||
|
for svc in "${SERVICES[@]}"; do
|
||||||
|
echo ""
|
||||||
|
echo "=== Building $svc ==="
|
||||||
|
|
||||||
|
# Convert to hyphenated name for image
|
||||||
|
img_name=$(echo "$svc" | tr '_' '-')
|
||||||
|
|
||||||
|
# Check if binary exists
|
||||||
|
if [ ! -f "target/release/$svc" ]; then
|
||||||
|
echo "Building $svc binary..."
|
||||||
|
cargo build --release --bin "$svc" 2>&1 || {
|
||||||
|
echo "ERROR: Failed to build $svc"
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
fi
|
||||||
|
|
||||||
|
# Create temp directory for build
|
||||||
|
tmpdir=$(mktemp -d)
|
||||||
|
cp "target/release/$svc" "$tmpdir/service"
|
||||||
|
|
||||||
|
# Build minimal image
|
||||||
|
cat > "$tmpdir/Dockerfile" << 'EOF'
|
||||||
|
FROM scratch
|
||||||
|
COPY service /app/service
|
||||||
|
COPY --from=alpine:latest /etc/ssl/certs/ca-certificates.crt /etc/ssl/certs/
|
||||||
|
USER 65532:65532
|
||||||
|
EXPOSE 8000
|
||||||
|
ENTRYPOINT ["/app/service"]
|
||||||
|
EOF
|
||||||
|
|
||||||
|
# Build and push
|
||||||
|
docker build -t "$REGISTRY/nxtgauge-rust-$img_name:latest" "$tmpdir" 2>&1 || {
|
||||||
|
echo "ERROR: Failed to build image for $svc"
|
||||||
|
rm -rf "$tmpdir"
|
||||||
|
continue
|
||||||
|
}
|
||||||
|
|
||||||
|
docker push "$REGISTRY/nxtgauge-rust-$img_name:latest" 2>&1 || echo "ERROR: Failed to push $svc"
|
||||||
|
|
||||||
|
rm -rf "$tmpdir"
|
||||||
|
echo "✓ $svc pushed"
|
||||||
|
done
|
||||||
|
|
||||||
|
echo ""
|
||||||
|
echo "All services built and pushed!"
|
||||||
Loading…
Add table
Reference in a new issue