Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 5b5a9ecbc3 | |||
| 932e889ee8 | |||
| 48ca873fe2 |
@@ -0,0 +1,81 @@
|
|||||||
|
name: Rebuild Docker images
|
||||||
|
|
||||||
|
on:
|
||||||
|
workflow_dispatch:
|
||||||
|
|
||||||
|
concurrency:
|
||||||
|
group: rebuild-docker-images
|
||||||
|
cancel-in-progress: false
|
||||||
|
|
||||||
|
jobs:
|
||||||
|
rebuild-and-push:
|
||||||
|
runs-on: ubuntu-latest
|
||||||
|
env:
|
||||||
|
REGISTRY: docker.dcglab.co.uk
|
||||||
|
IMAGE_BASE: docker.dcglab.co.uk/public/kb
|
||||||
|
REGISTRY_USERNAME: ${{ secrets.DOCKER_DCGLAB_CI_USERNAME }}
|
||||||
|
REGISTRY_PASSWORD: ${{ secrets.DOCKER_DCGLAB_CI_PASSWORD }}
|
||||||
|
steps:
|
||||||
|
- name: Check out repository
|
||||||
|
uses: actions/checkout@v4
|
||||||
|
|
||||||
|
- name: Log in to registry
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
test -n "$REGISTRY_USERNAME" || { echo "DOCKER_DCGLAB_CI_USERNAME is not available" >&2; exit 1; }
|
||||||
|
test -n "$REGISTRY_PASSWORD" || { echo "DOCKER_DCGLAB_CI_PASSWORD is not available" >&2; exit 1; }
|
||||||
|
printf '%s' "$REGISTRY_PASSWORD" | docker login "$REGISTRY" --username "$REGISTRY_USERNAME" --password-stdin
|
||||||
|
|
||||||
|
- name: Build all images from scratch
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
version="$(tr -d '[:space:]' < engine/VERSION)"
|
||||||
|
|
||||||
|
docker build --pull --no-cache --provenance=false --sbom=false \
|
||||||
|
-t "$IMAGE_BASE/engine:v${version}-nvidia" \
|
||||||
|
-t "$IMAGE_BASE/engine:latest-nvidia" \
|
||||||
|
-f engine/Dockerfile.nvidia engine
|
||||||
|
|
||||||
|
docker build --pull --no-cache --provenance=false --sbom=false \
|
||||||
|
-t "$IMAGE_BASE/engine:v${version}-cpu" \
|
||||||
|
-t "$IMAGE_BASE/engine:latest-cpu" \
|
||||||
|
-f engine/Dockerfile.cpu engine
|
||||||
|
|
||||||
|
docker build --pull --no-cache --provenance=false --sbom=false \
|
||||||
|
-t "$IMAGE_BASE/mcp:v${version}" \
|
||||||
|
-t "$IMAGE_BASE/mcp:latest" \
|
||||||
|
-f mcp/Dockerfile mcp
|
||||||
|
|
||||||
|
- name: Push and verify all tags
|
||||||
|
shell: bash
|
||||||
|
run: |
|
||||||
|
set -euo pipefail
|
||||||
|
version="$(tr -d '[:space:]' < engine/VERSION)"
|
||||||
|
images=(
|
||||||
|
"$IMAGE_BASE/engine:v${version}-nvidia"
|
||||||
|
"$IMAGE_BASE/engine:latest-nvidia"
|
||||||
|
"$IMAGE_BASE/engine:v${version}-cpu"
|
||||||
|
"$IMAGE_BASE/engine:latest-cpu"
|
||||||
|
"$IMAGE_BASE/mcp:v${version}"
|
||||||
|
"$IMAGE_BASE/mcp:latest"
|
||||||
|
)
|
||||||
|
|
||||||
|
push_image() {
|
||||||
|
local image="$1"
|
||||||
|
local attempt
|
||||||
|
for attempt in 1 2 3 4 5; do
|
||||||
|
docker push "$image" && return 0
|
||||||
|
if [[ "$attempt" -eq 5 ]]; then
|
||||||
|
echo "Failed to push $image after $attempt attempts" >&2
|
||||||
|
return 1
|
||||||
|
fi
|
||||||
|
sleep 10
|
||||||
|
done
|
||||||
|
}
|
||||||
|
|
||||||
|
for image in "${images[@]}"; do
|
||||||
|
push_image "$image"
|
||||||
|
docker manifest inspect "$image" >/dev/null
|
||||||
|
done
|
||||||
+5
-1
@@ -90,11 +90,15 @@ curl http://localhost:8000/api/v1/status | jq .version
|
|||||||
|
|
||||||
### Docker images
|
### Docker images
|
||||||
|
|
||||||
Images are pushed to `docker.dcglab.co.uk/kb/engine` with tags:
|
Images are pushed to `docker.dcglab.co.uk/public/kb/engine` with tags:
|
||||||
|
|
||||||
- `engine-v2.0.6-nvidia` / `engine-v2.0.6-cpu` — versioned
|
- `engine-v2.0.6-nvidia` / `engine-v2.0.6-cpu` — versioned
|
||||||
- `latest-nvidia` / `latest-cpu` — latest release
|
- `latest-nvidia` / `latest-cpu` — latest release
|
||||||
|
|
||||||
|
The release script authenticates to the registry using the
|
||||||
|
`DOCKER_DCGLAB_CI_USERNAME` and `DOCKER_DCGLAB_CI_PASSWORD` environment
|
||||||
|
variables.
|
||||||
|
|
||||||
Override the registry and org via environment variables:
|
Override the registry and org via environment variables:
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
|
|||||||
@@ -20,7 +20,7 @@ docker run -d --name kb-mcp \
|
|||||||
-e KB_API_KEY=your-engine-key \
|
-e KB_API_KEY=your-engine-key \
|
||||||
-e KB_MCP_API_KEY=your-agent-key \
|
-e KB_MCP_API_KEY=your-agent-key \
|
||||||
--restart unless-stopped \
|
--restart unless-stopped \
|
||||||
docker.dcglab.co.uk/kb/mcp:latest
|
docker.dcglab.co.uk/public/kb/mcp:latest
|
||||||
```
|
```
|
||||||
|
|
||||||
## MCP tools
|
## MCP tools
|
||||||
|
|||||||
@@ -33,7 +33,7 @@ docker run -d --name kb-engine \
|
|||||||
-e KB_DEVICE=auto \
|
-e KB_DEVICE=auto \
|
||||||
-e KB_API_KEY=your-secret-key \
|
-e KB_API_KEY=your-secret-key \
|
||||||
--restart unless-stopped \
|
--restart unless-stopped \
|
||||||
docker.dcglab.co.uk/kb/engine:latest-nvidia
|
docker.dcglab.co.uk/public/kb/engine:latest-nvidia
|
||||||
|
|
||||||
# CPU only (no GPU required — smaller image)
|
# CPU only (no GPU required — smaller image)
|
||||||
docker run -d --name kb-engine \
|
docker run -d --name kb-engine \
|
||||||
@@ -42,7 +42,7 @@ docker run -d --name kb-engine \
|
|||||||
-e KB_MODEL=all-MiniLM-L6-v2 \
|
-e KB_MODEL=all-MiniLM-L6-v2 \
|
||||||
-e KB_API_KEY=your-secret-key \
|
-e KB_API_KEY=your-secret-key \
|
||||||
--restart unless-stopped \
|
--restart unless-stopped \
|
||||||
docker.dcglab.co.uk/kb/engine:latest-cpu
|
docker.dcglab.co.uk/public/kb/engine:latest-cpu
|
||||||
```
|
```
|
||||||
|
|
||||||
Or use a compose file from the repo:
|
Or use a compose file from the repo:
|
||||||
@@ -111,10 +111,12 @@ Override via environment variables (`KB_ENGINE_URL`, `KB_API_KEY`) or CLI flags
|
|||||||
# Add notes
|
# Add notes
|
||||||
kb addnote "Always restart nginx after config changes"
|
kb addnote "Always restart nginx after config changes"
|
||||||
kb addnote "Server room is building 3, floor 2" --tags ops
|
kb addnote "Server room is building 3, floor 2" --tags ops
|
||||||
|
kb addnote "Deploy checklist" --wait
|
||||||
|
|
||||||
# Add files (async — uploads and exits immediately)
|
# Add files (async by default; --wait blocks until ingestion finishes)
|
||||||
kb addfile ~/docs/manual.pdf --tags admin
|
kb addfile ~/docs/manual.pdf --tags admin
|
||||||
kb addfile ~/notes/ --recursive
|
kb addfile ~/notes/ --recursive
|
||||||
|
kb addfile ~/docs/manual.pdf --wait
|
||||||
|
|
||||||
# Check ingestion progress
|
# Check ingestion progress
|
||||||
kb jobs
|
kb jobs
|
||||||
@@ -122,13 +124,16 @@ kb jobs
|
|||||||
# Search
|
# Search
|
||||||
kb search "how to install git"
|
kb search "how to install git"
|
||||||
kb search "deploy process" --tags ops --type pdf
|
kb search "deploy process" --tags ops --type pdf
|
||||||
|
kb find "vehicle handbook" --type pdf
|
||||||
|
|
||||||
# Update a note in place
|
# Update a note in place
|
||||||
kb updatenote 42 "revised note content"
|
kb updatenote 42 "revised note content"
|
||||||
|
|
||||||
# Manage
|
# Manage
|
||||||
kb list
|
kb list
|
||||||
kb info 1
|
kb list --title handbook
|
||||||
|
kb list --filename M38T_PHEV
|
||||||
|
kb info 1 --no-chunks
|
||||||
kb tags
|
kb tags
|
||||||
kb tag 1 --add important
|
kb tag 1 --add important
|
||||||
kb export 1 -o manual.pdf # download original file
|
kb export 1 -o manual.pdf # download original file
|
||||||
@@ -160,6 +165,7 @@ The engine is configured via environment variables (set in the compose file or v
|
|||||||
| `KB_INGEST_DEVICE` | `auto` | Docling layout detection device: `auto`, `cpu`, or `cuda` |
|
| `KB_INGEST_DEVICE` | `auto` | Docling layout detection device: `auto`, `cpu`, or `cuda` |
|
||||||
| `KB_API_KEY` | (none) | Optional Bearer token for API authentication |
|
| `KB_API_KEY` | (none) | Optional Bearer token for API authentication |
|
||||||
| `KB_SEARCH_THRESHOLD` | `0.01` | Minimum score for search results (filters noise) |
|
| `KB_SEARCH_THRESHOLD` | `0.01` | Minimum score for search results (filters noise) |
|
||||||
|
| `KB_MIN_CHUNK_ALNUM` | `3` | Minimum alphanumeric characters retained in Docling PDF/DOCX/HTML chunks (`0` disables) |
|
||||||
| `KB_RERANK_ENABLED` | `false` (`true` in nvidia compose) | Load a cross-encoder and rerank hybrid results server-side |
|
| `KB_RERANK_ENABLED` | `false` (`true` in nvidia compose) | Load a cross-encoder and rerank hybrid results server-side |
|
||||||
| `KB_RERANKER_MODEL` | `BAAI/bge-reranker-v2-m3` | Cross-encoder model for reranking |
|
| `KB_RERANKER_MODEL` | `BAAI/bge-reranker-v2-m3` | Cross-encoder model for reranking |
|
||||||
| `KB_RERANK_CANDIDATES` | `40` | Hybrid candidates scored by the reranker per query |
|
| `KB_RERANK_CANDIDATES` | `40` | Hybrid candidates scored by the reranker per query |
|
||||||
@@ -169,6 +175,21 @@ The engine is configured via environment variables (set in the compose file or v
|
|||||||
| `HF_HUB_OFFLINE` | (none) | Set to `1` to prevent model downloads (use cached only) |
|
| `HF_HUB_OFFLINE` | (none) | Set to `1` to prevent model downloads (use cached only) |
|
||||||
| `KB_DATA_PATH` | `./data` | Host path for bind mount (compose variable, not used by engine) |
|
| `KB_DATA_PATH` | `./data` | Host path for bind mount (compose variable, not used by engine) |
|
||||||
|
|
||||||
|
### Repairing legacy note titles
|
||||||
|
|
||||||
|
Notes created by older clients may have a synthetic `<uuid>_note.note` title.
|
||||||
|
The maintenance command previews only unambiguous matches by default:
|
||||||
|
|
||||||
|
```bash
|
||||||
|
cd engine
|
||||||
|
python -m kb.maintenance.backfill_note_titles
|
||||||
|
python -m kb.maintenance.backfill_note_titles --apply
|
||||||
|
```
|
||||||
|
|
||||||
|
The apply mode reloads the configured embedding model and refreshes each
|
||||||
|
affected note's title, enriched full-text content, and vector embedding. Back
|
||||||
|
up the data directory before running it against production.
|
||||||
|
|
||||||
## Data portability
|
## Data portability
|
||||||
|
|
||||||
The data directory contains everything: SQLite database, model cache, and staging files. To migrate between hosts:
|
The data directory contains everything: SQLite database, model cache, and staging files. To migrate between hosts:
|
||||||
|
|||||||
+48
-4
@@ -8,6 +8,7 @@ import (
|
|||||||
"path/filepath"
|
"path/filepath"
|
||||||
"sort"
|
"sort"
|
||||||
"strings"
|
"strings"
|
||||||
|
"time"
|
||||||
|
|
||||||
"github.com/kb-search/kb/internal/api"
|
"github.com/kb-search/kb/internal/api"
|
||||||
"github.com/kb-search/kb/internal/output"
|
"github.com/kb-search/kb/internal/output"
|
||||||
@@ -54,12 +55,16 @@ var addfileCmd = &cobra.Command{
|
|||||||
func init() {
|
func init() {
|
||||||
addfileCmd.Flags().String("tags", "", "tags (comma-separated)")
|
addfileCmd.Flags().String("tags", "", "tags (comma-separated)")
|
||||||
addfileCmd.Flags().BoolP("recursive", "r", false, "recursively add directory contents")
|
addfileCmd.Flags().BoolP("recursive", "r", false, "recursively add directory contents")
|
||||||
|
addfileCmd.Flags().Bool("wait", false, "wait for ingestion to finish")
|
||||||
|
addfileCmd.Flags().Duration("wait-timeout", 10*time.Minute, "maximum time to wait per ingestion job")
|
||||||
rootCmd.AddCommand(addfileCmd)
|
rootCmd.AddCommand(addfileCmd)
|
||||||
}
|
}
|
||||||
|
|
||||||
func runAddfile(cmd *cobra.Command, args []string) error {
|
func runAddfile(cmd *cobra.Command, args []string) error {
|
||||||
tags, _ := cmd.Flags().GetString("tags")
|
tags, _ := cmd.Flags().GetString("tags")
|
||||||
recursive, _ := cmd.Flags().GetBool("recursive")
|
recursive, _ := cmd.Flags().GetBool("recursive")
|
||||||
|
wait, _ := cmd.Flags().GetBool("wait")
|
||||||
|
timeout, _ := cmd.Flags().GetDuration("wait-timeout")
|
||||||
|
|
||||||
client := api.NewClient()
|
client := api.NewClient()
|
||||||
|
|
||||||
@@ -89,12 +94,25 @@ func runAddfile(cmd *cobra.Command, args []string) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
if output.IsJSON() {
|
if output.IsJSON() {
|
||||||
|
if !wait || result.Duplicate {
|
||||||
output.PrintJSON([]interface{}{result.Raw})
|
output.PrintJSON([]interface{}{result.Raw})
|
||||||
|
}
|
||||||
} else if result.Duplicate {
|
} else if result.Duplicate {
|
||||||
fmt.Println(result.duplicateMsg())
|
fmt.Println(result.duplicateMsg())
|
||||||
} else {
|
} else {
|
||||||
fmt.Printf("Queued: %s\n", filepath.Base(path))
|
fmt.Printf("Queued: %s\n", filepath.Base(path))
|
||||||
}
|
}
|
||||||
|
if wait && !result.Duplicate {
|
||||||
|
job, err := waitForJob(client, int(result.JobID), timeout)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if output.IsJSON() {
|
||||||
|
output.PrintJSON([]interface{}{job})
|
||||||
|
} else {
|
||||||
|
fmt.Printf("Ingested: %s (doc ID: %d, chunks: %d)\n", filepath.Base(path), job.DocumentID, job.ChunkCount)
|
||||||
|
}
|
||||||
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -122,6 +140,7 @@ func runAddfile(cmd *cobra.Command, args []string) error {
|
|||||||
}
|
}
|
||||||
|
|
||||||
var results []interface{}
|
var results []interface{}
|
||||||
|
var pending []*uploadResult
|
||||||
queued := 0
|
queued := 0
|
||||||
duplicates := 0
|
duplicates := 0
|
||||||
for _, f := range files {
|
for _, f := range files {
|
||||||
@@ -130,7 +149,9 @@ func runAddfile(cmd *cobra.Command, args []string) error {
|
|||||||
fmt.Fprintf(os.Stderr, "Error uploading %s: %v\n", f, err)
|
fmt.Fprintf(os.Stderr, "Error uploading %s: %v\n", f, err)
|
||||||
continue
|
continue
|
||||||
}
|
}
|
||||||
|
if !wait || result.Duplicate {
|
||||||
results = append(results, result.Raw)
|
results = append(results, result.Raw)
|
||||||
|
}
|
||||||
if result.Duplicate {
|
if result.Duplicate {
|
||||||
duplicates++
|
duplicates++
|
||||||
if !output.IsJSON() {
|
if !output.IsJSON() {
|
||||||
@@ -138,11 +159,25 @@ func runAddfile(cmd *cobra.Command, args []string) error {
|
|||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
queued++
|
queued++
|
||||||
|
pending = append(pending, result)
|
||||||
if !output.IsJSON() {
|
if !output.IsJSON() {
|
||||||
fmt.Printf("Queued: %s\n", filepath.Base(f))
|
fmt.Printf("Queued: %s\n", filepath.Base(f))
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if wait {
|
||||||
|
for _, result := range pending {
|
||||||
|
job, err := waitForJob(client, int(result.JobID), timeout)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if output.IsJSON() {
|
||||||
|
results = append(results, job)
|
||||||
|
} else {
|
||||||
|
fmt.Printf("Ingested job %d (doc ID: %d, chunks: %d)\n", int(result.JobID), job.DocumentID, job.ChunkCount)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
if output.IsJSON() {
|
if output.IsJSON() {
|
||||||
output.PrintJSON(results)
|
output.PrintJSON(results)
|
||||||
@@ -203,10 +238,19 @@ func uploadFile(client *api.Client, path, tags string) (*uploadResult, error) {
|
|||||||
return nil, err
|
return nil, err
|
||||||
}
|
}
|
||||||
|
|
||||||
var result interface{}
|
var raw json.RawMessage
|
||||||
if err := api.DecodeJSON(resp, &result); err != nil {
|
if err := api.DecodeJSON(resp, &raw); err != nil {
|
||||||
return nil, fmt.Errorf("failed to decode response: %w", err)
|
return nil, fmt.Errorf("failed to decode response: %w", err)
|
||||||
}
|
}
|
||||||
return &uploadResult{Raw: result}, nil
|
var queued struct {
|
||||||
|
JobID float64 `json:"job_id"`
|
||||||
|
}
|
||||||
|
if err := json.Unmarshal(raw, &queued); err != nil {
|
||||||
|
return nil, fmt.Errorf("failed to decode queued job: %w", err)
|
||||||
|
}
|
||||||
|
var result interface{}
|
||||||
|
if err := json.Unmarshal(raw, &result); err != nil {
|
||||||
|
return nil, fmt.Errorf("failed to decode queued response: %w", err)
|
||||||
|
}
|
||||||
|
return &uploadResult{Raw: result, JobID: queued.JobID}, nil
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
+25
-3
@@ -4,6 +4,7 @@ import (
|
|||||||
"fmt"
|
"fmt"
|
||||||
"net/http"
|
"net/http"
|
||||||
"os"
|
"os"
|
||||||
|
"time"
|
||||||
|
|
||||||
"github.com/kb-search/kb/internal/api"
|
"github.com/kb-search/kb/internal/api"
|
||||||
"github.com/kb-search/kb/internal/output"
|
"github.com/kb-search/kb/internal/output"
|
||||||
@@ -27,16 +28,20 @@ var addnoteCmd = &cobra.Command{
|
|||||||
|
|
||||||
func init() {
|
func init() {
|
||||||
addnoteCmd.Flags().String("tags", "", "tags (comma-separated)")
|
addnoteCmd.Flags().String("tags", "", "tags (comma-separated)")
|
||||||
|
addnoteCmd.Flags().Bool("wait", false, "wait for ingestion to finish")
|
||||||
|
addnoteCmd.Flags().Duration("wait-timeout", 10*time.Minute, "maximum time to wait for ingestion")
|
||||||
rootCmd.AddCommand(addnoteCmd)
|
rootCmd.AddCommand(addnoteCmd)
|
||||||
}
|
}
|
||||||
|
|
||||||
func runAddnote(cmd *cobra.Command, args []string) error {
|
func runAddnote(cmd *cobra.Command, args []string) error {
|
||||||
tags, _ := cmd.Flags().GetString("tags")
|
tags, _ := cmd.Flags().GetString("tags")
|
||||||
|
wait, _ := cmd.Flags().GetBool("wait")
|
||||||
|
timeout, _ := cmd.Flags().GetDuration("wait-timeout")
|
||||||
client := api.NewClient()
|
client := api.NewClient()
|
||||||
return submitNote(client, args[0], tags)
|
return submitNote(client, args[0], tags, wait, timeout)
|
||||||
}
|
}
|
||||||
|
|
||||||
func submitNote(client *api.Client, note, tags string) error {
|
func submitNote(client *api.Client, note, tags string, wait bool, timeout time.Duration) error {
|
||||||
fields := map[string]string{
|
fields := map[string]string{
|
||||||
"note": note,
|
"note": note,
|
||||||
}
|
}
|
||||||
@@ -74,15 +79,32 @@ func submitNote(client *api.Client, note, tags string) error {
|
|||||||
os.Exit(1)
|
os.Exit(1)
|
||||||
}
|
}
|
||||||
|
|
||||||
var result interface{}
|
var result struct {
|
||||||
|
JobID int `json:"job_id"`
|
||||||
|
Status string `json:"status"`
|
||||||
|
Filename string `json:"filename"`
|
||||||
|
}
|
||||||
if err := api.DecodeJSON(resp, &result); err != nil {
|
if err := api.DecodeJSON(resp, &result); err != nil {
|
||||||
return fmt.Errorf("failed to decode response: %w", err)
|
return fmt.Errorf("failed to decode response: %w", err)
|
||||||
}
|
}
|
||||||
|
|
||||||
if output.IsJSON() {
|
if output.IsJSON() {
|
||||||
|
if !wait {
|
||||||
output.PrintJSON(result)
|
output.PrintJSON(result)
|
||||||
|
}
|
||||||
} else {
|
} else {
|
||||||
fmt.Println("Queued: note")
|
fmt.Println("Queued: note")
|
||||||
}
|
}
|
||||||
|
if wait {
|
||||||
|
job, err := waitForJob(client, result.JobID, timeout)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if output.IsJSON() {
|
||||||
|
output.PrintJSON(job)
|
||||||
|
} else {
|
||||||
|
fmt.Printf("Ingested: note (doc ID: %d, chunks: %d)\n", job.DocumentID, job.ChunkCount)
|
||||||
|
}
|
||||||
|
}
|
||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -14,21 +14,25 @@ var examplesCmd = &cobra.Command{
|
|||||||
fmt.Print(`Add notes:
|
fmt.Print(`Add notes:
|
||||||
kb addnote "Remember to update DNS records"
|
kb addnote "Remember to update DNS records"
|
||||||
kb addnote "Server room is building 3" --tags ops
|
kb addnote "Server room is building 3" --tags ops
|
||||||
|
kb addnote "Deploy checklist" --wait
|
||||||
|
|
||||||
Add files:
|
Add files:
|
||||||
kb addfile report.pdf
|
kb addfile report.pdf
|
||||||
kb addfile ~/docs/ --recursive --tags reference
|
kb addfile ~/docs/ --recursive --tags reference
|
||||||
|
kb addfile report.pdf --wait
|
||||||
|
|
||||||
Search:
|
Search:
|
||||||
kb search "how to restart nginx"
|
kb search "how to restart nginx"
|
||||||
kb search "deploy" --tags ops --top 5
|
kb search "deploy" --tags ops --top 5
|
||||||
|
kb find "quarterly report" --type pdf
|
||||||
|
|
||||||
Update notes:
|
Update notes:
|
||||||
kb updatenote 42 "revised note content"
|
kb updatenote 42 "revised note content"
|
||||||
|
|
||||||
Manage documents:
|
Manage documents:
|
||||||
kb list --type pdf
|
kb list --type pdf
|
||||||
kb info 3
|
kb list --filename report.pdf
|
||||||
|
kb info 3 --no-chunks
|
||||||
kb tag 3 --add important,ops
|
kb tag 3 --add important,ops
|
||||||
kb remove 3 --yes
|
kb remove 3 --yes
|
||||||
`)
|
`)
|
||||||
|
|||||||
@@ -0,0 +1,85 @@
|
|||||||
|
package cmd
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
|
||||||
|
"github.com/kb-search/kb/internal/api"
|
||||||
|
"github.com/kb-search/kb/internal/output"
|
||||||
|
"github.com/spf13/cobra"
|
||||||
|
)
|
||||||
|
|
||||||
|
var findCmd = &cobra.Command{
|
||||||
|
Use: "find <query>",
|
||||||
|
Short: "Find documents by their indexed content",
|
||||||
|
Args: cobra.ExactArgs(1),
|
||||||
|
RunE: runFind,
|
||||||
|
}
|
||||||
|
|
||||||
|
func init() {
|
||||||
|
findCmd.Flags().IntP("top", "n", 10, "number of documents to return")
|
||||||
|
findCmd.Flags().String("tags", "", "filter by tags (comma-separated)")
|
||||||
|
findCmd.Flags().String("type", "", "filter by document type")
|
||||||
|
rootCmd.AddCommand(findCmd)
|
||||||
|
}
|
||||||
|
|
||||||
|
func runFind(cmd *cobra.Command, args []string) error {
|
||||||
|
top, _ := cmd.Flags().GetInt("top")
|
||||||
|
tags, _ := cmd.Flags().GetString("tags")
|
||||||
|
docType, _ := cmd.Flags().GetString("type")
|
||||||
|
body := map[string]interface{}{"query": args[0], "top": top}
|
||||||
|
if tags != "" {
|
||||||
|
body["tags"] = splitTags(tags)
|
||||||
|
}
|
||||||
|
if docType != "" {
|
||||||
|
body["doc_type"] = docType
|
||||||
|
}
|
||||||
|
|
||||||
|
resp, err := api.NewClient().Post("/api/v1/documents/find", body)
|
||||||
|
if err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if err := api.CheckError(resp); err != nil {
|
||||||
|
return err
|
||||||
|
}
|
||||||
|
if output.IsJSON() {
|
||||||
|
var raw interface{}
|
||||||
|
if err := api.DecodeJSON(resp, &raw); err != nil {
|
||||||
|
return fmt.Errorf("failed to decode response: %w", err)
|
||||||
|
}
|
||||||
|
output.PrintJSON(raw)
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
|
||||||
|
var docs []struct {
|
||||||
|
DocumentID int `json:"document_id"`
|
||||||
|
Title string `json:"title"`
|
||||||
|
Filename string `json:"original_filename"`
|
||||||
|
Type string `json:"doc_type"`
|
||||||
|
Score float64 `json:"score"`
|
||||||
|
HitCount int `json:"hit_count"`
|
||||||
|
TopChunk string `json:"top_chunk"`
|
||||||
|
}
|
||||||
|
if err := api.DecodeJSON(resp, &docs); err != nil {
|
||||||
|
return fmt.Errorf("failed to decode response: %w", err)
|
||||||
|
}
|
||||||
|
if len(docs) == 0 {
|
||||||
|
fmt.Println("No documents found.")
|
||||||
|
return nil
|
||||||
|
}
|
||||||
|
for i, doc := range docs {
|
||||||
|
preview := doc.TopChunk
|
||||||
|
if len(preview) > 200 {
|
||||||
|
preview = preview[:200] + "..."
|
||||||
|
}
|
||||||
|
fmt.Printf("\n%d. [%.4f] %s (doc:%d, hits:%d)\n", i+1, doc.Score, doc.Title, doc.DocumentID, doc.HitCount)
|
||||||
|
if doc.Filename != "" {
|
||||||
|
fmt.Printf(" Filename: %s\n", doc.Filename)
|
||||||
|
}
|
||||||
|
if doc.Type != "" {
|
||||||
|
fmt.Printf(" Type: %s\n", doc.Type)
|
||||||
|
}
|
||||||
|
fmt.Printf(" %s\n", preview)
|
||||||
|
}
|
||||||
|
fmt.Println()
|
||||||
|
return nil
|
||||||
|
}
|
||||||
+9
-2
@@ -17,12 +17,18 @@ var infoCmd = &cobra.Command{
|
|||||||
}
|
}
|
||||||
|
|
||||||
func init() {
|
func init() {
|
||||||
|
infoCmd.Flags().Bool("no-chunks", false, "return document metadata without chunk details")
|
||||||
rootCmd.AddCommand(infoCmd)
|
rootCmd.AddCommand(infoCmd)
|
||||||
}
|
}
|
||||||
|
|
||||||
func runInfo(cmd *cobra.Command, args []string) error {
|
func runInfo(cmd *cobra.Command, args []string) error {
|
||||||
client := api.NewClient()
|
client := api.NewClient()
|
||||||
resp, err := client.Get("/api/v1/documents/" + args[0])
|
noChunks, _ := cmd.Flags().GetBool("no-chunks")
|
||||||
|
path := "/api/v1/documents/" + args[0]
|
||||||
|
if noChunks {
|
||||||
|
path += "?include_chunks=false"
|
||||||
|
}
|
||||||
|
resp, err := client.Get(path)
|
||||||
if err != nil {
|
if err != nil {
|
||||||
fmt.Fprintln(os.Stderr, err)
|
fmt.Fprintln(os.Stderr, err)
|
||||||
os.Exit(1)
|
os.Exit(1)
|
||||||
@@ -48,6 +54,7 @@ func runInfo(cmd *cobra.Command, args []string) error {
|
|||||||
Tags []string `json:"tags"`
|
Tags []string `json:"tags"`
|
||||||
CreatedAt string `json:"created_at"`
|
CreatedAt string `json:"created_at"`
|
||||||
UpdatedAt string `json:"updated_at"`
|
UpdatedAt string `json:"updated_at"`
|
||||||
|
ChunkCount int `json:"chunk_count"`
|
||||||
Chunks []struct {
|
Chunks []struct {
|
||||||
ID int `json:"id"`
|
ID int `json:"id"`
|
||||||
Page interface{} `json:"page"`
|
Page interface{} `json:"page"`
|
||||||
@@ -65,7 +72,7 @@ func runInfo(cmd *cobra.Command, args []string) error {
|
|||||||
{"Tags", joinStrings(doc.Tags)},
|
{"Tags", joinStrings(doc.Tags)},
|
||||||
{"Created", doc.CreatedAt},
|
{"Created", doc.CreatedAt},
|
||||||
{"Updated", doc.UpdatedAt},
|
{"Updated", doc.UpdatedAt},
|
||||||
{"Chunks", fmt.Sprintf("%d", len(doc.Chunks))},
|
{"Chunks", fmt.Sprintf("%d", doc.ChunkCount)},
|
||||||
}
|
}
|
||||||
output.PrintKeyValue(pairs)
|
output.PrintKeyValue(pairs)
|
||||||
|
|
||||||
|
|||||||
+14
-2
@@ -13,18 +13,23 @@ import (
|
|||||||
var listCmd = &cobra.Command{
|
var listCmd = &cobra.Command{
|
||||||
Use: "list",
|
Use: "list",
|
||||||
Short: "List documents in the knowledge base",
|
Short: "List documents in the knowledge base",
|
||||||
|
Args: cobra.NoArgs,
|
||||||
RunE: runList,
|
RunE: runList,
|
||||||
}
|
}
|
||||||
|
|
||||||
func init() {
|
func init() {
|
||||||
listCmd.Flags().String("type", "", "filter by document type")
|
listCmd.Flags().String("type", "", "filter by document type")
|
||||||
listCmd.Flags().String("tags", "", "filter by tags (comma-separated)")
|
listCmd.Flags().String("tags", "", "filter by tags (comma-separated)")
|
||||||
|
listCmd.Flags().String("title", "", "filter by title substring")
|
||||||
|
listCmd.Flags().String("filename", "", "filter by original filename substring")
|
||||||
rootCmd.AddCommand(listCmd)
|
rootCmd.AddCommand(listCmd)
|
||||||
}
|
}
|
||||||
|
|
||||||
func runList(cmd *cobra.Command, args []string) error {
|
func runList(cmd *cobra.Command, args []string) error {
|
||||||
docType, _ := cmd.Flags().GetString("type")
|
docType, _ := cmd.Flags().GetString("type")
|
||||||
tags, _ := cmd.Flags().GetString("tags")
|
tags, _ := cmd.Flags().GetString("tags")
|
||||||
|
title, _ := cmd.Flags().GetString("title")
|
||||||
|
filename, _ := cmd.Flags().GetString("filename")
|
||||||
|
|
||||||
params := url.Values{}
|
params := url.Values{}
|
||||||
if docType != "" {
|
if docType != "" {
|
||||||
@@ -33,6 +38,12 @@ func runList(cmd *cobra.Command, args []string) error {
|
|||||||
if tags != "" {
|
if tags != "" {
|
||||||
params.Set("tags", tags)
|
params.Set("tags", tags)
|
||||||
}
|
}
|
||||||
|
if title != "" {
|
||||||
|
params.Set("title", title)
|
||||||
|
}
|
||||||
|
if filename != "" {
|
||||||
|
params.Set("filename", filename)
|
||||||
|
}
|
||||||
|
|
||||||
path := "/api/v1/documents"
|
path := "/api/v1/documents"
|
||||||
if len(params) > 0 {
|
if len(params) > 0 {
|
||||||
@@ -62,6 +73,7 @@ func runList(cmd *cobra.Command, args []string) error {
|
|||||||
var docs []struct {
|
var docs []struct {
|
||||||
ID int `json:"id"`
|
ID int `json:"id"`
|
||||||
Title string `json:"title"`
|
Title string `json:"title"`
|
||||||
|
Filename string `json:"original_filename"`
|
||||||
Type string `json:"doc_type"`
|
Type string `json:"doc_type"`
|
||||||
Tags []string `json:"tags"`
|
Tags []string `json:"tags"`
|
||||||
}
|
}
|
||||||
@@ -74,10 +86,10 @@ func runList(cmd *cobra.Command, args []string) error {
|
|||||||
return nil
|
return nil
|
||||||
}
|
}
|
||||||
|
|
||||||
headers := []string{"ID", "TITLE", "TYPE", "TAGS"}
|
headers := []string{"ID", "TITLE", "FILENAME", "TYPE", "TAGS"}
|
||||||
var rows [][]string
|
var rows [][]string
|
||||||
for _, d := range docs {
|
for _, d := range docs {
|
||||||
rows = append(rows, []string{fmt.Sprintf("%d", d.ID), d.Title, d.Type, joinStrings(d.Tags)})
|
rows = append(rows, []string{fmt.Sprintf("%d", d.ID), d.Title, d.Filename, d.Type, joinStrings(d.Tags)})
|
||||||
}
|
}
|
||||||
output.PrintTable(headers, rows)
|
output.PrintTable(headers, rows)
|
||||||
return nil
|
return nil
|
||||||
|
|||||||
@@ -67,3 +67,16 @@ func TestAddnoteCmd_TooManyArgs_ReturnsError(t *testing.T) {
|
|||||||
t.Errorf("expected 'accepts 1 arg' error, got: %s", errMsg)
|
t.Errorf("expected 'accepts 1 arg' error, got: %s", errMsg)
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
func TestListCmd_PositionalArgReturnsError(t *testing.T) {
|
||||||
|
rootCmd.SetArgs([]string{"list", "ignored-title"})
|
||||||
|
|
||||||
|
err := rootCmd.Execute()
|
||||||
|
if err == nil {
|
||||||
|
t.Fatal("expected error for positional list argument, got nil")
|
||||||
|
}
|
||||||
|
if !strings.Contains(err.Error(), "unknown command") &&
|
||||||
|
!strings.Contains(err.Error(), "accepts 0 arg") {
|
||||||
|
t.Errorf("expected positional argument error, got: %s", err)
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|||||||
@@ -0,0 +1,43 @@
|
|||||||
|
package cmd
|
||||||
|
|
||||||
|
import (
|
||||||
|
"fmt"
|
||||||
|
"time"
|
||||||
|
|
||||||
|
"github.com/kb-search/kb/internal/api"
|
||||||
|
)
|
||||||
|
|
||||||
|
type jobStatus struct {
|
||||||
|
ID int `json:"id"`
|
||||||
|
Status string `json:"status"`
|
||||||
|
DocumentID int `json:"document_id"`
|
||||||
|
ChunkCount int `json:"chunk_count"`
|
||||||
|
Error string `json:"error"`
|
||||||
|
}
|
||||||
|
|
||||||
|
func waitForJob(client *api.Client, jobID int, timeout time.Duration) (*jobStatus, error) {
|
||||||
|
deadline := time.Now().Add(timeout)
|
||||||
|
for {
|
||||||
|
resp, err := client.Get(fmt.Sprintf("/api/v1/jobs/%d", jobID))
|
||||||
|
if err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
if err := api.CheckError(resp); err != nil {
|
||||||
|
return nil, err
|
||||||
|
}
|
||||||
|
var job jobStatus
|
||||||
|
if err := api.DecodeJSON(resp, &job); err != nil {
|
||||||
|
return nil, fmt.Errorf("failed to decode job status: %w", err)
|
||||||
|
}
|
||||||
|
switch job.Status {
|
||||||
|
case "done", "skipped":
|
||||||
|
return &job, nil
|
||||||
|
case "failed":
|
||||||
|
return nil, fmt.Errorf("ingestion job %d failed: %s", jobID, job.Error)
|
||||||
|
}
|
||||||
|
if time.Now().After(deadline) {
|
||||||
|
return nil, fmt.Errorf("timed out waiting for ingestion job %d", jobID)
|
||||||
|
}
|
||||||
|
time.Sleep(time.Second)
|
||||||
|
}
|
||||||
|
}
|
||||||
@@ -13,6 +13,7 @@ services:
|
|||||||
- KB_INGEST_DEVICE=cpu
|
- KB_INGEST_DEVICE=cpu
|
||||||
- KB_API_KEY=${KB_API_KEY:-}
|
- KB_API_KEY=${KB_API_KEY:-}
|
||||||
- KB_SEARCH_THRESHOLD=${KB_SEARCH_THRESHOLD:-0.01}
|
- KB_SEARCH_THRESHOLD=${KB_SEARCH_THRESHOLD:-0.01}
|
||||||
|
- KB_MIN_CHUNK_ALNUM=${KB_MIN_CHUNK_ALNUM:-3}
|
||||||
- HF_HUB_OFFLINE=${HF_HUB_OFFLINE:-}
|
- HF_HUB_OFFLINE=${HF_HUB_OFFLINE:-}
|
||||||
restart: unless-stopped
|
restart: unless-stopped
|
||||||
|
|
||||||
|
|||||||
@@ -21,6 +21,7 @@ services:
|
|||||||
- KB_INGEST_DEVICE=${KB_INGEST_DEVICE:-auto}
|
- KB_INGEST_DEVICE=${KB_INGEST_DEVICE:-auto}
|
||||||
- KB_API_KEY=${KB_API_KEY:-}
|
- KB_API_KEY=${KB_API_KEY:-}
|
||||||
- KB_SEARCH_THRESHOLD=${KB_SEARCH_THRESHOLD:-0.01}
|
- KB_SEARCH_THRESHOLD=${KB_SEARCH_THRESHOLD:-0.01}
|
||||||
|
- KB_MIN_CHUNK_ALNUM=${KB_MIN_CHUNK_ALNUM:-3}
|
||||||
- KB_RERANK_ENABLED=${KB_RERANK_ENABLED:-true}
|
- KB_RERANK_ENABLED=${KB_RERANK_ENABLED:-true}
|
||||||
- KB_RERANKER_MODEL=${KB_RERANKER_MODEL:-BAAI/bge-reranker-v2-m3}
|
- KB_RERANKER_MODEL=${KB_RERANKER_MODEL:-BAAI/bge-reranker-v2-m3}
|
||||||
- KB_RERANK_CANDIDATES=${KB_RERANK_CANDIDATES:-40}
|
- KB_RERANK_CANDIDATES=${KB_RERANK_CANDIDATES:-40}
|
||||||
|
|||||||
@@ -24,6 +24,7 @@ class Config:
|
|||||||
self.reranker_model = os.environ.get("KB_RERANKER_MODEL", "BAAI/bge-reranker-v2-m3")
|
self.reranker_model = os.environ.get("KB_RERANKER_MODEL", "BAAI/bge-reranker-v2-m3")
|
||||||
self.rerank_candidates = int(os.environ.get("KB_RERANK_CANDIDATES", "40"))
|
self.rerank_candidates = int(os.environ.get("KB_RERANK_CANDIDATES", "40"))
|
||||||
self.bulk_safety_percent = int(os.environ.get("KB_BULK_SAFETY_PERCENT", "70"))
|
self.bulk_safety_percent = int(os.environ.get("KB_BULK_SAFETY_PERCENT", "70"))
|
||||||
|
self.min_chunk_alnum = int(os.environ.get("KB_MIN_CHUNK_ALNUM", "3"))
|
||||||
self.host = os.environ.get("KB_HOST", "0.0.0.0")
|
self.host = os.environ.get("KB_HOST", "0.0.0.0")
|
||||||
self.port = int(os.environ.get("KB_PORT", "8000"))
|
self.port = int(os.environ.get("KB_PORT", "8000"))
|
||||||
|
|
||||||
|
|||||||
@@ -25,6 +25,7 @@ from docling.document_converter import DocumentConverter, PdfFormatOption # noq
|
|||||||
from docling_core.transforms.chunker.hierarchical_chunker import ( # noqa: E402
|
from docling_core.transforms.chunker.hierarchical_chunker import ( # noqa: E402
|
||||||
HierarchicalChunker,
|
HierarchicalChunker,
|
||||||
)
|
)
|
||||||
|
from kb.ingest.quality import has_minimum_content
|
||||||
|
|
||||||
|
|
||||||
def _fixed_size_chunks(text: str, max_chars: int = 2000) -> list[str]:
|
def _fixed_size_chunks(text: str, max_chars: int = 2000) -> list[str]:
|
||||||
@@ -40,6 +41,7 @@ def _fixed_size_chunks(text: str, max_chars: int = 2000) -> list[str]:
|
|||||||
def chunk_document(
|
def chunk_document(
|
||||||
file_path: Path,
|
file_path: Path,
|
||||||
ingest_device: str = "cpu",
|
ingest_device: str = "cpu",
|
||||||
|
min_chunk_alnum: int = 3,
|
||||||
) -> list[dict]:
|
) -> list[dict]:
|
||||||
"""Convert and chunk a PDF/DOCX/HTML document using Docling.
|
"""Convert and chunk a PDF/DOCX/HTML document using Docling.
|
||||||
|
|
||||||
@@ -71,7 +73,7 @@ def chunk_document(
|
|||||||
chunks: list[dict] = []
|
chunks: list[dict] = []
|
||||||
for idx, chunk in enumerate(raw_chunks):
|
for idx, chunk in enumerate(raw_chunks):
|
||||||
text = chunk.text.strip() if hasattr(chunk, "text") else str(chunk).strip()
|
text = chunk.text.strip() if hasattr(chunk, "text") else str(chunk).strip()
|
||||||
if not text:
|
if not text or not has_minimum_content(text, min_chunk_alnum):
|
||||||
continue
|
continue
|
||||||
|
|
||||||
metadata: dict = {}
|
metadata: dict = {}
|
||||||
@@ -98,6 +100,8 @@ def chunk_document(
|
|||||||
if not full_text and hasattr(doc, "text"):
|
if not full_text and hasattr(doc, "text"):
|
||||||
full_text = doc.text
|
full_text = doc.text
|
||||||
for idx, piece in enumerate(_fixed_size_chunks(full_text)):
|
for idx, piece in enumerate(_fixed_size_chunks(full_text)):
|
||||||
|
if not has_minimum_content(piece, min_chunk_alnum):
|
||||||
|
continue
|
||||||
chunks.append({
|
chunks.append({
|
||||||
"text": piece,
|
"text": piece,
|
||||||
"chunk_index": idx,
|
"chunk_index": idx,
|
||||||
|
|||||||
@@ -0,0 +1,14 @@
|
|||||||
|
"""Small, conservative ingestion-quality checks."""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
|
||||||
|
def has_minimum_content(text: str, min_alnum: int = 3) -> bool:
|
||||||
|
"""Return whether text contains enough letters/numbers to be searchable.
|
||||||
|
|
||||||
|
Counting alphanumeric characters avoids indexing OCR fragments consisting
|
||||||
|
only of punctuation or one-character labels while retaining short IDs.
|
||||||
|
"""
|
||||||
|
if min_alnum <= 0:
|
||||||
|
return True
|
||||||
|
return sum(character.isalnum() for character in text) >= min_alnum
|
||||||
@@ -0,0 +1 @@
|
|||||||
|
"""Explicit maintenance commands for kb-engine data."""
|
||||||
@@ -0,0 +1,115 @@
|
|||||||
|
"""Repair notes whose title was generated from the old ``note`` fallback.
|
||||||
|
|
||||||
|
Preview changes by default::
|
||||||
|
|
||||||
|
python -m kb.maintenance.backfill_note_titles
|
||||||
|
|
||||||
|
Apply them, including refreshed FTS text and embeddings::
|
||||||
|
|
||||||
|
python -m kb.maintenance.backfill_note_titles --apply
|
||||||
|
"""
|
||||||
|
|
||||||
|
from __future__ import annotations
|
||||||
|
|
||||||
|
import argparse
|
||||||
|
import json
|
||||||
|
import re
|
||||||
|
import struct
|
||||||
|
|
||||||
|
from kb import database, embeddings
|
||||||
|
from kb.config import cfg
|
||||||
|
from kb.ingest.note import auto_title
|
||||||
|
|
||||||
|
|
||||||
|
_SYNTHETIC_NOTE = re.compile(
|
||||||
|
r"^[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}_note\.note$",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
def find_repairs(conn) -> list[dict]:
|
||||||
|
"""Return unambiguous synthetic note titles and their derived replacements."""
|
||||||
|
rows = conn.execute(
|
||||||
|
"""
|
||||||
|
SELECT d.id, d.title, d.original_filename, c.text, c.metadata
|
||||||
|
FROM documents d
|
||||||
|
JOIN chunks c ON c.document_id = d.id AND c.chunk_index = 0
|
||||||
|
WHERE d.doc_type = 'note'
|
||||||
|
ORDER BY d.id
|
||||||
|
"""
|
||||||
|
).fetchall()
|
||||||
|
repairs = []
|
||||||
|
for row in rows:
|
||||||
|
if not _SYNTHETIC_NOTE.fullmatch(row["title"] or ""):
|
||||||
|
continue
|
||||||
|
title = auto_title(row["text"] or "")
|
||||||
|
if not title:
|
||||||
|
continue
|
||||||
|
repairs.append({
|
||||||
|
"document_id": row["id"],
|
||||||
|
"old_title": row["title"],
|
||||||
|
"new_title": title,
|
||||||
|
"original_filename": row["original_filename"],
|
||||||
|
})
|
||||||
|
return repairs
|
||||||
|
|
||||||
|
|
||||||
|
def apply_repairs(conn, repairs: list[dict]) -> None:
|
||||||
|
"""Update titles, enriched text, FTS, and vectors for selected notes."""
|
||||||
|
for repair in repairs:
|
||||||
|
doc_id = repair["document_id"]
|
||||||
|
title = repair["new_title"]
|
||||||
|
chunks = conn.execute(
|
||||||
|
"SELECT id, text, metadata FROM chunks WHERE document_id = ? ORDER BY chunk_index",
|
||||||
|
(doc_id,),
|
||||||
|
).fetchall()
|
||||||
|
enriched = []
|
||||||
|
for chunk in chunks:
|
||||||
|
metadata = json.loads(chunk["metadata"] or "{}")
|
||||||
|
enriched.append(database.build_enriched_text(title, chunk["text"], metadata))
|
||||||
|
vectors = embeddings.embed_texts(enriched)
|
||||||
|
|
||||||
|
conn.execute(
|
||||||
|
"UPDATE documents SET title = ?, updated_at = current_timestamp WHERE id = ?",
|
||||||
|
(title, doc_id),
|
||||||
|
)
|
||||||
|
for chunk, text, vector in zip(chunks, enriched, vectors):
|
||||||
|
conn.execute(
|
||||||
|
"UPDATE chunks SET enriched_text = ? WHERE id = ?", (text, chunk["id"])
|
||||||
|
)
|
||||||
|
conn.execute("DELETE FROM chunks_vec WHERE chunk_id = ?", (chunk["id"],))
|
||||||
|
blob = struct.pack(f"{len(vector)}f", *vector)
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO chunks_vec(embedding, chunk_id) VALUES (?, ?)",
|
||||||
|
(blob, chunk["id"]),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
|
||||||
|
|
||||||
|
def main() -> None:
|
||||||
|
parser = argparse.ArgumentParser(description=__doc__)
|
||||||
|
parser.add_argument(
|
||||||
|
"--apply", action="store_true", help="apply repairs (the default is preview only)"
|
||||||
|
)
|
||||||
|
args = parser.parse_args()
|
||||||
|
|
||||||
|
conn = database.get_connection(cfg.db_path)
|
||||||
|
try:
|
||||||
|
repairs = find_repairs(conn)
|
||||||
|
for repair in repairs:
|
||||||
|
print(
|
||||||
|
f'{repair["document_id"]}: {repair["old_title"]!r} -> '
|
||||||
|
f'{repair["new_title"]!r}'
|
||||||
|
)
|
||||||
|
if not args.apply:
|
||||||
|
print(f"Previewed {len(repairs)} repair(s); rerun with --apply to update them.")
|
||||||
|
return
|
||||||
|
embeddings.load_model(cfg.model, cfg.device)
|
||||||
|
apply_repairs(conn, repairs)
|
||||||
|
print(f"Repaired {len(repairs)} note title(s).")
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
main()
|
||||||
@@ -7,11 +7,13 @@ from pathlib import Path
|
|||||||
from typing import Optional
|
from typing import Optional
|
||||||
|
|
||||||
from fastapi import HTTPException, Query
|
from fastapi import HTTPException, Query
|
||||||
|
from pydantic import BaseModel, Field
|
||||||
from fastapi.responses import FileResponse
|
from fastapi.responses import FileResponse
|
||||||
|
|
||||||
from main import app
|
from main import app
|
||||||
from kb.config import cfg
|
from kb.config import cfg
|
||||||
from kb.database import get_connection
|
from kb.database import get_connection
|
||||||
|
from kb.search import hybrid_search
|
||||||
|
|
||||||
logger = logging.getLogger("kb.routes.documents")
|
logger = logging.getLogger("kb.routes.documents")
|
||||||
|
|
||||||
@@ -20,11 +22,13 @@ logger = logging.getLogger("kb.routes.documents")
|
|||||||
async def list_documents(
|
async def list_documents(
|
||||||
type: Optional[str] = Query(None),
|
type: Optional[str] = Query(None),
|
||||||
tags: Optional[str] = Query(None),
|
tags: Optional[str] = Query(None),
|
||||||
|
title: Optional[str] = Query(None),
|
||||||
|
filename: Optional[str] = Query(None),
|
||||||
):
|
):
|
||||||
conn = get_connection(cfg.db_path)
|
conn = get_connection(cfg.db_path)
|
||||||
try:
|
try:
|
||||||
sql = """
|
sql = """
|
||||||
SELECT d.id, d.title, d.doc_type,
|
SELECT d.id, d.title, d.original_filename, d.source_path, d.doc_type,
|
||||||
(SELECT COUNT(*) FROM chunks c WHERE c.document_id = d.id) AS chunk_count,
|
(SELECT COUNT(*) FROM chunks c WHERE c.document_id = d.id) AS chunk_count,
|
||||||
d.created_at, d.updated_at
|
d.created_at, d.updated_at
|
||||||
FROM documents d
|
FROM documents d
|
||||||
@@ -37,6 +41,14 @@ async def list_documents(
|
|||||||
where.append("d.doc_type = ?")
|
where.append("d.doc_type = ?")
|
||||||
params.append(type)
|
params.append(type)
|
||||||
|
|
||||||
|
if title:
|
||||||
|
where.append("d.title LIKE ? COLLATE NOCASE")
|
||||||
|
params.append(f"%{title}%")
|
||||||
|
|
||||||
|
if filename:
|
||||||
|
where.append("d.original_filename LIKE ? COLLATE NOCASE")
|
||||||
|
params.append(f"%{filename}%")
|
||||||
|
|
||||||
if tags:
|
if tags:
|
||||||
tag_list = [t.strip() for t in tags.split(",") if t.strip()]
|
tag_list = [t.strip() for t in tags.split(",") if t.strip()]
|
||||||
for i, tag in enumerate(tag_list):
|
for i, tag in enumerate(tag_list):
|
||||||
@@ -70,6 +82,8 @@ async def list_documents(
|
|||||||
results.append({
|
results.append({
|
||||||
"id": row["id"],
|
"id": row["id"],
|
||||||
"title": row["title"],
|
"title": row["title"],
|
||||||
|
"original_filename": row["original_filename"],
|
||||||
|
"source_path": row["source_path"],
|
||||||
"doc_type": row["doc_type"],
|
"doc_type": row["doc_type"],
|
||||||
"tags": [t["name"] for t in tag_rows],
|
"tags": [t["name"] for t in tag_rows],
|
||||||
"chunk_count": row["chunk_count"],
|
"chunk_count": row["chunk_count"],
|
||||||
@@ -82,8 +96,49 @@ async def list_documents(
|
|||||||
conn.close()
|
conn.close()
|
||||||
|
|
||||||
|
|
||||||
|
class DocumentFindRequest(BaseModel):
|
||||||
|
query: str
|
||||||
|
top: int = Field(default=10, ge=1, le=100)
|
||||||
|
tags: Optional[list[str]] = None
|
||||||
|
doc_type: Optional[str] = None
|
||||||
|
|
||||||
|
|
||||||
|
@app.post("/api/v1/documents/find")
|
||||||
|
async def find_documents(req: DocumentFindRequest):
|
||||||
|
"""Return document-level results aggregated from hybrid chunk search."""
|
||||||
|
conn = get_connection(cfg.db_path)
|
||||||
|
try:
|
||||||
|
search_result = hybrid_search(
|
||||||
|
conn,
|
||||||
|
req.query,
|
||||||
|
cfg,
|
||||||
|
top=max(req.top * 10, 50),
|
||||||
|
tags=req.tags,
|
||||||
|
doc_type=req.doc_type,
|
||||||
|
)
|
||||||
|
documents: dict[int, dict] = {}
|
||||||
|
for result in search_result["results"]:
|
||||||
|
doc_id = result["document_id"]
|
||||||
|
if doc_id not in documents:
|
||||||
|
documents[doc_id] = {
|
||||||
|
"document_id": doc_id,
|
||||||
|
"title": result["title"],
|
||||||
|
"doc_type": result["doc_type"],
|
||||||
|
"source_path": result["source_path"],
|
||||||
|
"original_filename": result["original_filename"],
|
||||||
|
"tags": result["tags"],
|
||||||
|
"score": result["score"],
|
||||||
|
"hit_count": 0,
|
||||||
|
"top_chunk": result["text"],
|
||||||
|
}
|
||||||
|
documents[doc_id]["hit_count"] += 1
|
||||||
|
return list(documents.values())[: req.top]
|
||||||
|
finally:
|
||||||
|
conn.close()
|
||||||
|
|
||||||
|
|
||||||
@app.get("/api/v1/documents/{doc_id}")
|
@app.get("/api/v1/documents/{doc_id}")
|
||||||
async def get_document(doc_id: int):
|
async def get_document(doc_id: int, include_chunks: bool = Query(True)):
|
||||||
conn = get_connection(cfg.db_path)
|
conn = get_connection(cfg.db_path)
|
||||||
try:
|
try:
|
||||||
doc = conn.execute(
|
doc = conn.execute(
|
||||||
@@ -92,6 +147,11 @@ async def get_document(doc_id: int):
|
|||||||
if not doc:
|
if not doc:
|
||||||
raise HTTPException(status_code=404, detail="Document not found.")
|
raise HTTPException(status_code=404, detail="Document not found.")
|
||||||
|
|
||||||
|
chunk_count = conn.execute(
|
||||||
|
"SELECT COUNT(*) AS n FROM chunks WHERE document_id = ?", (doc_id,)
|
||||||
|
).fetchone()["n"]
|
||||||
|
chunks = []
|
||||||
|
if include_chunks:
|
||||||
chunks = conn.execute(
|
chunks = conn.execute(
|
||||||
"SELECT * FROM chunks WHERE document_id = ? ORDER BY chunk_index",
|
"SELECT * FROM chunks WHERE document_id = ? ORDER BY chunk_index",
|
||||||
(doc_id,),
|
(doc_id,),
|
||||||
@@ -114,6 +174,7 @@ async def get_document(doc_id: int):
|
|||||||
**dict(doc),
|
**dict(doc),
|
||||||
"has_file": has_file,
|
"has_file": has_file,
|
||||||
"tags": [t["name"] for t in tag_rows],
|
"tags": [t["name"] for t in tag_rows],
|
||||||
|
"chunk_count": chunk_count,
|
||||||
"chunks": [dict(c) for c in chunks],
|
"chunks": [dict(c) for c in chunks],
|
||||||
}
|
}
|
||||||
finally:
|
finally:
|
||||||
|
|||||||
@@ -10,6 +10,7 @@ from fastapi.responses import JSONResponse
|
|||||||
from main import app
|
from main import app
|
||||||
from kb.config import cfg
|
from kb.config import cfg
|
||||||
from kb.database import get_connection, create_job, get_job, list_jobs, get_document_by_hash
|
from kb.database import get_connection, create_job, get_job, list_jobs, get_document_by_hash
|
||||||
|
from kb.ingest.note import auto_title
|
||||||
from kb.staging import stage_file, stage_note
|
from kb.staging import stage_file, stage_note
|
||||||
|
|
||||||
|
|
||||||
@@ -32,6 +33,7 @@ async def submit_job(
|
|||||||
content_hash = hashlib.sha256(content).hexdigest()
|
content_hash = hashlib.sha256(content).hexdigest()
|
||||||
filename = file.filename
|
filename = file.filename
|
||||||
else:
|
else:
|
||||||
|
title = title or auto_title(note) or "note"
|
||||||
content = note.encode("utf-8")
|
content = note.encode("utf-8")
|
||||||
content_hash = hashlib.sha256(content).hexdigest()
|
content_hash = hashlib.sha256(content).hexdigest()
|
||||||
filename = None
|
filename = None
|
||||||
@@ -48,7 +50,7 @@ async def submit_job(
|
|||||||
if file:
|
if file:
|
||||||
staging_path = stage_file(cfg.staging_dir, file.filename, content)
|
staging_path = stage_file(cfg.staging_dir, file.filename, content)
|
||||||
else:
|
else:
|
||||||
staging_path = stage_note(cfg.staging_dir, title or "note", note)
|
staging_path = stage_note(cfg.staging_dir, title, note)
|
||||||
filename = staging_path.name
|
filename = staging_path.name
|
||||||
|
|
||||||
tags_list = [t.strip() for t in tags.split(",") if t.strip()] if tags else []
|
tags_list = [t.strip() for t in tags.split(",") if t.strip()] if tags else []
|
||||||
|
|||||||
+2
-1
@@ -422,7 +422,7 @@ def _enrich(
|
|||||||
"""
|
"""
|
||||||
SELECT c.id, c.text, c.chunk_index, c.metadata AS chunk_meta,
|
SELECT c.id, c.text, c.chunk_index, c.metadata AS chunk_meta,
|
||||||
d.id AS doc_id, d.title, d.doc_type, d.source_path,
|
d.id AS doc_id, d.title, d.doc_type, d.source_path,
|
||||||
d.created_at
|
d.created_at, d.original_filename
|
||||||
FROM chunks c
|
FROM chunks c
|
||||||
JOIN documents d ON c.document_id = d.id
|
JOIN documents d ON c.document_id = d.id
|
||||||
WHERE c.id = ?
|
WHERE c.id = ?
|
||||||
@@ -456,6 +456,7 @@ def _enrich(
|
|||||||
"doc_type": row[6],
|
"doc_type": row[6],
|
||||||
"source_path": row[7],
|
"source_path": row[7],
|
||||||
"created_at": row[8],
|
"created_at": row[8],
|
||||||
|
"original_filename": row[9],
|
||||||
"tags": [t[0] for t in tag_rows],
|
"tags": [t[0] for t in tag_rows],
|
||||||
"tag_contexts": {t[0]: t[1] for t in tag_rows if t[1]},
|
"tag_contexts": {t[0]: t[1] for t in tag_rows if t[1]},
|
||||||
}
|
}
|
||||||
|
|||||||
+3
-1
@@ -113,7 +113,9 @@ def _process_job(job_row) -> tuple[str, int | None, int]:
|
|||||||
chunks = chunk_note(text)
|
chunks = chunk_note(text)
|
||||||
elif doc_type == "pdf":
|
elif doc_type == "pdf":
|
||||||
from kb.ingest.docling_pipeline import chunk_document
|
from kb.ingest.docling_pipeline import chunk_document
|
||||||
chunks = chunk_document(staged_path, cfg.ingest_device)
|
chunks = chunk_document(
|
||||||
|
staged_path, cfg.ingest_device, cfg.min_chunk_alnum
|
||||||
|
)
|
||||||
elif doc_type == "markdown":
|
elif doc_type == "markdown":
|
||||||
text = staged_path.read_text(encoding="utf-8")
|
text = staged_path.read_text(encoding="utf-8")
|
||||||
from kb.ingest.markdown import chunk_markdown
|
from kb.ingest.markdown import chunk_markdown
|
||||||
|
|||||||
@@ -0,0 +1,102 @@
|
|||||||
|
"""Focused tests for document metadata lookup and aggregation."""
|
||||||
|
|
||||||
|
import pytest
|
||||||
|
|
||||||
|
from kb import database
|
||||||
|
from kb.routes import documents
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.fixture
|
||||||
|
def document_db(tmp_path, monkeypatch):
|
||||||
|
db_path = tmp_path / "kb.db"
|
||||||
|
conn = database.get_connection(db_path)
|
||||||
|
database.init_schema(conn, 3)
|
||||||
|
conn.execute(
|
||||||
|
"""
|
||||||
|
INSERT INTO documents(title, source_path, content_hash, doc_type, original_filename)
|
||||||
|
VALUES ('Vehicle Guide', '/data/staging/random.pdf', 'hash', 'pdf', 'M38T_manual.pdf')
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
doc_id = conn.execute("SELECT id FROM documents").fetchone()["id"]
|
||||||
|
conn.execute(
|
||||||
|
"INSERT INTO chunks(document_id, chunk_index, text, enriched_text) VALUES (?, 0, 'body', 'body')",
|
||||||
|
(doc_id,),
|
||||||
|
)
|
||||||
|
conn.commit()
|
||||||
|
conn.close()
|
||||||
|
monkeypatch.setattr(documents.cfg, "data_dir", tmp_path)
|
||||||
|
return doc_id
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_list_filters_title_and_original_filename(document_db):
|
||||||
|
by_title = await documents.list_documents(
|
||||||
|
type=None, tags=None, title="vehicle", filename=None
|
||||||
|
)
|
||||||
|
by_filename = await documents.list_documents(
|
||||||
|
type=None, tags=None, title=None, filename="m38t"
|
||||||
|
)
|
||||||
|
|
||||||
|
assert [item["id"] for item in by_title] == [document_db]
|
||||||
|
assert by_filename[0]["original_filename"] == "M38T_manual.pdf"
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_info_can_omit_chunks_without_losing_count(document_db):
|
||||||
|
result = await documents.get_document(document_db, include_chunks=False)
|
||||||
|
|
||||||
|
assert result["chunk_count"] == 1
|
||||||
|
assert result["chunks"] == []
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.asyncio
|
||||||
|
async def test_find_aggregates_chunk_hits_by_document(monkeypatch):
|
||||||
|
class Connection:
|
||||||
|
def close(self):
|
||||||
|
pass
|
||||||
|
|
||||||
|
monkeypatch.setattr(documents, "get_connection", lambda _path: Connection())
|
||||||
|
monkeypatch.setattr(
|
||||||
|
documents,
|
||||||
|
"hybrid_search",
|
||||||
|
lambda *_args, **_kwargs: {
|
||||||
|
"results": [
|
||||||
|
{
|
||||||
|
"document_id": 7,
|
||||||
|
"title": "Guide",
|
||||||
|
"doc_type": "pdf",
|
||||||
|
"source_path": "/staging/file",
|
||||||
|
"original_filename": "guide.pdf",
|
||||||
|
"tags": [],
|
||||||
|
"score": 0.8,
|
||||||
|
"text": "best hit",
|
||||||
|
},
|
||||||
|
{
|
||||||
|
"document_id": 7,
|
||||||
|
"title": "Guide",
|
||||||
|
"doc_type": "pdf",
|
||||||
|
"source_path": "/staging/file",
|
||||||
|
"original_filename": "guide.pdf",
|
||||||
|
"tags": [],
|
||||||
|
"score": 0.7,
|
||||||
|
"text": "second hit",
|
||||||
|
},
|
||||||
|
]
|
||||||
|
},
|
||||||
|
)
|
||||||
|
|
||||||
|
result = await documents.find_documents(
|
||||||
|
documents.DocumentFindRequest(query="guide")
|
||||||
|
)
|
||||||
|
|
||||||
|
assert result == [{
|
||||||
|
"document_id": 7,
|
||||||
|
"title": "Guide",
|
||||||
|
"doc_type": "pdf",
|
||||||
|
"source_path": "/staging/file",
|
||||||
|
"original_filename": "guide.pdf",
|
||||||
|
"tags": [],
|
||||||
|
"score": 0.8,
|
||||||
|
"hit_count": 2,
|
||||||
|
"top_chunk": "best hit",
|
||||||
|
}]
|
||||||
@@ -0,0 +1,19 @@
|
|||||||
|
"""Tests for filtering noisy OCR fragments."""
|
||||||
|
|
||||||
|
from kb.ingest.quality import has_minimum_content
|
||||||
|
|
||||||
|
|
||||||
|
def test_short_ocr_fragments_are_rejected():
|
||||||
|
assert not has_minimum_content('"')
|
||||||
|
assert not has_minimum_content("B")
|
||||||
|
assert not has_minimum_content("12")
|
||||||
|
|
||||||
|
|
||||||
|
def test_short_identifiers_and_real_text_are_retained():
|
||||||
|
assert has_minimum_content("BID")
|
||||||
|
assert has_minimum_content("00:15")
|
||||||
|
assert has_minimum_content("Useful text")
|
||||||
|
|
||||||
|
|
||||||
|
def test_filter_can_be_disabled():
|
||||||
|
assert has_minimum_content("B", min_alnum=0)
|
||||||
@@ -0,0 +1,75 @@
|
|||||||
|
"""Tests for automatic and backfilled note titles."""
|
||||||
|
|
||||||
|
import sqlite3
|
||||||
|
|
||||||
|
from kb.ingest.note import auto_title
|
||||||
|
from kb.maintenance import backfill_note_titles
|
||||||
|
from kb.maintenance.backfill_note_titles import find_repairs
|
||||||
|
|
||||||
|
|
||||||
|
def test_auto_title_strips_markdown_and_limits_length():
|
||||||
|
assert auto_title("## Useful heading\nBody") == "Useful heading"
|
||||||
|
assert auto_title("x" * 100) == "x" * 80
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_repairs_only_selects_synthetic_note_titles():
|
||||||
|
conn = sqlite3.connect(":memory:")
|
||||||
|
conn.row_factory = sqlite3.Row
|
||||||
|
conn.executescript(
|
||||||
|
"""
|
||||||
|
CREATE TABLE documents (
|
||||||
|
id INTEGER PRIMARY KEY, title TEXT, original_filename TEXT, doc_type TEXT
|
||||||
|
);
|
||||||
|
CREATE TABLE chunks (
|
||||||
|
id INTEGER PRIMARY KEY, document_id INTEGER, chunk_index INTEGER,
|
||||||
|
text TEXT, metadata TEXT
|
||||||
|
);
|
||||||
|
INSERT INTO documents VALUES
|
||||||
|
(1, '7dec828a-1234-4567-89ab-123456789abc_note.note',
|
||||||
|
'7dec828a-1234-4567-89ab-123456789abc_note.note', 'note'),
|
||||||
|
(2, 'A deliberate title', 'named.note', 'note');
|
||||||
|
INSERT INTO chunks VALUES
|
||||||
|
(1, 1, 0, '# Derived title\nBody', '{}'),
|
||||||
|
(2, 2, 0, 'Must not replace', '{}');
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
|
||||||
|
assert find_repairs(conn) == [{
|
||||||
|
"document_id": 1,
|
||||||
|
"old_title": "7dec828a-1234-4567-89ab-123456789abc_note.note",
|
||||||
|
"new_title": "Derived title",
|
||||||
|
"original_filename": "7dec828a-1234-4567-89ab-123456789abc_note.note",
|
||||||
|
}]
|
||||||
|
|
||||||
|
|
||||||
|
def test_apply_repairs_refreshes_title_search_text_and_vector(monkeypatch):
|
||||||
|
conn = sqlite3.connect(":memory:")
|
||||||
|
conn.row_factory = sqlite3.Row
|
||||||
|
conn.executescript(
|
||||||
|
"""
|
||||||
|
CREATE TABLE documents (id INTEGER PRIMARY KEY, title TEXT, updated_at TEXT);
|
||||||
|
CREATE TABLE chunks (
|
||||||
|
id INTEGER PRIMARY KEY, document_id INTEGER, chunk_index INTEGER,
|
||||||
|
text TEXT, metadata TEXT, enriched_text TEXT
|
||||||
|
);
|
||||||
|
CREATE TABLE chunks_vec (embedding BLOB, chunk_id INTEGER);
|
||||||
|
INSERT INTO documents VALUES (1, 'old', NULL);
|
||||||
|
INSERT INTO chunks VALUES (3, 1, 0, 'New title\nBody', '{}', 'old text');
|
||||||
|
INSERT INTO chunks_vec VALUES (X'00', 3);
|
||||||
|
"""
|
||||||
|
)
|
||||||
|
monkeypatch.setattr(
|
||||||
|
backfill_note_titles.embeddings,
|
||||||
|
"embed_texts",
|
||||||
|
lambda texts: [[1.0, 2.0, 3.0] for _ in texts],
|
||||||
|
)
|
||||||
|
|
||||||
|
backfill_note_titles.apply_repairs(conn, [{
|
||||||
|
"document_id": 1, "new_title": "New title"
|
||||||
|
}])
|
||||||
|
|
||||||
|
assert conn.execute("SELECT title FROM documents").fetchone()[0] == "New title"
|
||||||
|
assert conn.execute("SELECT enriched_text FROM chunks").fetchone()[0] == (
|
||||||
|
"New title\n\nNew title\nBody"
|
||||||
|
)
|
||||||
|
assert len(conn.execute("SELECT embedding FROM chunks_vec").fetchone()[0]) == 12
|
||||||
@@ -1,5 +1,10 @@
|
|||||||
# kb — Next Steps
|
# kb — Next Steps
|
||||||
|
|
||||||
|
> Implementation status (2026-08-22): `document_id`, metadata-only info,
|
||||||
|
> title/filename filters, document-level `find`, and short Docling chunk
|
||||||
|
> filtering are implemented on `feature/tasks-5-15-completion`. OCR is already
|
||||||
|
> enabled through RapidOCR; further OCR tuning remains measurement-gated.
|
||||||
|
|
||||||
UX improvements to make documents easier to find and inspect, prompted by a session where searching for an uploaded PDF (`M38T_PHEV_RHD_OM_EN_UK_20251209.pdf`, doc id 2077, 1801 chunks) surfaced lots of chunk hits but no obvious path back to the original document.
|
UX improvements to make documents easier to find and inspect, prompted by a session where searching for an uploaded PDF (`M38T_PHEV_RHD_OM_EN_UK_20251209.pdf`, doc id 2077, 1801 chunks) surfaced lots of chunk hits but no obvious path back to the original document.
|
||||||
|
|
||||||
## Problems observed
|
## Problems observed
|
||||||
|
|||||||
@@ -65,7 +65,7 @@ The project SHALL provide Docker Compose files for single-command deployment. Co
|
|||||||
|
|
||||||
#### Scenario: Pre-built image deployment
|
#### Scenario: Pre-built image deployment
|
||||||
- **WHEN** an admin wants to use a pre-built engine image without building from source
|
- **WHEN** an admin wants to use a pre-built engine image without building from source
|
||||||
- **THEN** the engine release notes SHALL include the exact `docker pull` command with the versioned tag (e.g. `docker.dcglab.co.uk/kb/engine:engine-v2.1.0-nvidia`)
|
- **THEN** the engine release notes SHALL include the exact `docker pull` command with the versioned tag (e.g. `docker.dcglab.co.uk/public/kb/engine:engine-v2.1.0-nvidia`)
|
||||||
|
|
||||||
#### Scenario: MCP allowed hosts in Compose
|
#### Scenario: MCP allowed hosts in Compose
|
||||||
- **WHEN** the kb-mcp service is defined in a Compose file
|
- **WHEN** the kb-mcp service is defined in a Compose file
|
||||||
|
|||||||
+22
-1
@@ -20,7 +20,7 @@ VERSION_FILE="$ENGINE_DIR/VERSION"
|
|||||||
# attestation manifests, making the image an OCI image index. The Registry v2
|
# attestation manifests, making the image an OCI image index. The Registry v2
|
||||||
# host at docker.dcglab.co.uk rejects those with a 500 on manifest PUT.
|
# host at docker.dcglab.co.uk rejects those with a 500 on manifest PUT.
|
||||||
REGISTRY="${REGISTRY:-docker.dcglab.co.uk}"
|
REGISTRY="${REGISTRY:-docker.dcglab.co.uk}"
|
||||||
IMAGE_ORG="${IMAGE_ORG:-}"
|
IMAGE_ORG="${IMAGE_ORG:-public}"
|
||||||
IMAGE_BASE="${REGISTRY}${IMAGE_ORG:+/${IMAGE_ORG}}/kb"
|
IMAGE_BASE="${REGISTRY}${IMAGE_ORG:+/${IMAGE_ORG}}/kb"
|
||||||
|
|
||||||
# Push retries — see push_image() below
|
# Push retries — see push_image() below
|
||||||
@@ -106,6 +106,16 @@ run() {
|
|||||||
fi
|
fi
|
||||||
}
|
}
|
||||||
|
|
||||||
|
registry_login() {
|
||||||
|
echo " $ docker login $REGISTRY --username \$DOCKER_DCGLAB_CI_USERNAME --password-stdin"
|
||||||
|
[[ "$DRY_RUN" == true ]] && return 0
|
||||||
|
|
||||||
|
printf '%s' "$DOCKER_DCGLAB_CI_PASSWORD" |
|
||||||
|
docker login "$REGISTRY" \
|
||||||
|
--username "$DOCKER_DCGLAB_CI_USERNAME" \
|
||||||
|
--password-stdin
|
||||||
|
}
|
||||||
|
|
||||||
# Push one image tag, retrying on transient registry failures.
|
# Push one image tag, retrying on transient registry failures.
|
||||||
#
|
#
|
||||||
# The engine images carry a ~5.6GB torch layer. Uploading it intermittently
|
# The engine images carry a ~5.6GB torch layer. Uploading it intermittently
|
||||||
@@ -165,6 +175,17 @@ echo ""
|
|||||||
echo "==> Pre-flight checks"
|
echo "==> Pre-flight checks"
|
||||||
|
|
||||||
if [[ "$DRY_RUN" == false ]]; then
|
if [[ "$DRY_RUN" == false ]]; then
|
||||||
|
if [[ -z "${DOCKER_DCGLAB_CI_USERNAME:-}" ]]; then
|
||||||
|
echo "Error: DOCKER_DCGLAB_CI_USERNAME is required" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
if [[ -z "${DOCKER_DCGLAB_CI_PASSWORD:-}" ]]; then
|
||||||
|
echo "Error: DOCKER_DCGLAB_CI_PASSWORD is required" >&2
|
||||||
|
exit 1
|
||||||
|
fi
|
||||||
|
|
||||||
|
registry_login
|
||||||
|
|
||||||
if git -C "$SCRIPT_DIR" rev-parse "$GIT_TAG" &>/dev/null; then
|
if git -C "$SCRIPT_DIR" rev-parse "$GIT_TAG" &>/dev/null; then
|
||||||
echo "Error: tag $GIT_TAG already exists"
|
echo "Error: tag $GIT_TAG already exists"
|
||||||
exit 1
|
exit 1
|
||||||
|
|||||||
Reference in New Issue
Block a user