my-homelab-configs/scripts/build-homelab-ai-index

213 lines
6.5 KiB
Python
Executable File

#!/usr/bin/env python3
"""Build a small local retrieval index from non-secret homelab repo files."""
from __future__ import annotations
import argparse
import fnmatch
import json
import math
import os
import pathlib
import re
import sys
import time
from collections import Counter
REPO_ROOT = pathlib.Path(__file__).resolve().parents[1]
DEFAULT_SOURCES_FILE = REPO_ROOT / "infra" / "ai" / "knowledge-sources.txt"
DEFAULT_EXCLUDES_FILE = REPO_ROOT / "infra" / "ai" / "knowledge-excludes.txt"
DEFAULT_INDEX_DIR = pathlib.Path(os.environ.get("LAB_AI_KNOWLEDGE_INDEX_DIR", "/data/homelab-ai/index"))
TOKEN_RE = re.compile(r"[A-Za-z0-9_./:-]{2,}")
MAX_FILE_BYTES = 512 * 1024
CHUNK_TARGET_LINES = 80
CHUNK_OVERLAP_LINES = 12
def tokenise(text: str) -> list[str]:
return [token.lower() for token in TOKEN_RE.findall(text)]
def load_patterns(path: pathlib.Path) -> list[str]:
patterns: list[str] = []
for raw_line in path.read_text(encoding="utf-8").splitlines():
line = raw_line.strip()
if not line or line.startswith("#"):
continue
patterns.append(line)
return patterns
def path_is_secret(path: pathlib.Path) -> bool:
lowered = str(path).lower()
secret_markers = (
".git/",
".terraform/",
".lab/",
"__pycache__/",
"terraform.tfstate",
".secret.",
".enc.",
"id_ed25519",
"cosign.key",
"keys.txt",
)
return any(marker in lowered for marker in secret_markers)
def path_matches_any(rel_path: str, patterns: list[str]) -> bool:
return any(fnmatch.fnmatch(rel_path, pattern) for pattern in patterns)
def candidate_files(patterns: list[str], exclude_patterns: list[str]) -> list[pathlib.Path]:
files: set[pathlib.Path] = set()
all_files = [path for path in REPO_ROOT.rglob("*") if path.is_file()]
for pattern in patterns:
if any(char in pattern for char in "*?["):
for path in all_files:
rel = path.relative_to(REPO_ROOT).as_posix()
if fnmatch.fnmatch(rel, pattern):
files.add(path)
else:
path = REPO_ROOT / pattern
if path.is_file():
files.add(path)
return sorted(
path
for path in files
if not path_is_secret(path)
and not path_matches_any(path.relative_to(REPO_ROOT).as_posix(), exclude_patterns)
)
def read_text(path: pathlib.Path) -> str | None:
try:
if path.stat().st_size > MAX_FILE_BYTES:
return None
return path.read_text(encoding="utf-8")
except UnicodeDecodeError:
return None
def chunk_file(path: pathlib.Path, text: str) -> list[dict[str, object]]:
rel = path.relative_to(REPO_ROOT).as_posix()
lines = text.splitlines()
chunks: list[dict[str, object]] = []
start = 0
while start < len(lines):
end = min(start + CHUNK_TARGET_LINES, len(lines))
body = "\n".join(lines[start:end]).strip()
if body:
chunks.append(
{
"id": f"{rel}:{start + 1}",
"path": rel,
"start_line": start + 1,
"end_line": end,
"text": body,
"tokens": tokenise(body),
}
)
if end == len(lines):
break
start = max(end - CHUNK_OVERLAP_LINES, start + 1)
if not chunks and text.strip():
chunks.append(
{
"id": f"{rel}:1",
"path": rel,
"start_line": 1,
"end_line": 1,
"text": text.strip(),
"tokens": tokenise(text),
}
)
return chunks
def build_index(files: list[pathlib.Path]) -> dict[str, object]:
chunks: list[dict[str, object]] = []
document_frequency: Counter[str] = Counter()
for path in files:
text = read_text(path)
if text is None:
continue
for chunk in chunk_file(path, text):
tokens = chunk["tokens"]
if not isinstance(tokens, list) or not tokens:
continue
chunks.append(chunk)
document_frequency.update(set(tokens))
chunk_count = len(chunks)
average_length = sum(len(chunk["tokens"]) for chunk in chunks) / chunk_count if chunk_count else 0
idf = {
token: math.log(1 + (chunk_count - freq + 0.5) / (freq + 0.5))
for token, freq in document_frequency.items()
}
return {
"version": 1,
"repo_root": str(REPO_ROOT),
"chunk_count": chunk_count,
"average_length": average_length,
"chunks": chunks,
"idf": idf,
}
def main() -> int:
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--sources-file", type=pathlib.Path, default=DEFAULT_SOURCES_FILE)
parser.add_argument("--excludes-file", type=pathlib.Path, default=DEFAULT_EXCLUDES_FILE)
parser.add_argument("--index-dir", type=pathlib.Path, default=DEFAULT_INDEX_DIR)
args = parser.parse_args()
if not args.sources_file.is_file():
print(f"missing sources file: {args.sources_file}", file=sys.stderr)
return 1
patterns = load_patterns(args.sources_file)
exclude_patterns = load_patterns(args.excludes_file) if args.excludes_file.is_file() else []
files = candidate_files(patterns, exclude_patterns)
index = build_index(files)
args.index_dir.mkdir(parents=True, exist_ok=True)
index_path = args.index_dir / "index.json"
manifest_path = args.index_dir / "manifest.json"
index_path.write_text(json.dumps(index, indent=2, sort_keys=True) + "\n", encoding="utf-8")
manifest_path.write_text(
json.dumps(
{
"generated_at_epoch": int(time.time()),
"sources_file": str(args.sources_file.relative_to(REPO_ROOT)),
"excludes_file": str(args.excludes_file.relative_to(REPO_ROOT)) if args.excludes_file.is_file() else None,
"file_count": len(files),
"chunk_count": index["chunk_count"],
"repo_root": str(REPO_ROOT),
"files": {
path.relative_to(REPO_ROOT).as_posix(): int(path.stat().st_mtime)
for path in files
},
},
indent=2,
sort_keys=True,
)
+ "\n",
encoding="utf-8",
)
print(f"indexed {len(files)} file(s), {index['chunk_count']} chunk(s) into {index_path}")
return 0
if __name__ == "__main__":
raise SystemExit(main())