diff --git a/.env.example b/.env.example new file mode 100644 index 00000000..444490bb --- /dev/null +++ b/.env.example @@ -0,0 +1,30 @@ +# ThothII Compose defaults. Copy this file to .env in the repository root. +# The root .env is loaded automatically by Docker Compose; do not put secrets here. + +COMPOSE_FILE=compose.yaml +COMPOSE_PROFILES= +THT_SECRETS_FILE=deploy/secrets/thothii.secrets + +THOTH_HTTP_PORT=8080 +AUTH_MODE=none +THOTH_PUBLIC_EXPOSURE=false +MAX_PI_PROCESSES=4 +PI_PROVIDER= +PI_MODEL= +PI_THINKING= + +# Set these for the selected DWH/vector/embedding adapters. +THT_DB_NAME= +THT_DWH_REST_URL= +THT_VEC_REST_URL= +THT_VEC_WRITE_REST_URL= +THT_OLLAMA_URL= +THT_DOCS_ROOT=/data/workspaces/example/evidence-source +THT_PROFILE=server + +# Local-vector defaults (used by the optional local-vector overlay). +THT_VECTOR_DATABASE=thoth +THT_VECTOR_BOOTSTRAP_USER=postgres +THT_VECTOR_MIGRATOR_USER=thoth_vector_migrator +THT_VECTOR_READER_USER=thoth_vector_reader +THT_VECTOR_WRITER_USER=thoth_vector_writer diff --git a/.github/workflows/container-multiarch.yml b/.github/workflows/container-multiarch.yml new file mode 100644 index 00000000..3e5c87bc --- /dev/null +++ b/.github/workflows/container-multiarch.yml @@ -0,0 +1,34 @@ +name: Container multi-architecture gate + +on: + pull_request: + paths: + - "backend/**" + - "frontend/**" + - "harness/**" + - "docker/**" + - "scripts/verify-container-images.sh" + - ".github/workflows/container-multiarch.yml" + workflow_dispatch: + +jobs: + verify: + runs-on: ubuntu-24.04 + strategy: + fail-fast: false + matrix: + platform: [linux/amd64, linux/arm64] + steps: + - uses: actions/checkout@v4 + - uses: docker/setup-qemu-action@v3 + - uses: docker/setup-buildx-action@v3 + with: + driver: docker + - name: Build, smoke, security-check, and inventory + env: + PLATFORM: ${{ matrix.platform }} + run: ./scripts/verify-container-images.sh + - uses: actions/upload-artifact@v4 + with: + name: container-inventory-${{ strategy.job-index }} + path: .artifacts/container-images/ diff --git a/.gitignore b/.gitignore index 59867390..520bcbbc 100644 --- a/.gitignore +++ b/.gitignore @@ -33,6 +33,14 @@ deploy/thothii.env ca-chain.pem config/ca-chain.pem +# ThothII deployment configuration and secret values (keep only the README tracked) +deploy/.env +deploy/compose.psd-local.yaml +deploy/workspaces/psd.yaml +deploy/secrets/* +!deploy/secrets/README.md +!deploy/secrets/*.example + # === Runtime data (sessions contain PII; indexes are derived) === harness/sessions/ harness/indexes/ @@ -55,3 +63,6 @@ htmlcov/ # === MkDocs build output === site/ + +# Generated container inventory / SBOM-equivalent verification artifacts +.artifacts/ diff --git a/.superpowers/sdd/adapter-final-fix-report.md b/.superpowers/sdd/adapter-final-fix-report.md new file mode 100644 index 00000000..dfdd4c7e --- /dev/null +++ b/.superpowers/sdd/adapter-final-fix-report.md @@ -0,0 +1,137 @@ +# Adapter Foundations final-review fix report + +Date: 2026-07-11 +Branch: `codex/portable-deployment` +Worktree: `/Users/mp/projects/ThothII/.worktrees/portable-deployment` +Binding findings: `.superpowers/sdd/adapter-final-review-findings.md` + +## Outcome + +All seven final-review findings are addressed as one coherent adapter-foundations change: + +1. HTTP vector reader and writer clients are independently optional. Capabilities reflect the + configured side; writer-only new and legacy configurations build successfully for targeted + writes; search without a reader raises public `VectorReadUnavailable`. +2. `VectorHealth` now reports read/write configured and reachable state independently, preserves + side-specific errors, and reports expected/observed embedding dimensions plus compatibility. + HTTP diagnostics cover read-only, write-only, both-up, and writer-down cases. Direct health + exposes its configured expected dimension without adding schema or migration work. +3. `ThothRestDwhAdapter` accepts `DatabaseIdentityConfig`, matching its resource contract. +4. Both vector adapters reject bools, floats, zero, and negative search limits using one exact + positive-integer guard. +5. Port tests explicitly cover public exports and frozen capability records. +6. A real `tht` subprocess test proves one legacy deprecation warning per config load on stderr + while JSON stdout remains parseable and uncontaminated. +7. The adapter plan and SDD progress explicitly constrain `build_vector_loader` to transitional + bulk sync and schedule its removal/migration in the local pgvector plan. Targeted memory and + solved-question writes remain on `build_vector_store(..., require_write=True)`. + +No pgvector schema or migration changes were made. + +## Files changed + +- `harness/tht/ports/vector.py` +- `harness/tht/ports/__init__.py` +- `harness/tht/adapters/vector/thoth_http.py` +- `harness/tht/adapters/vector/legacy_direct.py` +- `harness/tht/adapters/factory.py` +- `harness/tht/adapters/dwh/thoth_rest.py` +- `harness/tests/test_vector_port_contract.py` +- `harness/tests/test_adapter_factory.py` +- `harness/tests/test_config_resources.py` +- `harness/tests/test_config_legacy_compat.py` +- `harness/tests/test_adapter_command_regressions.py` +- `harness/tests/test_dwh_port_contract.py` +- `docs/superpowers/plans/2026-07-11-adapter-foundations.md` +- `.superpowers/sdd/progress.md` +- `.superpowers/sdd/adapter-final-fix-report.md` + +## TDD and verification evidence + +RED: + +```text +cd harness && .venv/bin/pytest tests/test_vector_port_contract.py \ + tests/test_adapter_factory.py tests/test_config_resources.py \ + tests/test_config_legacy_compat.py -q +``` + +Result: collection failed as expected because `VectorReadUnavailable` did not exist. After the +initial implementation, the same command exposed two expected contract/test-harness corrections: +dimension mismatch makes aggregate health unhealthy, and the installed CLI entry point is `tht` +rather than `python -m tht.cli`. + +GREEN, covering adapter/config/command regressions: + +```text +cd harness && .venv/bin/pytest tests/test_vector_port_contract.py \ + tests/test_adapter_factory.py tests/test_config_resources.py \ + tests/test_config_legacy_compat.py tests/test_adapter_command_regressions.py \ + tests/test_dwh_port_contract.py tests/test_memory_save_one.py \ + tests/test_solved_question.py tests/test_search_similar_kinds.py \ + tests/test_vector_dual_key.py -q +``` + +Result: `66 passed in 0.45s`. + +Docker availability: + +```text +docker info --format '{{.ServerVersion}}' +``` + +Result: `29.4.1` (available; command required Docker socket access). + +Full repository-default non-L2 harness suite, with Docker available for L0 tests: + +```text +cd harness && .venv/bin/pytest -q +``` + +Result: `433 passed, 5 deselected, 17 warnings in 9.14s`. The five deselections are the configured +L2/live-service tests. Warnings are existing legacy-workspace `FutureWarning` emissions. + +Scoped lint and diff hygiene: + +```text +cd harness && .venv/bin/ruff check tht/ports tht/adapters \ + tests/test_vector_port_contract.py tests/test_adapter_factory.py \ + tests/test_config_resources.py tests/test_config_legacy_compat.py \ + tests/test_adapter_command_regressions.py tests/test_dwh_port_contract.py +git diff --check +``` + +Result: `All checks passed!`; `git diff --check` produced no output. + +## Commit + +Commit subject: `fix(adapter): close final foundation review` + +The report is part of that same final commit. A Git object cannot contain its own SHA without +changing that SHA; the exact resulting commit ID is therefore recorded in the task handoff from +`git rev-parse HEAD` after creation. + +## Self-review + +- Reader/writer separation is preserved: search dereferences only `_reader`; hashes/upsert only + `_writer`; health probes each configured client independently and never substitutes one result + for the other. +- Writer failure contributes to aggregate `ok=False`, even when the reader succeeds. +- Dimension compatibility is derived only from configured embedding dimension and existing + `list_tables` metadata. Missing metadata remains `None`, not a guessed success/failure. +- The shared limit guard uses `type(limit) is int`, intentionally rejecting Python booleans and + numeric coercions before either adapter reaches its transport. +- Existing JSON/CLI behavior is preserved; the subprocess regression parses stdout as JSON and + counts exactly one deprecation marker on stderr. +- Scope remains adapter foundations. No vector DDL, schema initialization, or migration work was + introduced. + +## Concerns / follow-up + +- Write reachability uses the existing `list_tables` diagnostic on the separately authenticated + writer client. Deployments must allow that non-mutating diagnostic RPC to the writer credential; + failures are intentionally visible rather than hidden by reader success. +- Existing legacy-workspace tests emit 17 `FutureWarning`s in the full suite. This wave pins the + required production stderr behavior but does not migrate unrelated test fixtures. +- `build_vector_loader` remains transitional technical debt only for bulk sync, explicitly assigned + to `2026-07-11-local-pgvector-profile.md`. diff --git a/.superpowers/sdd/container-task-3-report.md b/.superpowers/sdd/container-task-3-report.md new file mode 100644 index 00000000..faa90627 --- /dev/null +++ b/.superpowers/sdd/container-task-3-report.md @@ -0,0 +1,206 @@ +# Container Packaging Task 3 Report + +## Status + +Implemented the multi-stage core application image, non-root runtime, pinned Pi installation, +container entrypoint, context exclusions, and an in-image health smoke test. + +## TDD / Build Evidence + +Initial RED: + +```text +docker build -f docker/core.Dockerfile -t thothii-core:test . +ERROR: failed to build: resolve : lstat docker: no such file or directory +``` + +The first sandboxed attempt could not access the Docker socket; the authorized rerun reached the +builder and failed for the expected reason: the Dockerfile did not exist. + +GREEN build: + +```text +sh -n docker/core-entrypoint.sh docker/smoke/core-smoke.sh +docker build --progress=plain -f docker/core.Dockerfile -t thothii-core:test . +``` + +Result: shell syntax exited 0; Docker build exited 0. A final rebuild after tightening +`.dockerignore` also exited 0 and transferred only 17.60 kB of changed context (the initial clean +build transferred 1.02 MB). + +## Runtime and Entrypoints + +- Runtime user is `10001:10001` (`thoth`), never root. +- Runtime contains Node `v22.19.0` and Python `3.12.13`. Python 3.12 is intentional because the + harness declares `requires-python = ">=3.12"` and also satisfies the deployment floor of 3.11+. +- Pi is installed exactly as `@earendil-works/pi-coding-agent@0.80.3`; its build-time and runtime + version probes both reported `0.80.3`. +- `server` starts `/app/backend/dist/server.js`; `doctor` routes to `tht doctor`; `preprocess` + routes to the future-facing `tht preprocess` command; explicit `tht ...` and arbitrary CLI + arguments route to the installed `tht` binary. +- The gate extension's `typebox` runtime dependency is installed from the harness lockfile. + +## Smoke and Diagnostic Results + +```text +docker run --rm thothii-core:test doctor +config: error - configuration is invalid or unreadable +data_root: ok +``` + +Result: expected exit 1 for absent mounted workspace configuration, with no traceback and no +secret-bearing validation detail. + +```text +docker run --rm --entrypoint /app/docker/smoke/core-smoke.sh thothii-core:test +backend listening on http://127.0.0.1:8787 +v22.19.0 +Python 3.12.13 +core smoke: ok +``` + +Result: exit 0. The script asserted non-root execution, `tht --help`, `pi --version`, runtime +version floors, and `GET /health` through curl. Fastify's returned display address was loopback; +the inspected container environment is `HOST=0.0.0.0`, and the compiled server passes that value +to `app.listen`. + +```text +docker run --rm thothii-core:test tht --version +0.1.0 +``` + +Result: arbitrary `tht` entrypoint exited 0. + +An explicit runtime assertion checked UID 10001, exact Node and Pi versions, Python 3.11+, and the +absence of `/app/harness/.env` and `/app/harness/workspaces`; it exited 0. + +## Image Size and Containment Inspection + +```text +docker image inspect thothii-core:test --format '{{.Size}} {{json .Config.User}} {{json .Config.Env}}' +221419008 "10001:10001" [...runtime paths and version metadata only...] +``` + +Image size: **221,419,008 bytes** (about 211.2 MiB). + +`docker history --no-trunc thothii-core:test` was inspected. It contains only Dockerfile commands, +the pinned public package name/version, base-image metadata, and non-sensitive runtime variables; +no credentials or customer paths were found. An in-image filename scan found only +`/app/harness/.pi/settings.json` among `.env`, key/certificate, and settings-name candidates; that +tracked Pi file contains theme/startup preferences, not secrets. The build asserts `.env` and +workspace directories are absent. + +`.dockerignore` excludes VCS/agent state, all environment files except examples, package-manager +credential files, SSH/private-key and certificate formats, local virtualenvs/node_modules/caches, +backend runtime data, customer workspaces, sessions, artifacts, indexes, corpus, and deployment +mount content. + +## Self-review + +- `git diff --check` is clean. +- Entrypoint processes use `exec`, preserving container signal handling. +- Backend production dependencies are pruned; TypeScript build tools remain in the build stage. +- The writable `/data` root is owned by UID 10001; application payload remains root-owned and + read-only to the runtime user. +- CA certificates and curl are present for HTTPS integrations and health probing. +- No existing source, customer workspace, secret, or unrelated progress-ledger change is included + in the task commit. + +## Concerns + +- The `tht preprocess` command is deliberately a future-facing routing contract; its CLI group is + scheduled in the Evidence/preprocessing plan and is not implemented in the current harness. +- Python dependencies are range-resolved because the existing harness has no Python lockfile. The + Pi package, Node runtime, and package-lock-backed Node dependency sets are pinned/reproducible. +- The image was built and smoked on Docker Desktop arm64. The chosen official multi-arch base + images and Pi package are architecture-neutral at the package level, but amd64 still needs a CI + build/smoke before being advertised as verified. + +## Reproducibility Review Fix + +The original image pinned Pi's direct version in the Dockerfile but resolved its transitives at +build time, and pip resolved all harness dependencies from ranges. Both paths now consume committed +locks. + +### Lock generation + +Pi uses the minimal `docker/pi-runtime/package.json` and its committed npm v3 lock. It was generated +with: + +```text +npm install --package-lock-only --ignore-scripts --no-audit --no-fund \ + --prefix docker/pi-runtime +``` + +The package manifest specifies exact `@earendil-works/pi-coding-agent` version `0.80.3`; a lock +inspection confirmed that same resolved package version. Docker installs it with: + +```text +npm ci --omit=dev --ignore-scripts --no-audit --no-fund +``` + +The Python lock was generated directly from the harness production metadata plus one explicit, +pinned PEP 517 build-backend input—not from a host `pip freeze`: + +```text +uv pip compile harness/pyproject.toml docker/python-runtime/build-requirements.in \ + --universal \ + --python-version 3.12 \ + --no-emit-package tht \ + --generate-hashes \ + --custom-compile-command \ + 'uv pip compile harness/pyproject.toml docker/python-runtime/build-requirements.in --universal --python-version 3.12 --no-emit-package tht --generate-hashes --output-file docker/python-runtime/requirements.lock' \ + --output-file docker/python-runtime/requirements.lock +``` + +`pytest`, `ruff`, and `testcontainers` are absent. All production direct and transitive packages +are exact and hashed. `setuptools==80.9.0` is explicit so the local harness install can use +`--no-build-isolation` without an unpinned build-time resolution. Refresh instructions are in +`docker/LOCKS.md`. + +### No-cache rebuild and verification + +Final build command: + +```text +docker build --no-cache -f docker/core.Dockerfile -t thothii-core:test . +``` + +Result: exit 0. The logs showed Pi `0.80.3`, Node `v22.19.0`, a hash-enforced Python dependency +install, explicit `setuptools==80.9.0`, and a non-isolated local `tht` wheel build. No isolated +build-dependency download occurred. + +Fresh runtime checks: + +```text +docker run --rm --entrypoint /app/docker/smoke/core-smoke.sh thothii-core:test +backend listening on http://127.0.0.1:8787 +v22.19.0 +Python 3.12.13 +core smoke: ok + +docker run --rm thothii-core:test tht --version +0.1.0 + +/opt/venv/bin/pip check +No broken requirements found. +``` + +An in-container package inspection reconfirmed Pi `0.80.3`. Non-root UID, runtime version floors, +doctor's expected concise exit 1/no traceback, `/health`, and arbitrary `tht` routing all passed. + +The full filename containment scan found no `.env`, PEM, private-key, P12, or PFX file in `/app`; +`/app/harness/workspaces` remains absent. Image environment and `docker history --no-trunc` were +re-inspected and contain only public package/build commands and non-sensitive runtime metadata. + +Final locked image size: + +```text +220003986 10001:10001 +``` + +That is **220,003,986 bytes** (about 209.8 MiB), 1,415,022 bytes smaller than the original image. + +Remaining concern: the universal lock is resolved for Python 3.12 and includes hashes/markers for +all supported platforms, but only Linux arm64 has been built and smoked locally; amd64 remains a CI +verification gate. diff --git a/.superpowers/sdd/container-task-4-report.md b/.superpowers/sdd/container-task-4-report.md new file mode 100644 index 00000000..3ea43714 --- /dev/null +++ b/.superpowers/sdd/container-task-4-report.md @@ -0,0 +1,82 @@ +# Container Packaging Task 4 Report + +## Status + +Implemented and verified runtime-configured frontend packaging. + +## Changes + +- Added the browser runtime contract `window.__THOTHII_CONFIG__.backendBaseUrl`. +- Loaded `/config.js` before the Vite module entrypoint. +- Made runtime configuration take precedence while preserving `VITE_BACKEND_URL` and the + existing `http://localhost:8787` client default for development and tests. +- Added a multi-stage frontend image that builds with Node and serves static assets as + unprivileged UID/GID `101:101` with nginx on port 8080. +- Added startup-time `BACKEND_BASE_URL` substitution (default `/api`). +- Added `/api/` reverse proxying to `core:8787`, SPA fallback, no-cache runtime config, + and SSE-safe proxy settings (`proxy_buffering off`, `proxy_cache off`, one-hour read timeout). + +## TDD evidence + +- RED: `npx vitest run src/api/runtime-config.test.ts` failed because + `./runtime-config` did not exist. +- GREEN: targeted runtime config suite passed (3 tests after preserving the legacy client + default). + +## Verification + +- `cd frontend && npx vitest run --reporter=dot && npx tsc -b && npm run build` — exit 0 + (40 test files, 185 tests; TypeScript and Vite production build passed). +- `docker build -f docker/frontend.Dockerfile -t thothii-frontend:test .` — success. +- Image metadata reports `USER 101:101`. +- Two-container isolated-network smoke: + - `/config.js` returned `window.__THOTHII_CONFIG__ = { backendBaseUrl: "/api" };` + - `/api/health` proxied to the core image and returned `{"status":"ok"}`. + - an unknown nested route returned the SPA `index.html`. + - active nginx config contained `proxy_buffering off`, `proxy_cache off`, and + `proxy_read_timeout 1h`. + - `/config.js` returned `Cache-Control: no-store`. +- `sh -n docker/frontend-entrypoint.sh` and `git diff --check` — exit 0. + +## Secret-leakage inspection + +- `.dockerignore` excludes `.env*` (except examples), credentials/key formats, dependency + trees, build outputs, backend data, and deployment data. +- The runtime web root contained no `.env*`, `.pem`, `.key`, `.p12`, or `.pfx` files. +- Image history contained build/package instructions only; no secret build arguments or + credential values were introduced by this task. + +## Self-review / concerns + +- nginx resolves the `core` hostname at startup, matching the planned Compose service name; + standalone runs therefore need a reachable network alias named `core`. +- Existing frontend test warnings (React refs/act, MSW unmatched incidental requests, Vite + chunk-size warnings) remain; they did not fail the requested gates and are unrelated to + this task. +- `.superpowers/sdd/progress.md` was already modified by the orchestrator and was intentionally + excluded from this task's commit. + +## P1 review fixes + +Follow-up commit work addressed both review findings: + +- Runtime configuration is now produced with `jq -cn --arg`, so `BACKEND_BASE_URL` is encoded + by a real JSON serializer rather than interpolated into JavaScript by `sed`. +- The image includes `frontend-config-smoke`, which strips only the fixed assignment wrapper, + parses the remaining JSON with `jq`, requires exactly the `backendBaseUrl` key, and compares + the decoded value to the environment input. +- The hostile smoke passed with quotes, backslashes, a literal newline, ampersand, pipe, and + `"; globalThis.PWNED=true; //` in the value. A breakout would leave non-JSON trailing input + and fail parsing. +- Added `joinBackendPath`, shared by API fetch and EventSource creation. It removes duplicate + boundary slashes for relative and absolute bases while keeping empty and `/` bases rooted. + +Follow-up verification: + +- RED: six join cases failed with `joinBackendPath is not a function` before implementation. +- Targeted: runtime config, API client, and EventSource suites — 14 tests passed. +- Full frontend gate — exit 0 (40 test files, 191 tests, TypeScript, Vite build). +- Rebuilt `thothii-frontend:test` successfully. +- Hostile config image smoke — `frontend runtime config smoke: ok`. +- Rebuilt two-container smoke — default `/api` config, proxied `/api/health`, SPA fallback, + and SSE-safe nginx directives all passed. diff --git a/.superpowers/sdd/evidence-task-1-report.md b/.superpowers/sdd/evidence-task-1-report.md new file mode 100644 index 00000000..c8c6dd18 --- /dev/null +++ b/.superpowers/sdd/evidence-task-1-report.md @@ -0,0 +1,98 @@ +# Evidence / Preprocessing Task 1 Report + +## Outcome + +Implemented the additive Evidence source port and canonical corpus records. Existing evidence, +search, vector, and session runtime code is unchanged. + +## Contract + +- `EvidenceSource` is a runtime-checkable protocol with `discover` and `acquire` operations. +- `SourceObject` and `AcquiredDocument` are frozen, reject extra fields, use independent metadata + defaults, and restrict metadata to Pydantic `JsonValue` values. +- `CanonicalDocument`, `CanonicalChunk`, and `CorpusManifest` are frozen and reject extra fields. +- Provenance includes stable source IDs, canonical URIs, fingerprints, modification time, and + content hashes. +- Pipeline versions are recorded on documents, chunks, and manifests. Manifests also carry schema + version, optional publish ID/vector generation, and paired embedding model/dimension fields. +- Credential-like metadata keys are rejected recursively. Credentials are not model fields and + therefore cannot enter serialized canonical artifacts through extras. + +## TDD evidence + +The initial focused run failed during collection because `tht.ports.evidence` and `tht.corpus` +did not exist. After implementation, the focused suite passed. + +## Verification + +- Focused models/protocol tests: 13 passed. +- Harness excluding Docker-backed L0 and the network-dependent wheel packaging test: 444 passed, + 5 deselected. +- Focused Ruff: passed. +- Full-repository Ruff remains blocked by 34 pre-existing findings outside the task files. +- An unrestricted `pytest -q` attempt reached 453 passed and 5 deselected, but reported 47 Docker + setup errors plus 4 Docker parity failures because the sandbox cannot access the Docker socket; + the wheel packaging test also failed because its isolated `uv build` needs unavailable network. + +## Concerns / follow-up + +- Pydantic's `frozen=True` prevents model field reassignment but does not recursively freeze list + and dict contents. `default_factory` prevents shared mutable defaults. Later pipeline stages should + treat these value objects as immutable and construct replacements rather than mutate collections. +- The adapter and normalization tasks should preserve the credential-free boundary by passing only + these records beyond acquisition. + +## Review hardening follow-up + +All six binding review areas were addressed in a separate TDD pass: + +- JSON metadata is recursively converted to immutable `FrozenDict`/tuple values while retaining + stable object/array JSON serialization. Manifest document and chunk collections are tuples. +- Secret-key matching now normalizes camelCase and punctuation. It rejects credential-specific + names (passwords, API keys, access/refresh tokens, client/private keys, session cookies and + authorization) recursively, while deliberate benign labels such as generic `token` and `secret` + remain valid. +- Canonical URIs require a scheme and reject userinfo or credential-bearing query parameters. +- Namespaced IDs, SHA-256 content hashes, timezone-aware UTC timestamps, embedding/vector + compatibility, unique IDs, chunk referential/provenance integrity, contiguous per-document + ordinals and pipeline-version consistency are validated. Nested Pydantic instances are always + revalidated so `model_copy(update=...)` cannot bypass a manifest boundary. +- Acquired arbitrary bytes have explicit base64 JSON encoding and validation, covered by a JSON + round-trip test. +- `EvidenceSourceError` classifies transient/retryable versus permanent failures and exposes only + recursively immutable, credential-screened JSON details. + +Follow-up verification: + +- Focused contract suite: 39 passed. +- Focused Ruff: passed. +- Harness excluding Docker-backed L0 and the network-dependent wheel packaging test: 470 passed, + 5 deselected. +- Fresh unrestricted harness attempt: 479 passed, 5 deselected; the same environmental boundary + remains (47 Docker socket setup errors, four Docker parity failures, one isolated `uv build` + network failure). + +## Final blocker follow-up + +The remaining four contract blockers were closed in a third TDD cycle: + +- `EvidenceSourceError` now always exposes the fixed public message/`args` value `evidence source + operation failed`; caller diagnostics are not retained. Category, details and args cannot be + reassigned, details remain recursively frozen and credential-screened, and an original exception + is available only when callers use standard exception chaining. +- Canonical document/chunk provenance stores only URI scheme, authority and path. Userinfo is + rejected; query strings and fragments are removed unconditionally, including AWS `X-Amz-*`, SAS + `sig`, and fragment token material. +- Binding model bases override Pydantic's unchecked `model_copy(update=...)`: merged values always + pass full field/model validation, so invalid copied records and top-level manifests fail. +- A canonical document/chunk `content_hash` must equal SHA-256 of the exact stored text encoded as + UTF-8. This establishes the normalization boundary explicitly: line-ending/frontmatter/text + normalization happens before model construction; the canonical models never rewrite content. + +Final follow-up verification: + +- Focused contract suite: 45 passed. +- Focused Ruff: passed. +- Harness excluding Docker-backed L0 and network-dependent packaging: 476 passed, 5 deselected. +- Fresh unrestricted harness attempt: 486 passed, 5 deselected, with the unchanged environmental + failures (47 Docker setup errors, four Docker parity failures, one isolated `uv build` failure). diff --git a/.superpowers/sdd/evidence-task-2-report.md b/.superpowers/sdd/evidence-task-2-report.md new file mode 100644 index 00000000..583ffda0 --- /dev/null +++ b/.superpowers/sdd/evidence-task-2-report.md @@ -0,0 +1,65 @@ +# Evidence Task 2 Report + +## Status + +Implemented filesystem and explicit-manifest HTTP Evidence source adapters, typed source +configuration with legacy compatibility, and factory construction. + +## Delivered behavior + +- Filesystem discovery is deterministic and rooted at a strict canonical directory. +- Symlink/path escapes are rejected before content is exposed. +- Discovery hashing and acquisition reads enforce a configurable byte limit. +- Filesystem fingerprints are content SHA-256 values; stable IDs derive from relative paths. +- HTTP accepts only explicit `http`/`https` manifest entries and keeps transport URLs private. +- HTTP provenance strips query strings/fragments, while config and adapter representations hide + signed or secret-bearing transport URLs. +- HTTP acquisition uses separate connect/read timeouts, streaming byte limits, bounded redirects, + private redirect rejection, and safe transient/permanent error classification. +- HTTP fingerprints prefer a deterministic ETag digest, then Last-Modified, then content SHA-256. +- `build_evidence_sources(cfg)` supports both typed `evidence.sources` entries and the legacy + `source_root` plus `evidence_dir` filesystem configuration. + +## TDD and verification + +- RED: focused tests initially failed during collection because the adapter package did not exist. +- GREEN: `15 passed` for filesystem, HTTP, and resource-config tests. +- Full harness: `548 passed, 5 deselected`. +- Changed-file Ruff: clean. +- Repository-wide Ruff remains non-clean due to 34 pre-existing findings in unrelated test files; + no unrelated lint files were modified. + +## Notes + +The approved `SourceObject` namespace grammar does not permit raw quoted ETags such as +`etag:"abc"`. The adapter therefore uses `etag:`: it preserves ETag-based +change identity without weakening the canonical contract or exposing validator contents. + +## Review hardening follow-up + +Four review findings were closed in a separate follow-up commit: + +- Filesystem access now anchors a persistent descriptor at the canonical root and walks each + component with `openat` semantics (`dir_fd`, `O_NOFOLLOW`, and `O_DIRECTORY`). The regular-file + check, bounded read, metadata, and hash all use the opened descriptor. Acquisition reopens by + the same path-safe mechanism and rejects a changed fingerprint. Deterministic tests swap both a + leaf and an ancestor to symlinks at open time. +- HTTP network policy defaults to public hosts only. Initial URLs and every redirect reject + userinfo, mixed public/private IPv4/IPv6 answers fail closed, and the connected peer must be a + public member of the previously validated DNS answer set before any body bytes are consumed. + Explicit `allow_private_hosts: true` is required for trusted private deployments and local tests. +- Every HTTP response is closed in a `finally` block, including redirects, status failures, + policy failures, oversized bodies, and mid-stream exceptions. +- ETag and Last-Modified values remain adapter-internal. Repeated discovery and acquisition send + conditional headers; a 304 reuses only previously verified cached bytes and identity. The LRU + content cache has an explicit byte bound (`max_cache_bytes`). Validators are not forwarded + across redirect origins. + +### Conditional cache binding correction + +The conditional cache now binds bytes and validators to both the canonical provenance key and the +exact final effective representation URL. Redirect traversal recomputes request headers per hop: +validators are sent only when that exact URL matches the cached final URL, never merely because a +redirect retains an origin. A same-origin path change therefore downloads and replaces the body. +The adapter accepts 304 only when the exact request carried a bound ETag or Last-Modified validator; +unsolicited and cross-origin 304 responses are permanent protocol errors. diff --git a/.superpowers/sdd/evidence-task-3-report.md b/.superpowers/sdd/evidence-task-3-report.md new file mode 100644 index 00000000..db9ad616 --- /dev/null +++ b/.superpowers/sdd/evidence-task-3-report.md @@ -0,0 +1,59 @@ +# Evidence Task 3 — deterministic normalization and chunking + +## Outcome + +- Added pure `normalize(acquired, pipeline_version)` and `chunk(document, policy)` transforms. +- Normalization enforces UTF-8 (including UTF-8 BOM), a 10 MiB input ceiling, LF line endings, + NFC Unicode, safe YAML frontmatter extraction, canonical provenance URIs, and hashes the exact + canonical UTF-8 text stored on the document. +- Undecodable, unsupported-charset, oversized, and invalid-frontmatter inputs fail explicitly; + byte content is never truncated. +- Chunking uses a versioned immutable policy, paragraph/word boundaries with deterministic + character-count hard splits for long tokens, contiguous ordinals, provenance metadata, exact + per-chunk hashes, and IDs derived from document hash + ordinal + policy version. +- Empty documents produce no chunks. Non-ASCII, CRLF equivalence, repeatability, policy changes, + duplicate-content ordinal collisions, and max-character limits are covered by tests. + +## TDD evidence + +- Initial focused test run failed during collection because both transform modules were absent. +- The EOF-frontmatter edge test was separately observed failing before its implementation. +- Final focused verification: `12 passed`. + +## Verification + +- `cd harness && .venv/bin/pytest tests/test_corpus_normalize.py tests/test_corpus_chunk.py -q` + — **12 passed**. +- `cd harness && .venv/bin/pytest -q` — **573 passed, 5 deselected**. The sandboxed attempt could + not access Docker; the approved rerun with local Docker access passed. +- Targeted Ruff over all four implementation/test files — **clean**. +- Full `cd harness && .venv/bin/ruff check .` — reports **34 pre-existing errors** in unrelated + legacy tests (unused imports and existing E702 semicolon lines); none are in Task 3 files. + +## Concerns + +- The 10 MiB normalization ceiling is deliberately explicit and independent of adapter download + limits. If deployment policy needs a different ceiling, it should become a versioned pipeline + configuration before ingestion is wired. +- Character limits use Python Unicode code points (`len`), not UTF-8 bytes or tokenizer tokens; + this is recorded in the chunk-policy metadata and tested with non-ASCII content. + +## Review hardening follow-up + +- Chunk IDs now bind the canonical document identity, document content hash, ordinal, chunk hash, + and a canonical SHA-256 fingerprint of every `ChunkPolicy` field. Identical content in separate + documents and same-version policies with different limits cannot collide. +- Boundary-aware slicing now retains separators in the slices. Concatenating every chunk exactly + reconstructs the canonical document for repeated spaces, tabs, blank lines, Markdown hard + breaks, fenced code, whitespace-only input, Unicode, and overlong tokens; every slice remains + within `max_chars`. +- Frontmatter uses a bounded `SafeLoader` variant: duplicate keys, anchors/aliases, structures + deeper than 20 nodes, and documents larger than 1000 composed nodes are rejected. YAML parse, + JSON type, credential-safety, and resulting canonical-model errors attributable to frontmatter + map to `PermanentNormalizationError(reason="invalid_frontmatter")`; invalid pipeline policy + remains a programmer-facing `ValueError`. +- Follow-up TDD evidence: the expanded focused suite first reported 11 expected failures against + the prior implementation, then passed **45/45** across normalization, chunking, and manifest + invariants. +- Follow-up full verification: **586 passed, 5 deselected**. Targeted Ruff is clean. Full Ruff + continues to report the same **34 unrelated pre-existing** violations in legacy tests. diff --git a/.superpowers/sdd/evidence-task-4-report.md b/.superpowers/sdd/evidence-task-4-report.md new file mode 100644 index 00000000..8cfd32bb --- /dev/null +++ b/.superpowers/sdd/evidence-task-4-report.md @@ -0,0 +1,107 @@ +# Evidence Task 4 — shared job envelope + +Status: complete + +## Delivered + +- Immutable `JobSpec`, `JobRun`, `JobReport`, per-stage state, sanitized error, and UTC + timestamp records. +- `run_job(spec, stages)` with a durable checkpoint at job start, before and after every stage, + and at terminal state. Successful stages are skipped when a prior run is resumed. +- Atomic JSON checkpoint/report replacement using a unique same-directory temporary file, + file `fsync`, atomic `os.replace`, and parent-directory `fsync`. +- Public reports contain fixed operational fields only. Workspace paths, stage return values, + exception messages, source content, credentials, and arbitrary metadata are not serialized. +- `WorkspaceJobLock` uses non-blocking kernel `flock` on a stable workspace/job-specific inode. + Locks are released by the kernel on process exit; lock files are never removed based on PID, + avoiding stale-lock and PID-reuse deletion races. Evidence and DWH use distinct lock files. +- Dry-run intent is immutable in the spec/report and exposed to every stage through `JobContext`. + +## TDD evidence + +Initial focused collection failed because `tht.jobs` did not exist. Tests then drove: + +- failure, sanitized reporting, resume, and idempotent successful-stage skipping; +- corrupt-checkpoint refusal before stage execution; +- JSON schema and path/secret/PII exclusion; +- dry-run propagation and ordered aware timestamps; +- multiprocessing exclusion, distinct Evidence/DWH jobs, traversal rejection, and recovery after + a lock-owning process crashes. + +Final focused result: + +```text +11 passed in 0.42s +``` + +## Verification + +```text +cd harness && .venv/bin/pytest -q +597 passed, 5 deselected, 17 warnings in 28.45s + +cd harness && .venv/bin/ruff check tht/jobs tests/test_job_runner.py tests/test_job_locking.py +All checks passed! +``` + +The full Ruff invocation was also run. It reports 34 pre-existing violations in unrelated legacy +tests; no Task 4 file is among them. L2 tests remain deselected by the repository configuration. + +## Operational notes + +- `fcntl.flock` intentionally targets the supported Linux/macOS deployment environments; it is not + a Windows locking implementation. +- The envelope does not publish or mutate an active corpus. Later pipeline stages must use + `JobContext.run_dir` for staging and perform their own final atomic publish only after validation. +- A dry run is an execution mode foundation: the runner exposes and records it; individual stages + remain responsible for suppressing external mutations. + +## Review hardening follow-up + +Four post-implementation findings were fixed test-first: + +1. Resume compatibility is now a canonical SHA-256 fingerprint over checkpoint schema version, + hashed workspace identity, job type, dry-run mode, explicit spec/pipeline versions, + configuration/input fingerprints, and the exact ordered explicit `stage_ids`. Any insertion, + removal, reorder, mode, identity, version, config, or input change rejects resume before a stage + executes. Omitting `resume_run_id` remains the explicit safe path for a new run. +2. Lock traversal now uses directory file descriptors with `O_DIRECTORY` and `O_NOFOLLOW`. + Lock files use `O_NOFOLLOW | O_CLOEXEC`; `fstat` requires a regular file owned by the current + UID with one link, and permissions are forced to `0600` (`0700` for private directories). + Pre-existing lock-file and lock-directory symlinks are rejected. +3. Stage failures now serialize only the fixed safe tuple `internal` / `stage_exception` / + `stage execution failed`. Neither exception class names nor messages are inspected for output; + a hostile exception-name/message regression test proves a terminal failed report is retained. +4. Job/run directory creation is no-follow, owner-checked, private, and durable. Each newly created + parent is fsynced, the run directory is fsynced before the first atomic file write, and the + existing file-fsync → replace → directory-fsync ordering has an explicit regression test. + +Follow-up verification: + +```text +focused job/lock suite: 27 passed in 0.45s +full harness suite: 613 passed, 5 deselected, 17 warnings in 29.65s +Task 4 scoped Ruff: All checks passed +``` + +Repository-wide Ruff continues to report the same 34 unrelated pre-existing legacy-test findings. + +## Final resume-integrity fix + +Resume is now read-only until the source checkpoint proves trustworthy. The runner loads the source +before allocating a new run ID or directory, validates the exact stage state/timestamp/error ledger, +rejects duplicate stage identifiers, and recomputes compatibility from every persisted compatibility +field plus the exact ordered persisted stage IDs. It first requires the stored fingerprint to match +that recomputation, then compares the trusted recomputation with the requested job fingerprint. + +Valid-JSON tampering tests cover removed, inserted/duplicated, reordered, and substituted stages; +input-field and stored-fingerprint changes; and invalid stage-state shapes. Every rejection occurs +before stage execution and asserts that the runs directory contains no orphan allocation. + +Final verification: + +```text +focused job/lock suite: 34 passed in 0.56s +full harness suite: 620 passed, 5 deselected, 17 warnings in 27.42s +Task 4 scoped Ruff: All checks passed +``` diff --git a/.superpowers/sdd/evidence-task-5-report.md b/.superpowers/sdd/evidence-task-5-report.md new file mode 100644 index 00000000..e778b59d --- /dev/null +++ b/.superpowers/sdd/evidence-task-5-report.md @@ -0,0 +1,82 @@ +# Evidence Task 5 report + +## Outcome + +Implemented an incremental Evidence corpus pipeline with immutable materialized generations, +generation-scoped vector records, and an fsynced atomic `ACTIVE` pointer. Runtime Evidence +artifact lookup reads the active canonical manifest and keeps a legacy source-tree fallback only +when no corpus has been published. + +The CLI is available as `tht preprocess evidence [--dry-run] [--resume RUN_ID] [--json]`. +JSON success and failure output is pristine and failure details are sanitized. + +## Safety and failure model + +- A workspace writer lock serializes preprocess writers; readers never take the lock. +- Generation directories, manifests, materialized files, locks, and `ACTIVE` reject symlink/path + escape cases and use owner-only durable writes. +- Vector records use generation-specific keys and metadata. The active manifest maps each active + document to its valid vector generation, allowing unchanged documents to retain their vectors. +- Runtime retrieval admits only active document IDs and their manifest-selected generations. + Removed documents and partial writes from failed generations are therefore unreachable. +- Embedding count and dimension checks occur before vector upsert; vector write count is checked + before staging/publish. Any failure leaves `ACTIVE` unchanged. +- Dry runs perform discovery/fingerprint planning only and never acquire, embed, write vectors, or + publish. Fully unchanged runs return the active generation without creating a replacement. +- Resume can safely retry idempotent generation-scoped upserts and publish an already staged, + compatibility-checked generation after a crash between staging and pointer replacement. + +## TDD evidence + +Initial focused collection failed because `tht.corpus.pipeline` and `tht.corpus.store` did not +exist. The implemented suite covers incremental skips, removals, model/policy rebuilds, acquire and +partial-vector failures, dry-run isolation, dimension validation, atomic reader snapshots, pointer +validation, symlink defense, and pristine CLI JSON. + +Fresh focused verification: + +```text +18 passed, 3 warnings in 0.39s +``` + +Command: + +```text +.venv/bin/pytest tests/test_corpus_pipeline.py tests/test_corpus_publish.py \ + tests/test_preprocess_cli.py tests/test_search_pack.py tests/test_session_documents.py -q +``` + +Scoped Ruff: `All checks passed!` + +Broader non-Docker/non-packaging run reached `560 passed, 5 deselected`; ten pre-existing HTTP +adapter tests could not bind localhost under the sandbox. The complete suite reached `570 passed, +5 deselected`, with the remaining failures/errors caused by denied Docker socket, localhost bind, +and offline wheel-build access. No task-focused test failed. + +## Remaining operational gate + +Live pgvector integration needs Docker or an authorized local pgvector endpoint. The compensation +strategy is logical isolation rather than destructive cleanup because the shared `VectorStore` +port intentionally exposes no delete/transaction API; unreachable failed generations can be +garbage-collected by a future maintenance job. + +## Review integration wave + +Added an enforceable `metadata_filter` vector-port contract and capability flags. Direct pgvector +places exact Evidence generation/document predicates in SQL before `LIMIT`; HTTP sends the same +filter to the RPC and deliberately does not use the legacy 404 fallback. The reader RPC script now +validates and applies that filter. Normal Evidence search and search-pack use an ACTIVE-aware +searcher that groups active documents by generation, executes complete server-filtered searches, +and merges the results. + +Added exact-generation Evidence cleanup to direct and HTTP writers plus the allowlisted writer RPC. +Pipeline failures compensate both staged filesystem state and vector writes; cleanup failures stay +sanitized and ACTIVE filtering remains the exposure boundary. Corpus-present session artifact +resolution now fails closed on corrupt/missing ACTIVE rather than falling through to source files. + +Focused review-wave verification: 45 passed, scoped Ruff clean. A mocked REST regression proves +the exact filter payload and fail-closed legacy 404 behavior. + +Still outstanding from the expanded review request: Task-4 JobRunner stage-by-stage integration, +published-generation retention/garbage collection, same-fd `dirfd` materialized-file reads, and +live local pgvector integration could not be completed in this wave. diff --git a/.superpowers/sdd/evidence-task-5b-report.md b/.superpowers/sdd/evidence-task-5b-report.md new file mode 100644 index 00000000..ad0fcfdf --- /dev/null +++ b/.superpowers/sdd/evidence-task-5b-report.md @@ -0,0 +1,94 @@ +# Evidence Task 5B implementation report + +## Status + +Integrated Evidence preprocessing with the Task 4 `JobRunner`. The CLI now accepts only a +32-character JobRunner run ID for `--resume`; generation IDs remain outputs. Runs persist the +exact ordered stages `discover`, `acquire_normalize_chunk`, `embed`, `vector_upsert`, +`stage_validate`, `publish`, and `retention_cleanup`. + +Successful-stage artifacts are copied into the new resume run before execution, allowing later +stages to continue without rediscovery, acquisition, normalization, chunking, or embedding. +Job compatibility includes workspace, configuration, discovered-input, pipeline, embedding, and +chunk-policy fingerprints. Generation-specific filesystem/vector compensation is retained, and a +compensated generation is rotated before retry. `ACTIVE` is mutated only by `publish`. + +Dry-run executes discovery/planning and makes every side-effecting stage a no-op. JSON output is +pristine and includes the JobRunner `run_id`, `resumed_from`, generation, plan, and publish status. + +## TDD evidence + +- RED: run-ID rejection and resume-artifact tests failed because generation IDs reached + configuration and resume runs had empty artifact directories. +- GREEN: the two regression tests passed after strict CLI validation and durable artifact carryover. +- Added pipeline job-plan and dry-run counting-fake coverage; both passed. + +## Fresh verification + +- Focused integration/search suite: `62 passed, 4 warnings`. +- Available harness suite excluding sandbox-blocked Docker, loopback HTTP-server, and networked + wheel-build tests: `559 passed, 5 deselected, 18 warnings`. +- Scoped Ruff: `All checks passed!`. +- `git diff --check`: clean. + +## Environment limitations and concerns + +The literal full harness invocation cannot complete in the managed sandbox: Docker socket access, +loopback HTTP test servers, and the `uv build` dependency resolution path are denied. It reached +`575 passed, 5 deselected` before those environment errors. The available-suite rerun above is +green. + +One pre-existing Pydantic serialization warning is exposed by the new end-to-end job test when +canonical metadata contains frozen tuple values; it does not contaminate CLI stdout. Retention is +an explicit stable no-op until a retention policy is configured. + +## Review fix wave — crash consistency and artifact integrity + +Addressed all five follow-up findings: + +- `JobRunner` now supports a test-only post-call/pre-checkpoint fault hook. Each stage seals a + canonical artifact manifest containing required flat filenames, SHA-256, byte size, producer + stage, and the full spec compatibility fingerprint. Resume validates the checkpoint and every + sealed artifact before allocating/copying a new run, rejecting missing, tampered, extra, nested, + or symlinked state. A sealed `running` stage is promoted after a simulated process crash; a + sealed `failed` stage is deliberately retried. +- Vector intent (exact record IDs and content hashes) is sealed before upsert. Execution reconciles + `existing_hashes` and writes only missing/mismatched rows. Crash-after-effect tests prove no + duplicate acquire, embed, or vector upsert. +- Raw upsert, stage, recovery-upsert, recovery-stage, and publish exceptions compensate the exact + generation. Compensation markers survive failed checkpoints; resume rotates the generation, + refreshes generation-bound artifacts, reconciles vectors, and stages idempotently. +- `CorpusStore.publish` is idempotent and failure-atomic. If replace succeeds but directory fsync + fails, it restores the previous `ACTIVE` value (or removes a newly created pointer), fsyncs the + rollback, and re-raises. Pipeline cleanup refuses to discard a generation referenced by ACTIVE. +- Added crash/resume coverage after all seven ordered stages; corrupt/missing plan, manifest, and + embeddings; unsafe extra paths; nonexistent run IDs; raw vector/stage failures; and post-replace + ACTIVE rollback. + +Fresh fix-wave verification: + +- Focused jobs/corpus/CLI/search suite: `82 passed, 17 warnings`. +- Available harness suite (same sandbox exclusions described above): + `579 passed, 5 deselected, 31 warnings`. +- Scoped Ruff and `git diff --check`: clean. + +## Final P1 fix — effect state and checkpoint-bound manifest roots + +- Stage checkpoints now distinguish `intent` from `completed`. Vector intent is atomically sealed + and checkpointed before upsert. A process-level `BaseException` after a partial multi-record + write leaves the stage `running/intent`; resume never promotes it and instead reconciles + `existing_hashes`, writing only the missing records. The completed state is persisted only after + reconciliation returns successfully. +- Every stage now persists its completed artifact state while still `running`, before the + post-call fault hook. The checkpoint binds the SHA-256 of canonical `artifact-manifest.json`, + effect state, exact producer stage, and exact required-file mapping. Resume validates this root + and all bindings before promotion or copying. +- Added process-interruption coverage proving the already-written vector record is not submitted + twice, remaining records are written, and publish completes only after reconciliation. Added + coordinated artifact/manifest, spec-binding, and producer-binding tamper rejection tests. + +Fresh verification: + +- Focused jobs/corpus/CLI/search suite: `86 passed, 18 warnings`. +- Available broad harness suite: `583 passed, 5 deselected, 32 warnings`. +- Scoped Ruff and `git diff --check`: clean. diff --git a/.superpowers/sdd/evidence-task-5c-report.md b/.superpowers/sdd/evidence-task-5c-report.md new file mode 100644 index 00000000..556bd02b --- /dev/null +++ b/.superpowers/sdd/evidence-task-5c-report.md @@ -0,0 +1,188 @@ +# Evidence Task 5C report + +## Delivered + +- Added `vector.retain_published_generations` (default `3`, validation minimum `1`). +- Retention runs only after publication. It keeps ACTIVE, the newest configured generations, + and generations referenced by running or resumable failed job checkpoints. +- Cleanup deletes the exact Evidence generation from the vector store before removing its + immutable filesystem directory. Vector failures retain filesystem metadata for retry and + produce credential-free partial reports. +- Added idempotent `tht preprocess evidence gc [--dry-run] --json` reconciliation with pristine + JSON output. +- Materialized document reads now open generation/documents components with directory file + descriptors and `O_NOFOLLOW`, require a regular file owned by the process with one link, and + hash the bytes read from the same descriptor against the canonical manifest. +- HTTP generation deletion is pinned to `delete_vector_generation` with exact + table/kind/generation arguments. Legacy 404 responses fail closed with an actionable, + sanitized migration message. + +## Evidence + +- Focused retention, safe-read, CLI, and HTTP contract tests: `51 passed` (Docker-backed direct + parametrizations excluded from that focused invocation). +- Real Docker pgvector adapter suites: `33 passed`. +- Full harness suite, including Docker-backed tests: `668 passed, 5 deselected`. +- Changed-file Ruff: clean. +- `git diff --check`: clean. + +The five deselected tests are the repository's opt-in `l2` tests requiring external services; +they are not local pgvector tests. Test output retains pre-existing Pydantic serialization and +legacy-config deprecation warnings. + +## Review fix wave + +- Publication is now explicit and durable (`PUBLISHED` marker). Retention candidates require a + valid generation manifest and publication marker (ACTIVE remains backward-compatible), so + staged and malformed directories neither consume retention slots nor become deletion targets. +- The policy retains ACTIVE plus exactly `N-1` newest rollback publications, ordered by durable + publication time and generation id. Running and failed-resumable JobRunner checkpoints protect + every referenced plan generation. +- `VectorStore` now exposes exact Evidence generation inventory. Direct pgvector uses a constrained + `SELECT DISTINCT` over `kind='evidence'` and `metadata.vector_generation`; HTTP uses the + allowlisted `list_evidence_generations` RPC and fails closed on legacy 404. The writer RPC SQL, + revokes, and grants are packaged in `create_vector_writer_rpc.sql`. +- Explicit GC reconciles the union of published filesystem generations and vector-only orphans, + preserving vector-before-filesystem deletion and retry semantics. +- `run_as_job` holds the same corpus writer lock across checkpoint recovery, staging, publish, and + retention. Explicit GC already uses this lock, serializing candidate snapshots with publishers. +- Session artifact consumers no longer receive the corpus source path after validation. They get + an owned, read-only copy atomically written from the bytes read and hash-validated on the same + descriptor. + +Fresh verification after the fix wave: full harness `672 passed, 5 deselected`; Docker pgvector, +HTTP parity, and migration suites `43 passed`; exact direct inventory/delete integration `1 passed`; +changed-file Ruff and `git diff --check` clean. + +## Final hardening verification + +- Canonical generation validation is exact (`^gen:[0-9a-f]{32}$`) before HTTP/direct deletion; + malformed HTTP inventory rows fail closed rather than entering the GC candidate set. +- Added explicit protection coverage for running and failed-resumable JobRunner checkpoints, plus + a second-GC idempotence assertion for vector-only orphan reconciliation. +- Added deterministic concurrent locking coverage: a job paused after discovery retains the corpus + writer lock, explicit GC blocks, then completes after publication without deleting the active run. +- Added a descriptor-race regression: replacing the corpus pathname immediately after `read(2)` + leaves the atomically materialized session-owned copy byte-for-byte equal to the validated ACTIVE + document and its manifest hash. + +Final fresh evidence: Docker pgvector/HTTP/migration suites `48 passed`; full harness `680 passed, +5 external L2 deselected`; changed-file Ruff and `git diff --check` clean. + +## Integrated Task 5 dependency fixes + +- GC now distinguishes filesystem retention from vector dependencies. ACTIVE and the newest + `N-1` published manifests keep their directories; every exact generation in their + `document_generations` maps remains vector-protected even after its old publication directory is + evicted. Job-protected manifests receive the same dependency treatment. +- The real four-publication Docker lifecycle now includes an unchanged document whose vectors come + from the first generation. With retention `N=2`, only the final two publication directories remain + while the first generation's vectors remain searchable from ACTIVE and survive restart/explicit GC. +- Evidence lookup is always wrapped by the ACTIVE-aware searcher. With no corpus/ACTIVE, Evidence + returns no rows and search packs cannot expose legacy vectors; non-Evidence kinds are unchanged. +- Session artifact resolution holds the corpus writer lock, snapshots the active manifest once, and + materializes bytes using that exact `manifest_id`, preventing a concurrent publish/retain-1 GC from + changing or deleting the selected source generation. + +Focused unit tests, the updated real Docker lifecycle, changed-file Ruff, and `git diff --check` pass. +The final full harness invocation completed with exit code 0, including the concurrently added DWH +JobRunner tests. + +## Final ACTIVE search review fixes + +- `ActiveEvidenceSearcher` now treats default (`kinds=None`) and mixed-kind searches as explicit + split queries: non-Evidence kinds are queried separately, while Evidence is queried only with + ACTIVE manifest generation/document predicates applied server-side before every limit. +- Results are merged deterministically by descending similarity then stable id and truncated once + to the caller's global `top_n`. Pure non-Evidence searches retain their original delegate path. +- The corpus writer lock now covers manifest snapshot construction and all corresponding vector + queries, preventing retain-1 publication/GC from switching or deleting generations mid-search. +- Removed the public post-LIMIT `active_evidence_hits` helper; no public Evidence path performs + client filtering after limit. + +Focused default/mixed/no-ACTIVE/search-pack tests pass, the real Docker pgvector lifecycle passes, +and the final full harness plus scoped Ruff/diff invocation completed with exit code 0. + +## Workspace-scoped Evidence isolation + +- Evidence manifests, vector metadata, and record keys now carry the stable JobRunner workspace id + derived from the configured workspace identity (config stem), never credentials or absolute paths. +- Every ACTIVE server-side predicate includes `workspace_id`. Legacy unscoped rows therefore fail + closed and cannot appear in Evidence results. +- Vector generation inventory and deletion require the workspace namespace across the port, direct + pgvector adapter, HTTP client/adapter, and allowlisted RPC SQL. Legacy unscoped RPC overloads are + explicitly dropped during migration; destructive SQL matches collection, kind, generation, and + workspace together. +- GC recovers the persisted namespace from ACTIVE for explicit/restarted cleanup and can only list + or delete that workspace's generations. Real shared-pgvector coverage proves deleting a generation + for workspace A preserves the same generation in workspace B. +- `PipelineResult.model_dump` now serializes fields explicitly instead of `dataclasses.asdict`, + avoiding deepcopy of immutable `FrozenDict` metadata while preserving pristine JSON CLI output. + +Final focused verification: `89 passed` across corpus/CLI JSON, direct/HTTP parity, migrations, and +real Docker pgvector lifecycle; scoped Ruff and `git diff --check` clean. A contemporaneous full-suite +run reached unrelated Task 6 immutable-file tamper tests; those files were deliberately not changed. + +## Immutable corpus/workspace binding + +- A corpus root becomes bound to the workspace id persisted in its ACTIVE manifest. Job, non-job, + explicit GC, and ACTIVE search entry points compare the configured namespace before discovery, + vector access, staging, deletion, or ACTIVE mutation. +- Reusing the same paths after renaming a workspace now fails closed with a typed/sanitized message: + use a new corpus root or perform an intentional explicit rebuild. Unscoped legacy manifests also + fail this ownership check. +- Tests prove unchanged-document reuse cannot silently mix workspace A vectors into a workspace B + manifest, and that mismatched job, GC, and search paths perform no vector/filesystem mutations. + +Focused workspace-binding, search-pack, preprocess JSON, and scoped Ruff/diff tests pass. + +Compatibility follow-up: direct/internal `CorpusPipeline` instances now distinguish an omitted +workspace identity from an explicit config/job identity. An unbound instance adopts the persisted +ACTIVE owner (or `default` only for a brand-new direct corpus), preserving safe resume/GC tests and +the real pgvector lifecycle. Explicit config/job identities still fail closed on any mismatch. The +two reported regressions, workspace mismatch guards, real Docker lifecycle, scoped Ruff/diff, and +the full harness suite all pass. + +Final fail-closed follow-up: persisted ACTIVE ownership is now validated under the corpus lock before +every configured search delegate, including default, mixed, pack, and non-Evidence-only operations. +Malformed or missing `metadata.workspace_id` is intrinsically rejected even for unbound direct +callers; source discovery, vector operations, GC, files, and ACTIVE remain untouched. Focused tests, +real Docker lifecycle, scoped Ruff/diff, and the full harness regression run pass. + +Final lock/preflight follow-up: `CorpusPipeline.gc()` now acquires the corpus writer lock itself for +ownership validation through vector/filesystem cleanup. The store lock is thread-reentrant so nested +job retention is safe without weakening cross-thread/process exclusion; the CLI wrapper no longer +double-locks. Search find/pack performs locked corpus ownership preflight immediately after config +load, before DWH leasing, vector/searcher factories, embeddings, or schema work. Focused concurrency +and fail-closed tests, real Docker lifecycle, scoped Ruff/diff, and the full harness pass. + +## Compact public Evidence reports + +- Public `PipelineResult.model_dump()` is now a bounded operational envelope: terminal status, + run/resume/publication/generation/manifest identifiers, capped changed/unchanged/removed source + identifiers, and aggregate document/chunk counts. Full manifests, bodies, and metadata remain + internal/on disk and are never serialized to CLI stdout. +- `tht preprocess evidence` exits `1` for any durable terminal status other than `succeeded` in + both JSON and text modes. JSON stdout remains one pristine sanitized object; text mode emits one + compact stderr error without traceback, exception identity, evidence content, or credentials. +- Tests cover a real failed acquisition job, sensitive evidence content, capped thousand-item + summaries, bounded report size, and smoke-compatible changed/unchanged fields. + +Focused tests and scoped Ruff/diff pass. The contemporaneous full suite reaches an unrelated Task 6 +DWH snapshot fixture missing its newly required workspace identity. + +### Safe result representation and exact text totals + +- `PipelineResult.manifest` is explicitly excluded from dataclass representation and the custom + representation is fixed-size operational data only. It omits manifest ids, documents, chunks, + content, metadata, and errors; `str(result)` inherits the same safe representation. +- Text-mode Evidence success output reads the uncapped aggregate totals from `payload["counts"]` + rather than the intentionally capped identifier arrays. +- Regression coverage builds a thousand-document/chunk manifest containing content and + credential-like metadata secrets, checks bounded `repr`/`str`, and verifies exact totals above + the 100-item public-array cap. + +Focused Evidence verification passes (`67 passed`), and scoped Ruff is clean. The full harness run +is not green in this sandbox: Docker-backed tests cannot access the daemon, wheel packaging cannot +use the restricted build environment, and concurrent Task 6 DWH binding changes currently fail two +DWH tests. None of those failures touch the Evidence files in this follow-up. diff --git a/.superpowers/sdd/evidence-task-5d-report.md b/.superpowers/sdd/evidence-task-5d-report.md new file mode 100644 index 00000000..9d87b141 --- /dev/null +++ b/.superpowers/sdd/evidence-task-5d-report.md @@ -0,0 +1,50 @@ +# Evidence Task 5D — Real pgvector lifecycle gate + +## Status + +Complete. The Docker-backed L0 gate uses one persistent `pgvector/pgvector:pg16` +database and the production migrations, direct reader/writer `PgVectorStore`, +`CorpusStore`, `CorpusPipeline.run_as_job`/JobRunner, ACTIVE Evidence retrieval, +search-pack fusion, owned session artifact copy, retention, and explicit GC. + +## Lifecycle covered + +- Four real corpus publications with retention set to two generations. +- A higher-similarity stale vector proves ACTIVE metadata filtering happens before LIMIT + for normal Evidence retrieval and the search-pack fusion path. +- A removed source is absent from ACTIVE retrieval and cannot be copied to a session. +- An injected process death occurs after one real committed vector upsert. Resume uses the + real run ID, preserves that record, fills the missing records, and produces no duplicate keys. +- Database engines and direct store objects are disposed/recreated before persisted ACTIVE + retrieval is checked again. +- An exact canonical vector-only orphan generation is discovered and removed by explicit GC. +- Filesystem and vector inventories converge exactly to ACTIVE plus one rollback; a second GC + is a no-op. +- Owned session artifact bytes and SHA-256 match the ACTIVE canonical document. + +## Production bug found and fixed + +Production migration `003_roles.sql` intentionally restricted `vector_writer`, but omitted +the privileges used by the production generation lifecycle: `SELECT(metadata)` for inventory +and `DELETE` for cleanup on `vectors.evidence`. Consequently a real job published successfully +and then failed in `retention_cleanup` on its first run. + +Added versioned migration `004_evidence_generation_gc.sql` granting only those two Evidence +generation-management privileges. Runtime application code was not redesigned. + +## Verification + +- Target lifecycle: `1 passed` (Docker-backed). +- Full harness: `681 passed, 5 deselected`. +- Scoped Ruff: passed. +- `git diff --check`: passed. + +The existing Pydantic serialization and legacy-workspace deprecation warnings remain unchanged. + +## Follow-up assertion correction + +The removal phase now retains the removed canonical document ID/ref before publication and +asserts both fields are absent from post-resume ACTIVE Evidence hits. It reruns the real +search-pack fusion after removal, proves active fourth-generation content is positively +returned in both paths, and proves the removed content remains absent. The owned session +artifact lookup for the retained removed ID remains empty. diff --git a/.superpowers/sdd/evidence-task-6-report.md b/.superpowers/sdd/evidence-task-6-report.md new file mode 100644 index 00000000..b1d347ab --- /dev/null +++ b/.superpowers/sdd/evidence-task-6-report.md @@ -0,0 +1,49 @@ +# Evidence Task 6 — final fd-anchored DWH correction + +All DWH generation state below `.tht-dwh` is now accessed relative to the directory descriptor +retained by the shared/exclusive generation lease. ACTIVE reads, atomic temp writes, replacement, +fsync, and rollback use `openat`/`replaceat` operations. Generation staging, validation, +reconciliation, resume checks, retention classification, and recursive deletion likewise use owned +root/generations/candidate descriptors with `O_NOFOLLOW`; locked operations no longer reopen +generation paths through `workspace_root`. + +Portable reader snapshots are copied from validated generation file descriptors into private 0700 +process-owned temporary directories while the shared lease is held. This avoids Linux-only +`/proc/self/fd` paths and prevents a renamed/replaced `.tht-dwh` pathname from redirecting later +schema or LSH reads. Lease-scoped copies are removed on exit and standalone snapshots are removed +at process exit. + +Deterministic adversarial tests rename the DWH root after lease acquisition during ACTIVE reads, +ACTIVE publication, and retention cleanup. Each test proves the replacement tree is never read, +written, or deleted; the descriptor-pinned original either completes consistently or fails closed. +Existing owner binding, legacy rejection, crash reconciliation, resume, atomic rollback, retention, +and reader/writer exclusion behavior remains covered. + +## Final review correction + +Snapshot materialization now reads the manifest and every owned artifact exactly once through the +already-open generation descriptor, validates each hash against those exact bytes, and writes the +same byte objects to the private snapshot. A deterministic second-read mutation test proves hostile +pickle bytes can neither pass validation nor enter the snapshot. Reconciliation closes the ACTIVE +generation descriptor in a `finally` block on matches, mismatches, and exceptions. Pipeline-owned +snapshot directories are removed and deregistered after `run_job` on both successful and failed +runs, preventing repeated pipeline use from accumulating temporary directories or registry entries. + +The cleanup boundary now begins immediately after snapshot materialization. Resume checkpoint +validation and `JobSpec` construction are guarded by the same release routine as `run_job`, so +corrupt/mismatched resume state or constructor failure clears the pipeline holder, removes the +private directory, and restores the snapshot registry to its prior state before propagating. + +## Shipped preprocessing startup contract + +Local-vector preprocessing now uses a dedicated Compose override. Both one-shot jobs depend on a +successfully completed `vector-migrate`, whose transitive chain waits for database health and role +reconciliation. The generic preprocessing overlay remains independently renderable and contains no +local-vector services or password secrets. README commands include the local override and build the +job image before running. + +The real clean-project smoke no longer injects dependencies or manually starts, reconciles, or +migrates PostgreSQL. Its first shipped `compose run preprocess-evidence` demonstrably creates the +database, waits for health, runs reconciliation and migration, then runs the Evidence job. Unchanged +rerun, changed-source publish, DWH preprocessing, ACTIVE verification, and injected-failure cleanup +all pass through the same shipped dependency path. diff --git a/.superpowers/sdd/evidence-task-7-report.md b/.superpowers/sdd/evidence-task-7-report.md new file mode 100644 index 00000000..2c217cd2 --- /dev/null +++ b/.superpowers/sdd/evidence-task-7-report.md @@ -0,0 +1,93 @@ +# Evidence preprocessing Task 7 report + +Implemented the S3-compatible Evidence adapter, explicit preprocessing Compose overlay, and +operational gates. + +- S3 discovery uses bounded paginator pages, page size, and total objects; acquisition enforces a + byte ceiling and always closes streaming bodies. +- Provenance is canonical `s3://bucket/key`. Versioned objects use `s3-version:`; + unversioned objects use a hashed exact ETag, and acquisition refuses validator drift. +- The adapter uses boto3/botocore rather than custom signing. TLS verification is enabled by + default. Custom HTTP and private endpoints require independent explicit opt-ins; endpoint + userinfo is rejected and public custom endpoints are DNS-policy checked. +- Access, secret, and session credentials support file-secret resolution into masked `SecretStr` + config fields. They are never emitted in provenance, reports, errors, or Compose environment. +- `deploy/compose.preprocess.yaml` provides separate one-shot Evidence and DWH jobs and is inert + unless explicitly included with the `preprocess` profile. +- `scripts/preprocess-smoke.sh` verifies both services render without secret material and pins an + unchanged rerun plus a modified generation through deterministic pipeline tests. + +Verification: focused S3/HTTP/filesystem/config tests 34 passed; operational smoke 2 passed; core +image with locked boto3 extra built; full harness 702 passed, 5 deselected; scoped Ruff and diff +checks passed. + +Operational risk: custom S3-compatible endpoints remain part of the deployment trust boundary. +Private endpoint access must be explicitly enabled and should be restricted by container egress +policy in production. S3 list consistency semantics are provider-defined; version IDs are preferred +over ETags wherever bucket versioning is available. + +## Review correction + +The Compose overlay now uses committed, purpose-built Evidence and DWH workspace files with +job-specific dependencies. Its services create their lock roots and mount only the vector secrets +they consume. The operational smoke is a real isolated Compose project: real pgvector migrations, +a deterministic in-project embeddings endpoint, actual Evidence CLI JSON across initial/unchanged/ +mutated runs, exact ACTIVE verification, an actual DWH introspection job, and owned cleanup. + +S3 custom endpoints now fail closed unless declared trusted; HTTP and private loopback endpoints +need additional independent opt-ins. Boto uses forced path-style addressing. Custom endpoints reject +userinfo, query, fragment, and non-root paths. Buckets use strict DNS syntax; listed keys must remain +under prefix and within the S3 byte bound; validators must be nonempty/bounded. Because +ListObjectsV2 does not provide version IDs, discovery honestly fingerprints the exact ETag and +acquisition rejects ETag drift. + +Final correction verification: S3/config focused 20 passed; full harness 721 passed, 5 deselected; +real Compose smoke and image build passed; scoped Ruff, shell syntax, and diff checks passed. + +## Final security review correction + +Literal non-global IPv4/IPv6 endpoints now require the private-endpoint opt-in without claiming DNS +pinning for hostnames. Pagination uses explicit continuation requests and never fetches page +`max_pages + 1`. IP-shaped buckets, leading-slash prefixes, empty/overlong/control-character keys, +and absent validators fail closed. Acquisition accepts only the exact stored `SourceObject` and +compares the response ETag with the stored discovery validator. The real smoke snapshots generation +directory counts after every run and has an injected-failure cleanup mode; cleanup fails if Compose +down fails or any owned container, volume, or network remains. + +The canonical smoke correction counts only root-level `corpus/gen-<32 hex>` directories. It exposed +that the durable job path still published an empty unchanged generation; the pipeline now returns +the existing ACTIVE generation without staging a directory when compatibility and all source +fingerprints are unchanged. The smoke therefore proves directory deltas `+1`, `+0`, `+1`. +Failure injection runs a real exit-97 command after resources exist and reaches the EXIT trap. +Cleanup aggregates Compose-down, residual container/volume/network, and temp-directory failures +while preserving the original failure status. S3 prefixes are validated before any client request +for leading slash, UTF-8 byte length, controls, and DEL. + +## Canonical unchanged-run correction + +The durable job now persists a deterministic source snapshot keyed by source identity. Each entry +binds canonical URI, exact source fingerprint, UTC modification time, canonical immutable metadata, +and explicit media type and size contract fields. The manifest also binds document-to-source +provenance, supplied config/input fingerprints, compatibility, embedding settings, and pipeline and +chunk-policy versions. + +An unchanged run reuses ACTIVE only when ownership, bindings, the complete snapshot, document +provenance, materialized document hashes, and every required vector ID/content hash match exactly. +Snapshot changes rebuild only the affected sources; job input/config changes publish a new manifest +while retaining valid stable vector-generation dependencies. Missing or corrupt legacy contract +metadata, documents, or vectors fails closed and rebuilds. The Compose smoke now explicitly expects +the unchanged no-op to report `published=false` while proving generation deltas `+1`, `+0`, `+1`. + +## Corrupt ACTIVE reconstruction correction + +ACTIVE reuse now reconstructs each source contract from the persisted discovery snapshot and checks +the deterministic document identity, canonical URI, source fingerprint, UTC modification time, +source metadata, applicable media type, content hash, and pipeline identity against the owned +materialized document. The persisted document-source map carries the same exact binding. + +Chunks are recomputed under the current chunk policy and must match the manifest exactly in count, +order, IDs, ordinals, content, hashes, linkage, provenance, and policy metadata. Vector health must +report the configured dimension, and every recomputed chunk must have its generation-scoped vector +ID with the exact content hash. Missing, altered, or extra chunks and corrupt document or vector +contracts therefore disable the no-op and rebuild, while a valid unchanged run still performs no +source acquisition. diff --git a/.superpowers/sdd/model-provider-credential-report.md b/.superpowers/sdd/model-provider-credential-report.md new file mode 100644 index 00000000..7fcdc16c --- /dev/null +++ b/.superpowers/sdd/model-provider-credential-report.md @@ -0,0 +1,16 @@ +# Model provider credential boundary + +The backend accepts only an absolute `THT_MODEL_API_KEY_FILE` reference. `PiProcessManager` reads +and validates it afresh before each hosted-provider spawn, rejects symlinks, non-regular/hard-linked, +empty, whitespace-containing, oversized, unreadable, or permissively-mode files, and accepts Docker +0444 secrets only beneath `/run/secrets`. Failures are sanitized and occur before child creation. + +Provider names are normalized and mapped to Pi-recognized variables. The child environment removes +the generic path, deprecated `PI_PROVIDER_API_KEY`, and all unselected known provider keys before +injecting only the selected key. Values never enter argv, settings, health, or diagnostics. Local +providers remain keyless and unknown hosted providers fail closed. + +The production Compose overlay mounts `model_api_key` read-only and points the backend at its file; +the deployment render smoke proves the value is absent from rendered configuration. Entrypoint, +root README, Pi configuration guide, environment example, and secrets operator guide document the +new contract and reject the legacy generic value variable. diff --git a/.superpowers/sdd/pgvector-final-fix-report.md b/.superpowers/sdd/pgvector-final-fix-report.md new file mode 100644 index 00000000..e33ee639 --- /dev/null +++ b/.superpowers/sdd/pgvector-final-fix-report.md @@ -0,0 +1,59 @@ +# Local pgvector whole-plan final fix report + +## Outcome + +All four binding final-review findings are closed. + +1. `PgVectorStore.health()` checks namespace `USAGE` independently for reader and writer + before inspecting vector types. Real PostgreSQL tests revoke only schema `USAGE`, prove both + health sides false and operations unavailable, then grant it back and prove recovery. +2. Direct reader/writer passwords use workspace `password_file` references. Compose mounts the + two files read-only into core and exposes only `_FILE` paths. Rendered Compose and live + `docker inspect` checks prove secret contents are absent. +3. Direct search failures map to `VectorReadUnavailable`; hash/upsert failures map to + `VectorWriteUnavailable`. Messages are fixed and sanitized, original exceptions remain chained, + and upsert rollback is preserved. +4. The shared secret policy uses Linux `stat -c` with macOS `stat -f` fallback. Host files permit + only `0600`/`0400`; Docker's read-only `0444` is accepted only beneath `/run/secrets`. Tests and + operator docs pin this exact policy. + +## TDD evidence + +The new config, mode, schema-usage, unavailable-connection, and permission regressions failed +before their implementations. The first live secret-policy run also caught GNU `stat -f` accepting +an incompatible format invocation; detection now tries the native Linux form first. The next live +run caught smoke-generated rotation fixtures at `0644`; fixtures now model the documented host +policy. + +## Verification + +- Real direct pgvector + HTTP parity: `31 passed`. +- Full harness from `harness/`: `493 passed, 5 deselected`. +- Live `local-vector` rotation, restart persistence, inspect boundary, and backup/restore: pass. +- Core image vector migration discovery/status smoke: pass. +- External and local Compose deployment security contracts: pass. +- Config/port focused suite: `26 passed`. +- Secret policy, bootstrap rotation, and backup/restore safety scripts: pass. +- Changed Python Ruff, shell syntax, and `git diff --check`: pass. + +One attempted full-harness invocation from the repository root produced a path-dependent failure +in an existing test that opens `workflow.yaml` relative to CWD. It was immediately rerun using the +documented `cd harness && .venv/bin/pytest -q` command and passed completely. + +## Operational notes + +Workspace files contain file paths, never direct passwords. Secret contents necessarily exist in +the in-process validated `DatabaseConfig` used to establish PostgreSQL connections, but are not +serialized by doctor/Compose/inspect paths. Docker Desktop file-backed secrets may appear as bind +mounts; the safe runtime exception is therefore based on the read-only service mount location +`/run/secrets`, while source files remain owner-only on the host. + +## External-profile regression follow-up + +Local pgvector is now an explicit `deploy/compose.local-vector.yaml` overlay. The base Compose and +production external override contain no direct vector password declarations, mounts, or `_FILE` +variables, so external deployments do not resolve or require local password files. A real lifecycle +gate unsets all local secret-file variables, renders external config, builds and starts core, waits +for health, and inspects the live container for absence of local direct-vector secret paths. The +local overlay retains its live inspect assertion (paths present, values absent), rotation, restart +persistence, and transactional backup/restore drill. diff --git a/.superpowers/sdd/pgvector-task-1-report.md b/.superpowers/sdd/pgvector-task-1-report.md new file mode 100644 index 00000000..f2025c83 --- /dev/null +++ b/.superpowers/sdd/pgvector-task-1-report.md @@ -0,0 +1,95 @@ +# Local pgvector Task 1 report + +## Status + +Implemented the direct `PgVectorStore` behind the transport-neutral `VectorStore` port. +The adapter uses separate optional reader and writer database configurations, derives +capabilities from configured authority, validates strict positive search limits, filters kinds +in SQL before limiting, and merges multi-collection results by cosine similarity. + +All collection identifiers are selected from the fixed `schema_records`, `evidence`, and +`memory` allowlist and composed with `psycopg2.sql.Identifier`. Values, vectors, kinds, hashes, +and limits remain bound parameters. Collection/kind mismatches fail with `VectorStoreError`. + +Upserts preserve the canonical metadata shape, use `record_key` conflict semantics, update the +transport hash and embedding, and leave semantic metadata fields intact. Health probes reader +and writer independently and reports observed `vector(N)` dimensions against the configured +embedding dimension. + +## Configuration and factory + +`pgvector_direct` now accepts explicit optional `reader` and `writer` `DatabaseConfig` entries. +The former `connection` entry remains supported as a deprecated read-only compatibility path. +`build_vector_store(..., require_write=True)` accepts writer-only direct configurations and +fails early when no explicit writer is present. + +The transitional `build_vector_loader` bulk-sync path remains in place. It uses an explicit +direct writer when present, or the legacy `connection`; it deliberately does not treat a new +reader-only credential as writable. No production schema migration was added. + +## TDD and verification + +- RED: the new tests initially failed at collection because `PgVectorStore` did not exist. +- Docker L0 pgvector tests: `11 passed`. +- Direct + HTTP parity/factory/config focus: `51 passed`. +- Full harness: `461 passed, 5 deselected`. +- Changed-file Ruff lint: clean. +- Changed-file Ruff format check: clean. +- `git diff --check`: clean. + +The repository-wide `ruff check .` still reports 34 pre-existing test-file findings outside +Task 1; none are in changed files. The full pytest suite emits 17 existing legacy-config +deprecation warnings. + +## Scope and concerns + +- Test fixtures create only the three existing vector tables needed to exercise the adapter; + migration/versioning remains Task 2. +- The legacy single `connection` form stays read-only through the public port, matching its + previous adapter behavior, while remaining available to the explicitly documented bulk-loader + transition. + +## Review fix wave + +The Task 1 review findings were addressed in a follow-up TDD cycle: + +- Search now validates requested kinds against the global known-kind set, intersects valid kinds + with each collection, and skips unrelated collections. A direct-versus-HTTP parity test covers + the multi-collection case. +- Health requires all three allowlisted tables, an `embedding vector(N)` column on every table, + the expected dimension on every table, and the appropriate read or write table privileges for + each configured side. Empty and partial schemas return deterministic, credential-free details; + unexpected database failures expose only their exception class. +- The Docker L0 fixture now provisions separate least-privilege reader and writer roles. Tests + prove the reader cannot insert, the writer cannot execute the cosine-search SELECT, and the + adapter still routes search to the reader and upsert/hash operations to the writer. Direct + upsert uses an atomic `INSERT ... ON CONFLICT DO NOTHING` followed by `UPDATE` for an existing + key, avoiding broad SELECT authority while retaining conflict-safe hash/upsert semantics. + +Fresh verification after the fix wave: + +- Docker L0 + HTTP port/search parity: `42 passed` (earlier checkpoint); the final L0 file has + `16 passed` including the stricter raw-role search denial. +- Expanded focused adapter/config suite: `56 passed`. +- Full harness: `466 passed, 5 deselected`. +- Changed-file Ruff lint/format and `git diff --check`: clean. + +## Sequence privilege health follow-up + +Writer health now resolves the real serial/identity sequence for the `id` column of every +required collection using `pg_get_serial_sequence`. It requires `USAGE` on each resolved +sequence, which is the privilege used by the adapter's implicit `nextval`; sequence `SELECT` is +not required because no adapter operation reads sequence state. + +The Docker fixture includes a writer role with complete table/hash-column authority but no +sequence grant. Its health is deterministically unhealthy and a new-key upsert fails. Granting +only sequence `USAGE` makes health green and the same port upsert succeeds. Sequence discovery is +guarded for partial schemas so a missing `id` column produces the existing sanitized schema +diagnostic instead of a PostgreSQL error. + +Fresh verification for this follow-up: + +- Docker pgvector L0 after formatting: `17 passed`. +- Expanded focused adapter/config/parity suite: `57 passed`. +- Full harness: `467 passed, 5 deselected`. +- Changed-file Ruff lint/format and `git diff --check`: clean. diff --git a/.superpowers/sdd/pgvector-task-2-report.md b/.superpowers/sdd/pgvector-task-2-report.md new file mode 100644 index 00000000..6fcac409 --- /dev/null +++ b/.superpowers/sdd/pgvector-task-2-report.md @@ -0,0 +1,82 @@ +# Local pgvector Task 2 report + +## Outcome + +Implemented ordered, idempotent production migrations and the `tht vector migrate` +interface, including `tht vector migrate --status --json` with pristine JSON output. + +## Implementation + +- `001_extensions.sql` installs pgvector. +- `002_schema_tables.sql` creates `vectors.schema_records`, `vectors.evidence`, and + `vectors.memory` with the `VectorWriteRecord` columns and `vector(768)` embeddings. +- `003_roles.sql` creates passwordless `NOLOGIN` reader/writer roles. Deployments inject + credentials (or grant these roles to separately-created login roles); no production secret + is stored in the repository. +- Reader authority is schema usage plus table `SELECT`. +- Writer authority is schema usage, table `INSERT`/`UPDATE`, narrow hash-probe column `SELECT`, + and sequence `USAGE`. It has no `DELETE`, broad row `SELECT`, DDL, or ownership authority. +- The migration runner discovers ordered SQL files, records SHA-256 checksums in + `public.tht_vector_migrations`, serializes runners with a transaction-scoped advisory lock, + and applies the full pending batch in one transaction. +- Status distinguishes applied, pending, and checksum-drifted migrations. Apply refuses drift. + A failed migration rolls back both prior migrations in that batch and ledger writes. + +## TDD evidence + +RED was observed with a real `pgvector/pgvector:pg16` testcontainer: 6 failures for the missing +module, missing command, and missing schema. + +GREEN verification: + +- Focused migration + direct adapter integration: `23 passed`. +- Full harness from the documented `harness/` cwd: `473 passed, 5 deselected`. +- Targeted Ruff (`tht` plus the new L0 test): clean. +- `git diff --check`: clean. + +The new L0 coverage exercises clean install, idempotent rerun, pristine JSON status, checksum +drift, transaction rollback, exact tables/columns/dimensions, role isolation, sequence authority, +and the real `PgVectorStore.health()` plus `VectorWriteRecord` upsert path. + +## Existing repository lint baseline + +The requested full `ruff check .` was run. It reports 34 pre-existing violations in unrelated +test files (unused imports and one-line semicolon statements). None are in Task 2 files; changing +them would exceed this task's scope. The complete harness test gate is green. + +## Self-review + +No unresolved Task 2 correctness concern found. One deliberate contract choice is worth noting: +writer `INSERT` and `UPDATE` are table-level because the approved direct adapter health probe uses +`has_table_privilege` for those authorities. Least privilege is retained by withholding broad +`SELECT`, `DELETE`, DDL, ownership, and credentials. + +## Review fix wave + +The post-implementation review found four production-boundary gaps. They are fixed as follows: + +- Migration SQL now ships inside the `tht` wheel (`tht/migrations/vector`) via explicit + setuptools package-data and is discovered through `importlib.resources`, rather than relying on + a source-checkout-relative directory. +- Both status and apply reject ledger versions absent from the installed manifest, including + nonnumeric future version labels. This treats a binary/database downgrade as drift instead of + silently reporting a healthy state. +- Migration files are ordered by parsed integer version; spellings such as `2` and `02` are + rejected as duplicate versions. +- Every migration transaction pins `search_path` locally to `pg_catalog, pg_temp`; catalog calls + and the ledger are schema-qualified. pgvector is installed into the locked `vectors` schema, + tables use `vectors.vector`, and `PgVectorStore` qualifies vector casts and the cosine operator. + A hostile admin default path with a writable shadow schema cannot redirect migration objects. +- The core image build asserts CLI discovery. Image verification now starts an ephemeral pgvector + database, runs the installed image's migration command, and compares pristine apply/status JSON. + +Additional verification after the fix wave: + +- Focused migration, adapter, hostile-path, and wheel suite: `27 passed`. +- Full harness: `477 passed, 5 deselected`. +- Production core image build: passed, including build-time CLI discovery. +- Core-image apply/status smoke against `pgvector/pgvector:pg16`: passed. +- Changed production and test files: Ruff clean; `git diff --check` clean. +- Full Ruff remains at the same 34 pre-existing unrelated test-file findings documented above. + +No dependency changed, so the committed Python requirements lock did not require regeneration. diff --git a/.superpowers/sdd/pgvector-task-3-report.md b/.superpowers/sdd/pgvector-task-3-report.md new file mode 100644 index 00000000..32199b42 --- /dev/null +++ b/.superpowers/sdd/pgvector-task-3-report.md @@ -0,0 +1,133 @@ +# Task 3 report — optional local pgvector profile + +## Status + +Implemented and verified the `local-vector` Compose profile. + +- `vector-db` uses pgvector 0.8.5 on PostgreSQL 16, pinned to the official multi-arch + manifest digest. +- `vector_data` is a project-scoped named volume and is not shared with application data. +- database readiness gates the packaged one-shot `vector-migrate` job; core declares the + migration completion dependency while remaining usable in the pre-existing external profile. +- bootstrap, migrator, reader, and writer identities are distinct. Bootstrap and migration + credentials are supplied as Compose secrets; the application receives only reader/writer + credentials. +- `deploy/workspaces/local-vector.yaml` selects `pgvector_direct` with separate reader and + writer connections. +- the base loopback port binding, `AUTH_MODE=none`, and `THOTH_PUBLIC_EXPOSURE=false` defaults + are unchanged. + +## Red/green evidence + +The initial Compose contract did not list `vector-db`, as required by the brief. The first real +smoke then failed migration 002 because bootstrap installed the vector extension in `public`. +The bootstrap was corrected to create the `vectors` schema under the migration owner and install +the extension there. A clean-volume rerun passed. + +## Verification + +- `./scripts/local-vector-smoke.sh`: PASS + - isolated generated Compose project and credentials + - clean migration plus idempotent status rerun + - reader/writer privilege health + - one-record upsert and similarity search + - restart of both `core` and `vector-db` + - persisted search result after restart + - project-only volume cleanup +- `./scripts/test-container-deployment.sh`: PASS +- `./scripts/test-backend-url-policy.sh`: PASS +- `docker compose --profile local-vector config --quiet`: PASS +- harness: 477 passed, 5 deselected +- backend: 84 passed; TypeScript typecheck PASS +- frontend: 226 passed; TypeScript typecheck PASS +- `git diff --check`: PASS + +## Self-review / concerns + +- Compose cannot make a dependency required only under one profile. The core dependency uses + `required: false` so the established `external` profile does not activate local infrastructure; + under `local-vector`, `compose up --wait` still fails if `vector-migrate` exits nonzero, and the + smoke verifies that successful migration precedes the healthy stack. +- Reader/writer passwords are injected into core environment variables because Compose service + attributes cannot be conditional by profile. Bootstrap and migrator credentials remain + file-backed secrets and are never exposed to core. +- The smoke intentionally refuses the operator project name `thothii` and removes only its unique + project namespace and volumes. + +## Follow-up hardening — credential reconciliation and cleanup ownership + +Review findings were resolved in a separate follow-up: + +- Replaced fresh-volume-only initialization with `vector-reconcile`, an idempotent one-shot that + runs after database health and before `vector-migrate`. It authenticates with only the bootstrap + admin secret, safely creates missing identities, reconciles role attributes and passwords on + existing volumes, restores memberships/ownership, and leaves vector data untouched. +- The migrator is explicitly `NOSUPERUSER NOCREATEDB NOCREATEROLE`. Schema/database ownership is + sufficient for all packaged migrations because reconciliation creates the two group roles first. +- The live smoke rotates migrator, reader, and writer secrets on the same populated volume, rejects + the old reader credential, reruns migrations, recreates core with the new runtime credentials, + and retrieves the record written before rotation and again after database/core restart. +- Smoke project names are no longer caller-controlled. Each run creates a unique namespace and + ownership token. Containers, networks, and volumes carry the ownership label; preflight refuses + any collision and cleanup verifies every discovered resource before `down --volumes`. +- Added a dynamic fake-Docker contract suite for caller override, collision, and mismatched cleanup + labels, plus a real-Docker collision probe using a unique labeled volume. + +Follow-up verification: + +- `./scripts/local-vector-smoke.sh`: PASS, including live secret rotation and persisted retrieval +- `./scripts/test-local-vector-smoke-safety.sh`: PASS +- `./scripts/test-local-vector-smoke-live-collision.sh`: PASS +- harness: 477 passed, 5 deselected +- backend: 84 passed; TypeScript typecheck PASS +- frontend: 226 passed; TypeScript typecheck PASS +- Compose security, backend URL, config, shell syntax, and diff checks: PASS + +Remaining operational constraint: the bootstrap admin secret must continue to match the PostgreSQL +bootstrap account stored in the volume. Runtime migrator/reader/writer rotation is supported without +data deletion; bootstrap-account password rotation is a distinct database-administration operation. + +## Final hardening — bootstrap account rotation + +The remaining operational constraint is now covered by +`scripts/vector-rotate-bootstrap-password.sh OLD_SECRET_FILE NEW_SECRET_FILE`: + +- It does not rely on `POSTGRES_PASSWORD_FILE` after initialization. +- It pre-stages the deployment-file replacement in the same directory, authenticates to the live + database with the explicit old file, and changes only the authenticated bootstrap role. +- Passwords are passed as connection parameters and rendered with psycopg2 SQL composition, so + shell and SQL metacharacters are not interpolated. +- A second connection must authenticate with the new password before the command succeeds. If that + verification fails, the still-open old connection restores the old database password. +- Only after verified database login does an atomic rename replace the current deployment secret. + Wrong-old authentication and verification failures leave deployment configuration unchanged. + +Final live smoke evidence on one existing `vector_data` volume: + +- wrong-old bootstrap rotation rejected; current deployment secret unchanged +- bootstrap password with quote characters rotated successfully +- old bootstrap login rejected and new login accepted +- `vector-reconcile`, packaged migrations, and core health passed afterward +- the vector record written before rotation remained searchable after rotation and after a further + database/core restart + +Final tests: + +- `./scripts/test-vector-bootstrap-rotation.sh`: PASS +- `./scripts/local-vector-smoke.sh`: PASS with negative and positive live bootstrap rotation +- existing local-vector collision/safety and Compose deployment contracts: PASS + +## Final identity and secret-policy alignment + +- `THT_VECTOR_BOOTSTRAP_USER` is now passed through core as well as vector-db and reconciliation, + so the rotation helper uses the authoritative configured role instead of defaulting to `postgres`. +- Rotation and reconciliation source the same raw-file `secret-policy.sh`: non-empty and no + whitespace, including trailing newlines. Rotation validates both files before Docker, + PostgreSQL, or atomic replacement staging; `test-vector-secret-policy.sh` pins empty, newline, + internal-space, and valid metacharacter cases. +- Fake-Docker tests prove a non-default identity reaches the helper path and whitespace rejection + performs no Docker call and creates no staged replacement. +- The real smoke runs the entire stack as `thoth_bootstrap_smoke`. Its whitespace-negative case + leaves the deployment file unchanged and proves the existing database login still succeeds; + non-default-account bootstrap rotation, reconciliation, migration, core health, restart, and + persisted retrieval all pass. diff --git a/.superpowers/sdd/pgvector-task-4-report.md b/.superpowers/sdd/pgvector-task-4-report.md new file mode 100644 index 00000000..38fe1607 --- /dev/null +++ b/.superpowers/sdd/pgvector-task-4-report.md @@ -0,0 +1,94 @@ +# Local pgvector Task 4 report + +## Outcome + +Implemented adapter parity gates and an operator-safe custom-format backup/restore workflow. + +- Direct and HTTP stores now share validation, configured-dimension rejection, and deterministic + similarity ordering with record ID as the tie-break. +- The parity fixture exercises identical records through real pgvector and the HTTP RPC contract: + kind filtering, ordering, hashes, replacement upserts, invalid collection/kind errors, and query + plus write dimensions. +- Backup explicitly allowlists the three vector tables and migration ledger, refuses overwrite, + writes through a partial file, and uses a custom compressed archive. +- Restore requires explicit active-source and target coordinates. It compares PostgreSQL system + identifier plus database OID (robust across DNS aliases), refuses the active database, checks for + an empty target unless force is explicit, and restores with exit-on-error. +- Passwords are accepted only through validated secret files, converted to private temporary + `PGPASSFILE`s, and never placed in command arguments or success/error logs. +- Role passwords/login identities are deliberately not dumped. The target must have the approved + passwordless group roles and pgvector extension reconciled before restore; archived ACLs restore + the reader/writer grants. + +## TDD and semantic alignment + +The first parity run exposed the intended HTTP differences: it accepted unknown collections and +wrong dimensions. Direct pgvector also had no stable order for equal cosine distance. The adapters +were aligned, and the final focused real-pgvector gate passed: **25 passed**. + +The first recovery run caught an incorrect probe username before restore. The second caught an +intersection between `pg_dump --schema` and the explicit public ledger table. The third confirmed +the archive contents but caught missing target group roles. Each defect was corrected and the +complete drill was rerun from a fresh generated project. + +## Live recovery smoke + +`./scripts/local-vector-smoke.sh --backup-restore`: **PASS**. + +- generated/owned source Compose project and source `vector_data` +- distinct restore container and distinct named restore volume +- migration and role health, secret rotation, restart persistence +- real custom backup, then deliberate mutation of the active source record +- same-database identity guard evaluated before restore +- restore into the separate target only +- restored hash equals the pre-mutation backup, proving retrieval parity +- migration ledger has all three applied versions +- all three restored embedding columns report `vectors.vector(768)` +- ownership-checked cleanup; the active operator project/volume is never addressed + +## Verification + +- parity + direct adapter: 25 passed +- full harness: 485 passed, 5 deselected +- changed Python files: Ruff clean +- shell syntax: clean +- `git diff --check`: clean +- full Ruff: unchanged repository baseline of 34 unrelated pre-existing test-file violations + +## Self-review and operational constraints + +The restore account must be able to read `pg_control_system()` for the robust cluster-identity +comparison and create/restore the selected objects. This is intentionally an administrative +recovery operation, not a runtime reader/writer action. `--force-nonempty` is explicit but still +uses `pg_restore --clean --if-exists`; operators should prefer a new database/volume and validate +migration status, health, and known retrieval before endpoint cutover. + +## Post-review hardening + +All five final review findings were addressed in a follow-up commit: + +- Restore now requires a physically separate PostgreSQL cluster and refuses any equal + `system_identifier`, independent of database OID or hostname. +- `pg_restore` combines `--single-transaction` with `--exit-on-error`. The live drill creates an + existing vector sentinel, deliberately fails late during a forced restore, and proves the + original sentinel row/hash remains unchanged before performing the successful restore. +- Backup uses a mode-0600 `mktemp` in the output directory, atomically renames it, and cleans only + that owned path. A fake-command test pins symlink-clobber resistance and preserves an adversarial + legacy `.partial` symlink and its target. +- HTTP parity now traverses the real `VectorRestClient` transport boundary. It asserts RPC URL/key + and kinds payloads, legacy 404 fallback, response conversion, malformed metadata tolerance, and + canonical `VectorRestError` to `VectorStoreError` mapping. +- The restored target runs role/secret reconciliation and a real `PgVectorStore` with separate + reader/writer logins. Health, known-record search, writer upsert, hash probe, schema/table/column/ + sequence authority, and 768-dimensional compatibility are therefore verified through the + production adapter. Reconciliation now restores group-role schema `USAGE`, which table-selected + archives cannot carry. + +### Atomic no-replace backup publication + +The final publication review is also closed. The private same-directory archive is published with +an atomic hard-link create rather than rename-overwrite semantics. If any process creates the final +file or symlink after preflight but before publication, `ln` fails with `EEXIST`, the backup exits +nonzero, the concurrent destination remains byte-for-byte intact, and the trap removes only the +randomly named temporary archive owned by this invocation. The fake `pg_dump` safety test creates +that destination immediately before returning and pins the failure and cleanup behavior. diff --git a/.superpowers/sdd/progress.md b/.superpowers/sdd/progress.md new file mode 100644 index 00000000..c6b9d14a --- /dev/null +++ b/.superpowers/sdd/progress.md @@ -0,0 +1,14 @@ +# Portable deployment SDD progress + +Plan: `docs/superpowers/plans/2026-07-11-adapter-foundations.md` +Branch: `codex/portable-deployment` +Worktree: `/Users/mp/projects/ThothII/.worktrees/portable-deployment` + +Task 1: complete (commits e02e61e..a4eb6cc, review clean) +Task 1 final-review follow-up: public exports and frozen capability records now have explicit regressions. +Task 2: complete (commits a4eb6cc..f6302b3, review clean after authorized contract correction) +Task 3: complete (commits f6302b3..fe8d70d, review clean after authorized write-envelope correction) +Task 4: complete (commits fe8d70d..1e0911b, review clean) +Task 4 final-review follow-up: a real `tht` subprocess now proves exactly one legacy warning on stderr and pristine JSON stdout. +Task 5: complete (commits 1e0911b..dbbab6d, review clean after two fix waves) +Final adapter review fix wave: complete (`fix(adapter): close final foundation review`). HTTP vector reader/writer endpoints are independently optional; writer-only targeted memory/solved writes are supported. Vector health reports each side separately plus configured/observed embedding dimensions. `build_vector_loader` remains an explicitly tracked bulk-sync-only exception scheduled for the local pgvector migration plan; it is not used by interactive/targeted writes. diff --git a/.superpowers/sdd/task-2-report.md b/.superpowers/sdd/task-2-report.md new file mode 100644 index 00000000..f9061d8c --- /dev/null +++ b/.superpowers/sdd/task-2-report.md @@ -0,0 +1,30 @@ +# Task 2 report — root Compose startup + +Status: DONE + +Implemented the root Compose defaults and the single bundle declaration: + +- added `.env.example` with automatic Compose defaults (`COMPOSE_FILE=compose.yaml`, an empty + profile, and the relative `THT_SECRETS_FILE` path); +- removed the mandatory `external` profile from `core` and `frontend`; +- mounted `deploy/secrets/thothii.secrets` at `/run/secrets/thothii.secrets` and passed only the + mounted path into the core container; +- changed the production overlay to inherit that bundle instead of declaring per-secret mounts; +- removed the local overlay's legacy `env_file` dependency; +- added the versioned bundle template and `.gitignore` exception; +- updated deployment security checks and added `scripts/test-default-compose.sh`. + +Focused verification: + +```text +./scripts/test-default-compose.sh # default Compose contract passed. +./scripts/test-container-deployment.sh # container deployment security contract passed. +./scripts/test-preprocess-compose-config.sh # preprocess compose config: ok +docker compose --env-file .env.example config --quiet + (with a temporary mode-0600 bundle via THT_SECRETS_FILE) +git diff --check +``` + +The local-vector and preprocess service secret declarations remain for Task 3, which converts +those services to the same bundle helper. Documentation and smoke command migration is reserved +for Task 4. diff --git a/.superpowers/sdd/task-3-report.md b/.superpowers/sdd/task-3-report.md new file mode 100644 index 00000000..b6696028 --- /dev/null +++ b/.superpowers/sdd/task-3-report.md @@ -0,0 +1,59 @@ +# Task 3 report — one secret bundle for local services + +## Status + +Complete. Local pgvector bootstrap, reconciliation, migration, and preprocess services now +mount only `/run/secrets/thothii.secrets`. `deploy/vector/secret-policy.sh` validates the +whole bundle (allowlist, duplicate/empty/unknown keys, comments/blank lines, mode and symlink +policy) and returns only the requested value. The core entrypoint exposes DWH/vector/CA values +to the harness and materializes short-lived 0600 password files for workspace resolution. + +## TDD evidence + +- RED: `./scripts/test-preprocess-compose-config.sh` failed on the pre-existing + `vector_reader_password` Compose secret declaration. +- GREEN: the same command passes after the bundle conversion and verifies local-vector + workspace interpolation and shared secret mounts. +- `./scripts/test-vector-secret-policy.sh` covers comments/blank lines and rejects an + unrelated duplicate key. + +## Verification + +- `./scripts/test-vector-secret-policy.sh` — passed. +- `./scripts/test-preprocess-compose-config.sh` — passed. +- `./scripts/test-vector-backup-restore-safety.sh` — passed. +- `./scripts/test-default-compose.sh` — passed. +- `./scripts/test-container-deployment.sh` — passed. +- `./scripts/local-vector-smoke.sh` — passed with real Docker (bootstrap rotation, role + reconciliation, migration, persistence and restart). +- `./scripts/preprocess-smoke.sh` — passed with real Docker (unchanged rerun, mutation, DWH + job, ACTIVE publication and cleanup). +- `./scripts/preprocess-smoke.sh --cleanup-failure` — passed. +- `git diff --check` and `sh -n` gates — passed. + +## Critical review fix + +`buildPiChildEnv` now removes `THT_DWH_API_KEY`, `THT_VEC_API_KEY`, `THT_VEC_WRITE_API_KEY`, +`THT_SSL_CA`, `THT_CA`, and their file metadata before spawning Pi. A regression test proves +that neither secret values nor bundle/file metadata are inherited by the Pi child. + +## Commits + +- `70a19f2 feat(compose): use one secret bundle for local services` +- `d500563 fix(security): scrub deployment secrets from Pi child` +- `8518a73 fix(security): scrub raw deployment secret values` + +## Concern + +The rotation helper retains its old/new scratch-file CLI contract; smoke tests keep those files +outside Compose and mount only the bundle. + +## Whole-branch review fixes + +- `core-entrypoint.sh` validates `THT_SECRETS_FILE` fail-closed before optional lookups; malformed, + duplicate, unknown, oversized, or overlong bundles stop startup with sanitized diagnostics. +- Runtime password files are cleaned after child exit via signal forwarding and `wait`, rather + than being orphaned by `exec`. +- The shell loader accepts CRLF bundles (Windows/Notepad) consistently with the TypeScript loader. +- Optional key lookup distinguishes an absent key from an invalid value; present malformed + credentials now stop entrypoint startup instead of being silently ignored. diff --git a/.superpowers/sdd/task-4-report.md b/.superpowers/sdd/task-4-report.md new file mode 100644 index 00000000..0ac3a0ad --- /dev/null +++ b/.superpowers/sdd/task-4-report.md @@ -0,0 +1,55 @@ +# Task 4 report — one-command Docker documentation + +## Status + +Implemented. The installation documentation now uses the canonical flow: + +```sh +cp .env.example .env +cp deploy/secrets/thothii.secrets.example deploy/secrets/thothii.secrets +chmod 600 deploy/secrets/thothii.secrets +docker compose up --build -d +``` + +Updated: + +- `README.md` with root `.env` defaults, one bundle, optional overlay presets, CA limitation, + preprocessing, and migration notes. +- `docs/installazione-docker-4-contesti.md` rewritten with exact files to create/edit and the + four requested contexts (co-located DB/vector, Mac, Windows, and remote DB/Evidence server). +- `docs/index.md` link text for the one-command installation. +- `deploy/secrets/README.md` bundle syntax, permissions, runtime mount verification, CA handling, + and migration guidance. +- `scripts/docker-smoke.sh` now creates a disposable mode-0600 bundle and exercises the default + Compose services without the legacy `external` profile. +- `scripts/test-default-compose.sh` asserts the exact installation command, tracked templates, + and absence of the legacy setup in the guide. +- `scripts/test-container-deployment.sh` now validates the bundle mount and rejects legacy + per-secret references; `.dockerignore` explicitly re-includes only the required vector policy + helper so the Docker build context remains safe. +- The Mac/Windows/local-vector and remote-server snippets now include required DWH/database and + Evidence-root settings. `deploy/env.example` is explicitly deprecated and no longer selects a + different Compose overlay. + +The docs explicitly state that a PEM CA chain cannot be put in the strict single-line bundle. A +reviewed Compose override/secret-manager mount is required for `THT_SSL_CA`. Direct PostgreSQL +workspace examples are marked as advanced and require a separate reviewed runtime password mount; +the base bundle mount is the only default mount. + +## Verification + +- `sh -n scripts/docker-smoke.sh scripts/test-default-compose.sh` — passed. +- `./scripts/test-default-compose.sh` — passed. +- `./scripts/test-container-deployment.sh` — passed after migrating its local-vector assertions + to the single bundle and checking the `.dockerignore` deployment allowlist. +- `git diff --check` — passed. +- `./scripts/test-docker-smoke.sh` — passed after updating its static assertion to the default + no-profile invocation. +- `docker buildx build --file docker/core.Dockerfile --check .` — passed; BuildKit reported no + warnings after the `.dockerignore` parent-directory fix. + +## Concerns + +The legacy `scripts/vector-rotate-bootstrap-password.sh` maintenance helper still accepts +old/new standalone files. Its output is intentionally documented as a transitional interface; +the resulting value must be copied into the bundle before restarting local-vector services. diff --git a/README.md b/README.md new file mode 100644 index 00000000..8059d471 --- /dev/null +++ b/README.md @@ -0,0 +1,206 @@ +# ThothII + +ThothII is a human-reviewed NL-to-SQL workflow with a React frontend and a Fastify/Pi/`tht` +core. The portable deployment runs exactly two application services; data services remain +external in this profile. + +## Docker Compose: one-command startup + +Requirements: Docker Engine with Compose v2. The default project starts only the two +application images; DWH, vector and embedding services can be remote or supplied by an +optional overlay. + +From a fresh clone, run these commands from the repository root: + +```sh +cp .env.example .env +cp deploy/secrets/thothii.secrets.example deploy/secrets/thothii.secrets +chmod 600 deploy/secrets/thothii.secrets +# Edit .env (non-secret endpoints) and deploy/secrets/thothii.secrets (KEY=VALUE lines). +docker compose up --build -d +``` + +The root `.env` is loaded automatically by Compose. It defaults to `compose.yaml`, an empty +profile, and `THT_SECRETS_FILE=deploy/secrets/thothii.secrets`; no `--env-file`, `-f`, or +`--profile` flag is required for the normal installation. Add or edit YAML workspace descriptors +under `deploy/workspaces/`; they are mounted read-only and relative `roots` resolve beneath +`/data/workspaces/`. Open (set `THOTH_HTTP_PORT` in +`.env` to choose another loopback port). + +The bundle contains only values, one per line (`THT_MODEL_API_KEY=...`, DWH/vector keys, and +the optional local-vector passwords). It is ignored by Git and never copied into either image. +Do not put credentials in `.env`, workspace YAML, URLs, or Compose interpolation values. + +### Optional overlays + +Overlays are selected in `.env`, so the operational command remains the same. On Unix-like +systems use `:` between files; on Windows use `;`: + +```dotenv +# Remote DWH/vector/embedding services with authenticated reverse proxy: +COMPOSE_FILE=compose.yaml:deploy/compose.production.yaml +COMPOSE_PROFILES= + +# Local pgvector (Mac/Windows or a standalone application server): +COMPOSE_FILE=compose.yaml:deploy/compose.local-vector.yaml +COMPOSE_PROFILES=local-vector +``` + +After changing `.env`, apply the selected configuration with `docker compose up --build -d`. +Preprocessing is an explicit opt-in preset: append +`deploy/compose.preprocess.yaml:deploy/compose.preprocess-local-vector.yaml` and set +`COMPOSE_PROFILES=local-vector,preprocess`; then run the job with +`docker compose run --rm preprocess-evidence` or `preprocess-dwh`. + +Application state, including settings, sessions, artifacts, and indexes, lives in the named +`thoth_data` volume mounted at `/data`. `docker compose down` keeps that volume. Only an +explicit destructive command such as `docker compose down --volumes` removes it. + +The frontend depends on the core health check and proxies `/health` and `/api/*` to it. The +application health endpoint intentionally checks process readiness only; external dependency +diagnostics are exposed by `tht doctor` and do not prevent the UI from starting. + +Run the end-to-end packaging check with: + +```sh +./scripts/docker-smoke.sh +``` + +The smoke script validates Compose, builds and waits for both services, checks health through +the frontend, verifies SSE response headers, restarts the core, and confirms `/data` survives. +Each run uses a unique Compose project and removes that project's containers, network, and test +volume afterward. It never targets the fixed `thothii` operator project or its volume. Set +`SMOKE_PROJECT` to a different explicit project name for reproducible debugging, and set +`KEEP_SMOKE_RESOURCES=1` to retain that smoke project's resources for inspection; remove them +later with `docker compose --project-name "$SMOKE_PROJECT" down --volumes`. + +## Optional local pgvector and recovery + +The local-vector overlay reads `THT_VECTOR_BOOTSTRAP_PASSWORD`, +`THT_VECTOR_MIGRATOR_PASSWORD`, `THT_VECTOR_READER_PASSWORD`, and +`THT_VECTOR_WRITER_PASSWORD` from the same bundle. Its `vector_data` volume is independent of +application state; passwords are selected at runtime and are never passed as URL arguments. + +## Preprocessing jobs and S3 Evidence + +The included job workspaces target the local-vector profile. Put the four local-vector password +keys in the bundle, set `THT_OLLAMA_URL`, mount Evidence at `/data/source/evidence`, then select +the preprocessing preset in `.env`: + +```dotenv +COMPOSE_FILE=compose.yaml:deploy/compose.local-vector.yaml:deploy/compose.preprocess.yaml:deploy/compose.preprocess-local-vector.yaml +COMPOSE_PROFILES=local-vector,preprocess +``` + +Run `docker compose run --rm preprocess-evidence` or +`docker compose run --rm preprocess-dwh`. The overlay makes each job wait for the vector +database health check, role reconciliation, and a successful migration; no separate database +startup or migration command is required. + +S3 Evidence uses the optional `tht[s3]` dependency and canonical `s3://bucket/key` provenance. +AWS endpoints are used when no custom URL is supplied. Every custom endpoint is an explicit egress +trust-boundary opt-in and uses path-style addressing; private and HTTP endpoints require additional +independent opt-ins. Literal non-global IPv4/IPv6 addresses are classified locally; hostnames are +not DNS-pinned, so trusted custom-endpoint deployments must enforce their destination with network +egress policy. Store access key, secret key, and session token as secret references in +deployment configuration—never in Compose environment values or source URIs. Discovery and reads +are bounded by configured page, object, and byte limits. + +Create a versioned PostgreSQL custom-format backup (the filename is operator-controlled, so use +an immutable timestamp or release identifier): + +```sh +./scripts/vector-backup.sh \ + --host 127.0.0.1 --port 5432 --database thoth --user thoth_backup \ + --password-file /secure/thoth/vector-backup-password \ + --output /secure/backups/thoth-vectors-2026-07-12.dump +``` + +The dump contains the three allowlisted `vectors` tables, their data and ACLs, plus the +`public.tht_vector_migrations` ledger. Login roles and passwords are deliberately not copied: +provision/reconcile the approved role names on the target first, and install the `vector` +extension in its `vectors` schema. The target must otherwise contain no vector tables or ledger. + +Restore always names both the currently active source and a target on a physically distinct +PostgreSQL cluster. The script compares PostgreSQL system identity, so host aliases or a different +database in the active cluster cannot bypass the guard. It refuses a non-empty target unless +`--force-nonempty` is explicit, and the clean restore is one transaction: + +```sh +./scripts/vector-restore.sh \ + --active-host vector-db --active-database thoth --active-user thoth_backup \ + --active-password-file /secure/thoth/vector-active-password \ + --target-host vector-db-restore --target-database thoth --target-user thoth_restore \ + --target-password-file /secure/thoth/vector-restore-password \ + --input /secure/backups/thoth-vectors-2026-07-12.dump +``` + +After restore, run `tht vector migrate --status --json`, adapter health, and a known retrieval +query against the target before changing any deployment endpoint. Never test recovery against the +active `vector_data` volume. `./scripts/local-vector-smoke.sh --backup-restore` performs this drill +with disposable source and target volumes. + +## Production trust boundary and secrets + +ThothII does not implement OIDC. Do not expose its application port directly to a network. +The production pattern is an authenticated host reverse proxy that: + +- terminates TLS and authenticates every request; +- removes any client-supplied identity header; +- injects one trusted `X-Authenticated-User` value; +- proxies to the loopback-only ThothII frontend. + +[`deploy/nginx-authenticated-proxy.conf.example`](deploy/nginx-authenticated-proxy.conf.example) +shows the contract using nginx `auth_request`; replace the placeholder authentication gateway +with the organization's reviewed identity proxy. `AUTH_MODE=upstream` trusts this boundary and +rejects requests without the identity header. Setting `THOTH_PUBLIC_EXPOSURE=true` with any other +auth mode fails during core startup. + +Production credentials use the one Compose secret bundle, not `.env`. Put the required keys in +`deploy/secrets/thothii.secrets` and select the production overlay in `.env`: + +```dotenv +THT_MODEL_API_KEY=replace-me +THT_DWH_API_KEY=replace-me +THT_VEC_API_KEY=replace-me +THT_VEC_WRITE_API_KEY=replace-me +``` + +The bundle is mounted read-only as `/run/secrets/thothii.secrets` and must be mode `0600` or +`0400` on the host. Docker's runtime `0444` mode is accepted only beneath `/run/secrets`; see +[`deploy/secrets/README.md`](deploy/secrets/README.md). A PEM CA chain is deliberately not a +bundle value: PEM contains whitespace and is rejected by the strict parser. Keep the CA chain in +the host/secret-manager materialization and add a reviewed Compose override that mounts it at +`/run/secrets/ca-chain.pem` and sets `THT_SSL_CA` when a private CA is required. The base bundle +does not create that mount. The frontend remains on loopback; the authenticated host proxy is the +only public listener. + +Set the selected model provider in application settings (or `PI_PROVIDER`). For each Pi spawn the +backend validates and reads `THT_MODEL_API_KEY` from the bundle, then exposes its value only as the provider's +recognized child variable (for example `ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, or +`ZAI_API_KEY`). Neither the generic file path nor deprecated `PI_PROVIDER_API_KEY` is inherited by +Pi. Local providers such as Ollama require no model key. + +`THT_MODEL_API_KEY` supports Pi providers whose authentication is exactly one key: +`ant-ling`, `anthropic`, `cerebras`, `deepseek`, `fireworks`, `github-copilot`, `google` +(including the `gemini` alias), `google-vertex` when using its API-key mode, `groq`, +`huggingface`, `kimi-coding`, `minimax`, `minimax-cn`, `mistral`, `moonshotai`, +`moonshotai-cn`, `nvidia`, `openai`, `opencode`, `opencode-go`, `openrouter`, `together`, +`vercel-ai-gateway`, `xai`, the four `xiaomi*` providers, `zai`, and `zai-coding-cn`. +Compound providers are deliberately unsupported: `amazon-bedrock`, `azure-openai-responses`, +`cloudflare-workers-ai`, and `cloudflare-ai-gateway` require multiple credential/configuration +values. Selecting one fails before Pi starts; ambient AWS, Azure, and Cloudflare credentials are +still scrubbed. Supporting them requires a future dedicated provider-specific configuration. + +## Reproducible image verification + +Base images use exact tags and immutable multi-platform manifest digests. Dependency update and +residual OS-repository limitations are documented in [`docker/LOCKS.md`](docker/LOCKS.md). +Run the shared architecture gate with `PLATFORM=linux/amd64` or `PLATFORM=linux/arm64`: + +```sh +PLATFORM=linux/arm64 ./scripts/verify-container-images.sh +``` + +It builds both images, runs common version/runtime/security smokes, and emits an image/package +inventory beneath `.artifacts/container-images/`. CI runs the same script for both architectures. diff --git a/backend/src/app.ts b/backend/src/app.ts index 33fe00fe..23a26e39 100644 --- a/backend/src/app.ts +++ b/backend/src/app.ts @@ -34,6 +34,7 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc thtBin: config.thtBin, harnessDir: config.harnessDir, configPath: process.env.THT_CONFIG ?? "config/tht.yaml", + dataRoot: config.dataRoot, }); const mgr = deps?.mgr ?? new PiProcessManager(config, deps?.spawnFn ? { spawnFn: deps.spawnFn } : undefined); const hub = new SseHub(); @@ -41,7 +42,12 @@ export function buildApp(config: AppConfig, deps?: BuildAppDeps): FastifyInstanc const listModels = deps?.listModels ?? createPiModelLister(config); const getSettings = deps?.getSettings ?? (() => effectiveSettings(config, loadSettings(config))); - app.addHook("preHandler", authPreHandler(config.authMode)); + const authenticate = authPreHandler(config.authMode); + app.addHook("preHandler", async (req, reply) => { + // Process readiness is intentionally unauthenticated for local container/proxy probes. + if (req.url === "/health") return; + return authenticate(req, reply); + }); app.get("/health", async () => ({ status: "ok" })); sessionRoutes(app, { mgr, tht: tht as ThtRunner, hub, getSettings, diff --git a/backend/src/auth/auth.ts b/backend/src/auth/auth.ts index 00615667..39adc88e 100644 --- a/backend/src/auth/auth.ts +++ b/backend/src/auth/auth.ts @@ -1,6 +1,6 @@ import type { FastifyRequest, FastifyReply } from "fastify"; -export function authPreHandler(mode: "none" | "mock" | "oidc") { +export function authPreHandler(mode: "none" | "mock" | "upstream") { return async (req: FastifyRequest, reply: FastifyReply) => { if (mode === "none") { (req as any).user = { id: "dev@local" }; @@ -9,8 +9,11 @@ export function authPreHandler(mode: "none" | "mock" | "oidc") { id: (req.headers["x-mock-user"] as string) ?? "mock", }; } else { - reply.code(501); - throw new Error("OIDC non configurato (MVP: usa none/mock)"); + const id = req.headers["x-authenticated-user"]; + if (typeof id !== "string" || id.trim() === "") { + return reply.code(401).send({ error: "authenticated upstream identity required" }); + } + (req as any).user = { id }; } }; } diff --git a/backend/src/config.ts b/backend/src/config.ts index 1bae9bf8..fae338d7 100644 --- a/backend/src/config.ts +++ b/backend/src/config.ts @@ -1,22 +1,60 @@ +import path from "node:path"; + export interface AppConfig { host: string; port: number; harnessDir: string; thtBin: string; piBin: string; - authMode: "none" | "mock" | "oidc"; + authMode: "none" | "mock" | "upstream"; defaults: { provider?: string; model?: string; thinking?: string }; maxPiProcesses: number; settingsFile: string; + dataRoot?: string; ollamaEnsureTimeoutMs: number; + secretsFile?: string; + secretFiles: Readonly>; + modelApiKeyFile?: string; } export function loadConfig(env: Record): AppConfig { + const authMode = env.AUTH_MODE ?? "none"; + if (!(["none", "mock", "upstream"] as const).includes(authMode as AppConfig["authMode"])) { + throw new Error(`unsupported AUTH_MODE=${authMode}; use none, mock, or upstream`); + } + if (env.THOTH_PUBLIC_EXPOSURE === "true" && authMode !== "upstream") { + throw new Error("public exposure requires AUTH_MODE=upstream behind a trusted proxy"); + } + const modelApiKeyFile = env.THT_MODEL_API_KEY_FILE; + if (modelApiKeyFile !== undefined && ( + modelApiKeyFile.trim() !== modelApiKeyFile + || modelApiKeyFile.length === 0 + || modelApiKeyFile.includes("\0") + || !path.isAbsolute(modelApiKeyFile) + )) { + throw new Error("model credential configuration is invalid"); + } + const secretsFile = env.THT_SECRETS_FILE; + if (secretsFile !== undefined && ( + secretsFile.trim() !== secretsFile || secretsFile.length === 0 || secretsFile.includes("\0") + || !path.isAbsolute(secretsFile) + )) throw new Error("secret bundle configuration is invalid"); + const secretFiles: Record = {}; + for (const name of [ + "THT_MODEL_API_KEY_SECRET_FILE", "THT_DWH_API_KEY_SECRET_FILE", "THT_VEC_API_KEY_SECRET_FILE", + "THT_VEC_WRITE_API_KEY_SECRET_FILE", "THT_CA_SECRET_FILE", "THT_VECTOR_BOOTSTRAP_PASSWORD_SECRET_FILE", + "THT_VECTOR_MIGRATOR_PASSWORD_SECRET_FILE", "THT_VECTOR_READER_PASSWORD_SECRET_FILE", + "THT_VECTOR_WRITER_PASSWORD_SECRET_FILE", + ]) secretFiles[name] = env[name]; return { host: env.HOST ?? "127.0.0.1", port: Number(env.PORT ?? 8787), harnessDir: env.THT_HARNESS_DIR ?? "../harness", thtBin: env.THT_BIN ?? "tht", piBin: env.PI_BIN ?? "pi", - authMode: (env.AUTH_MODE as AppConfig["authMode"]) ?? "none", + authMode: authMode as AppConfig["authMode"], defaults: { provider: env.PI_PROVIDER, model: env.PI_MODEL, thinking: env.PI_THINKING }, maxPiProcesses: Number(env.MAX_PI_PROCESSES ?? 4), settingsFile: env.SETTINGS_FILE ?? "data/settings.json", + dataRoot: env.THT_DATA_ROOT, ollamaEnsureTimeoutMs: Number(env.OLLAMA_ENSURE_TIMEOUT_MS ?? 60000), + secretsFile, + secretFiles, + modelApiKeyFile, }; } diff --git a/backend/src/config/secret-bundle.ts b/backend/src/config/secret-bundle.ts new file mode 100644 index 00000000..b46746a3 --- /dev/null +++ b/backend/src/config/secret-bundle.ts @@ -0,0 +1,137 @@ +import { + closeSync, constants, fstatSync, lstatSync, openSync, readFileSync, + type Stats, +} from "node:fs"; + +/** Keys accepted by the deployment bundle. Keep this list intentionally explicit. */ +export const SECRET_BUNDLE_KEYS = Object.freeze([ + "THT_MODEL_API_KEY", "THT_DWH_API_KEY", "THT_VEC_API_KEY", "THT_VEC_WRITE_API_KEY", + "THT_CA", "THT_SSL_CA", "THT_VECTOR_BOOTSTRAP_PASSWORD", "THT_VECTOR_MIGRATOR_PASSWORD", + "THT_VECTOR_READER_PASSWORD", "THT_VECTOR_WRITER_PASSWORD", "PI_PROVIDER_API_KEY", +] as const); + +const ALLOWED = new Set(SECRET_BUNDLE_KEYS); +const LEGACY_FILES: Readonly> = { + THT_MODEL_API_KEY: "THT_MODEL_API_KEY_SECRET_FILE", + THT_DWH_API_KEY: "THT_DWH_API_KEY_SECRET_FILE", + THT_VEC_API_KEY: "THT_VEC_API_KEY_SECRET_FILE", + THT_VEC_WRITE_API_KEY: "THT_VEC_WRITE_API_KEY_SECRET_FILE", + THT_CA: "THT_CA_SECRET_FILE", + THT_SSL_CA: "THT_CA_SECRET_FILE", + THT_VECTOR_BOOTSTRAP_PASSWORD: "THT_VECTOR_BOOTSTRAP_PASSWORD_SECRET_FILE", + THT_VECTOR_MIGRATOR_PASSWORD: "THT_VECTOR_MIGRATOR_PASSWORD_SECRET_FILE", + THT_VECTOR_READER_PASSWORD: "THT_VECTOR_READER_PASSWORD_SECRET_FILE", + THT_VECTOR_WRITER_PASSWORD: "THT_VECTOR_WRITER_PASSWORD_SECRET_FILE", +}; + +const MAX_BUNDLE_BYTES = 64 * 1024; +const MAX_LINE_BYTES = 16 * 1024; + +export interface SecretBundleConfig { + secretsFile?: string; + secretFiles?: Readonly>; + /** Accepted for callers that pass the raw process environment. */ + THT_SECRETS_FILE?: string; +} + +/** Injectable filesystem boundary used by the race-condition tests. */ +export interface SecretBundleFsOps { + lstat(path: string): Stats; + open(path: string, flags: number): number; + fstat(fd: number): Stats; + read(fd: number): string; + close(fd: number): void; +} + +const realFs: SecretBundleFsOps = { + lstat: lstatSync, + open: openSync, + fstat: fstatSync, + read: (fd) => readFileSync(fd, "utf8"), + close: closeSync, +}; + +function unavailable(): Error { return new Error("secret bundle is unavailable"); } + +function secureStat(info: Stats, docker: boolean): boolean { + const mode = info.mode & 0o777; + if (!info.isFile() || info.isSymbolicLink() || info.nlink !== 1 || info.size > MAX_BUNDLE_BYTES) return false; + if (docker) { + return (info.uid === 0 && mode === 0o444) + || (info.uid === (process.getuid?.() ?? info.uid) && (mode === 0o400 || mode === 0o600)); + } + return info.uid === (process.getuid?.() ?? info.uid) && (mode === 0o400 || mode === 0o600); +} + +function readSecure(file: string, fs: SecretBundleFsOps): string { + let fd: number | undefined; + try { + if (!file || file.trim() !== file || file.includes("\0")) throw unavailable(); + const docker = file.startsWith("/run/secrets/") && !file.slice("/run/secrets/".length).includes("/"); + if (file.startsWith("/run/secrets/") && !docker) throw unavailable(); + if (docker) { + const parent = fs.lstat("/run/secrets"); + if (!parent.isDirectory() || parent.uid !== 0 || (parent.mode & 0o022) !== 0) throw unavailable(); + } + const before = fs.lstat(file); + if (!secureStat(before, docker)) throw unavailable(); + fd = fs.open(file, constants.O_RDONLY | constants.O_NOFOLLOW); + const opened = fs.fstat(fd); + if (!secureStat(opened, docker) || before.dev !== opened.dev || before.ino !== opened.ino) throw unavailable(); + return fs.read(fd); + } catch { + throw unavailable(); + } finally { + if (fd !== undefined) try { fs.close(fd); } catch { /* sanitized by design */ } + } +} + +function parseBundle(text: string): ReadonlyMap { + const values = new Map(); + const lines = text.split("\n"); + for (const raw of lines) { + if (raw.length > MAX_LINE_BYTES) throw unavailable(); + const line = raw.endsWith("\r") ? raw.slice(0, -1) : raw; + const trimmed = line.trim(); + if (!trimmed || trimmed.startsWith("#")) continue; + const match = /^([A-Z][A-Z0-9_]*)=(.*)$/.exec(line); + if (!match) throw unavailable(); + const [, key, value] = match; + if (!ALLOWED.has(key) || values.has(key) || value.length === 0 || /[\r\n]/.test(value)) { + throw unavailable(); + } + values.set(key, value); + } + return values; +} + +export function loadSecretBundle(file: string): ReadonlyMap { + return loadSecretBundleWithFs(file, realFs); +} + +/** Same loader with an injectable filesystem boundary; useful for TOCTOU tests. */ +export function loadSecretBundleWithFs(file: string, fs: SecretBundleFsOps): ReadonlyMap { + try { return parseBundle(readSecure(file, fs)); } catch { throw unavailable(); } +} + +/** Resolve a value from the bundle, with the pre-bundle *_SECRET_FILE fallback. */ +export function secretValue(config: SecretBundleConfig, key: string): string | undefined { + const bundlePath = config.secretsFile ?? config.THT_SECRETS_FILE; + if (bundlePath) { + const found = loadSecretBundle(bundlePath).get(key); + if (found !== undefined) return found; + } + const legacyName = LEGACY_FILES[key]; + const legacyPath = legacyName + ? config.secretFiles?.[legacyName] ?? (() => { + const raw = (config as unknown as Record)[legacyName]; + return typeof raw === "string" ? raw : undefined; + })() + : undefined; + if (!legacyPath) return undefined; + const value = readSecure(legacyPath, realFs); + if (!value || /\s/.test(value)) throw unavailable(); + return value; +} + +export function legacySecretEnvNames(): Readonly> { return LEGACY_FILES; } diff --git a/backend/src/pi/list-models.ts b/backend/src/pi/list-models.ts index a7600d0f..5e30e348 100644 --- a/backend/src/pi/list-models.ts +++ b/backend/src/pi/list-models.ts @@ -1,7 +1,8 @@ import { spawn as nodeSpawn, type ChildProcessWithoutNullStreams } from "node:child_process"; -import { join } from "node:path"; import type { AppConfig } from "../config.js"; import { RpcClient } from "../rpc/rpc-client.js"; +import { buildPiChildEnv } from "./provider-credentials.js"; +import { secretValue } from "../config/secret-bundle.js"; export interface PiModel { provider: string; @@ -11,7 +12,11 @@ export interface PiModel { } interface Opts { - spawnFn?: () => ChildProcessWithoutNullStreams; + spawnFn?: ( + command: string, + args: string[], + options: { cwd: string; env: NodeJS.ProcessEnv }, + ) => ChildProcessWithoutNullStreams; ttlMs?: number; nowMs?: () => number; } @@ -24,22 +29,21 @@ interface Opts { export function createPiModelLister(cfg: AppConfig, opts: Opts = {}): () => Promise { const ttlMs = opts.ttlMs ?? 60_000; const now = opts.nowMs ?? (() => Date.now()); - const spawnFn = - opts.spawnFn ?? - (() => { - const harnessVenvBin = join(cfg.harnessDir, ".venv", "bin"); - return nodeSpawn(cfg.piBin, ["--mode", "rpc"], { - cwd: cfg.harnessDir, - env: { ...process.env, PATH: `${harnessVenvBin}:${process.env.PATH ?? ""}` }, - }) as ChildProcessWithoutNullStreams; - }); + const spawnFn = opts.spawnFn ?? nodeSpawn; let cache: { at: number; models: PiModel[] } | null = null; return async function listModels(): Promise { if (cache && now() - cache.at < ttlMs) return cache.models; - const child = spawnFn(); + const env = buildPiChildEnv({ + provider: cfg.defaults.provider, + credentialValue: secretValue(cfg, "THT_MODEL_API_KEY"), + credentialFile: cfg.modelApiKeyFile, + }); + delete env.THT_DATA_ROOT; + if (cfg.dataRoot !== undefined) env.THT_DATA_ROOT = cfg.dataRoot; + const child = spawnFn(cfg.piBin, ["--mode", "rpc"], { cwd: cfg.harnessDir, env }); child.stderr.resume(); const rpc = new RpcClient(child); try { diff --git a/backend/src/pi/pi-process-manager.ts b/backend/src/pi/pi-process-manager.ts index e724ee83..70e2d6f3 100644 --- a/backend/src/pi/pi-process-manager.ts +++ b/backend/src/pi/pi-process-manager.ts @@ -1,9 +1,10 @@ import { spawn as nodeSpawn, type ChildProcessWithoutNullStreams } from "node:child_process"; -import { join } from "node:path"; import type { AppConfig } from "../config.js"; import { RpcClient } from "../rpc/rpc-client.js"; import { SessionBridge } from "../bridge/session-bridge.js"; import type { ThtRunner } from "../tht/tht-runner.js"; +import { buildPiChildEnv, canonicalPiProvider } from "./provider-credentials.js"; +import { secretValue } from "../config/secret-bundle.js"; export interface SessionRuntime { rpc: RpcClient; @@ -11,46 +12,65 @@ export interface SessionRuntime { child: ChildProcessWithoutNullStreams; } -/** Injected test double signature: produce a child process, no args needed. */ -type SpawnFn = () => ChildProcessWithoutNullStreams; +/** Injectable child-process boundary; callbacks may ignore arguments in simpler tests. */ +type SpawnFn = ( + command: string, + args: string[], + options: { cwd: string; env: NodeJS.ProcessEnv }, +) => ChildProcessWithoutNullStreams; export class PiProcessManager { private runtimes = new Map(); - private spawnFn: (sessionId: string, author: string) => ChildProcessWithoutNullStreams; + private spawnFn: ( + sessionId: string, author: string, provider: string | undefined, + ) => ChildProcessWithoutNullStreams; constructor(private cfg: AppConfig, opts?: { spawnFn?: SpawnFn }) { if (opts?.spawnFn) { - this.spawnFn = () => opts.spawnFn!(); + this.spawnFn = (sessionId, author, provider) => + this.spawnPi(opts.spawnFn!, sessionId, author, provider); } else { - this.spawnFn = (sessionId: string, author: string) => { - const harnessVenvBin = join(cfg.harnessDir, ".venv", "bin"); - const env: NodeJS.ProcessEnv = { - ...process.env, - THT_SESSION: sessionId, - THT_AUTHOR: author, - PATH: `${harnessVenvBin}:${process.env.PATH ?? ""}`, - }; - // pi 0.73 (the @mariozechner rebrand) removed the `--approve` flag: rpc mode is - // headless and runs tools without an approval gate, so passing it makes pi exit - // with "Unknown option: --approve". Args are intentionally just `--mode rpc`. - const child = nodeSpawn(cfg.piBin, ["--mode", "rpc"], { - cwd: cfg.harnessDir, - env, - }); - // Drain stderr so the child's stderr buffer never blocks the process. - child.stderr.resume(); - return child; - }; + this.spawnFn = (sessionId, author, provider) => + this.spawnPi(nodeSpawn, sessionId, author, provider); } } + private spawnPi( + spawnFn: SpawnFn, sessionId: string, author: string, provider: string | undefined, + ): ChildProcessWithoutNullStreams { + const env = buildPiChildEnv({ + provider, + credentialValue: secretValue(this.cfg, "THT_MODEL_API_KEY"), + credentialFile: this.cfg.modelApiKeyFile, + additions: { THT_SESSION: sessionId, THT_AUTHOR: author }, + }); + // The Thoth gate executes the deterministic `tht` CLI as a Pi tool. Give only + // this managed session process the adapter values already loaded by the core + // entrypoint; the generic provider helper continues to scrub them by default. + for (const name of [ + "THT_DWH_API_KEY", "THT_VEC_API_KEY", "THT_VEC_WRITE_API_KEY", "THT_SSL_CA", + ] as const) { + if (process.env[name] !== undefined) env[name] = process.env[name]; + } + delete env.THT_DATA_ROOT; + if (this.cfg.dataRoot !== undefined) env.THT_DATA_ROOT = this.cfg.dataRoot; + // pi 0.73 removed `--approve`: rpc mode is headless and its argv is intentionally minimal. + const child = spawnFn(this.cfg.piBin, ["--mode", "rpc"], { + cwd: this.cfg.harnessDir, + env, + }); + // Drain stderr so the child's stderr buffer never blocks the process. + child.stderr.resume?.(); + return child; + } + count(): number { return this.runtimes.size; } get(id: string): SessionRuntime | undefined { return this.runtimes.get(id); } async spawnFor( sessionId: string, - o: { provider?: string; model?: string; thinking?: string; author?: string; mode?: "new" | "resume" }, + o: { provider?: string; model?: string; thinking?: string; author?: string; question?: string; mode?: "new" | "resume" }, ): Promise { // Idempotent per session id: tear down any existing runtime for this id // first (before the cap check) so a resume/respawn neither leaks the old @@ -64,7 +84,8 @@ export class PiProcessManager { throw new Error("max Pi processes reached"); } const author = o.author ?? "dev@local"; - const child = this.spawnFn(sessionId, author); + const provider = canonicalPiProvider(o.provider ?? this.cfg.defaults.provider); + const child = this.spawnFn(sessionId, author, provider); const rpc = new RpcClient(child); const bridge = new SessionBridge(rpc); const rt: SessionRuntime = { rpc, bridge, child }; @@ -85,7 +106,6 @@ export class PiProcessManager { } }); - const provider = o.provider ?? this.cfg.defaults.provider; const model = o.model ?? this.cfg.defaults.model; const thinking = o.thinking ?? this.cfg.defaults.thinking; @@ -98,7 +118,7 @@ export class PiProcessManager { const message = o.mode === "resume" ? `/riprendi-sessione ${sessionId}` - : `/nuova-domanda "kickoff"`; + : `/nuova-domanda ${JSON.stringify(o.question ?? "")}`; rpc.send({ type: "prompt", message }); return rt; } diff --git a/backend/src/pi/provider-credentials.ts b/backend/src/pi/provider-credentials.ts new file mode 100644 index 00000000..525f82ae --- /dev/null +++ b/backend/src/pi/provider-credentials.ts @@ -0,0 +1,162 @@ +import { + closeSync, constants, fstatSync, lstatSync, openSync, readFileSync, + type Stats, +} from "node:fs"; + +/** Audited against @earendil-works/pi-ai 0.80.3 auth plus its locked AWS credential chain. */ +export const PI_0803_CREDENTIAL_ENV_NAMES = Object.freeze([ + "AI_GATEWAY_API_KEY", "ANTHROPIC_API_KEY", "ANTHROPIC_OAUTH_TOKEN", "ANT_LING_API_KEY", + "AWS_ACCESS_KEY_ID", "AWS_BEARER_TOKEN_BEDROCK", "AWS_CONFIG_FILE", + "AWS_CONTAINER_AUTHORIZATION_TOKEN", "AWS_CONTAINER_AUTHORIZATION_TOKEN_FILE", + "AWS_CONTAINER_CREDENTIALS_FULL_URI", "AWS_CONTAINER_CREDENTIALS_RELATIVE_URI", "AWS_PROFILE", + "AWS_ROLE_ARN", "AWS_ROLE_SESSION_NAME", "AWS_SECRET_ACCESS_KEY", "AWS_SESSION_TOKEN", + "AWS_SHARED_CREDENTIALS_FILE", "AWS_WEB_IDENTITY_TOKEN_FILE", "AZURE_OPENAI_API_KEY", + "CEREBRAS_API_KEY", "CLOUDFLARE_ACCOUNT_ID", "CLOUDFLARE_API_KEY", + "CLOUDFLARE_GATEWAY_ID", "COPILOT_GITHUB_TOKEN", "DEEPSEEK_API_KEY", "FIREWORKS_API_KEY", + "GCLOUD_PROJECT", "GEMINI_API_KEY", "GOOGLE_APPLICATION_CREDENTIALS", "GOOGLE_CLOUD_API_KEY", + "GOOGLE_CLOUD_LOCATION", "GOOGLE_CLOUD_PROJECT", "GROQ_API_KEY", "HF_TOKEN", + "KIMI_API_KEY", "MINIMAX_API_KEY", "MINIMAX_CN_API_KEY", "MISTRAL_API_KEY", + "MOONSHOT_API_KEY", "NVIDIA_API_KEY", "OPENCODE_API_KEY", "OPENAI_API_KEY", + "OPENROUTER_API_KEY", "TOGETHER_API_KEY", "XAI_API_KEY", "XIAOMI_API_KEY", + "XIAOMI_TOKEN_PLAN_AMS_API_KEY", "XIAOMI_TOKEN_PLAN_CN_API_KEY", + "XIAOMI_TOKEN_PLAN_SGP_API_KEY", "ZAI_API_KEY", "ZAI_CODING_CN_API_KEY", +]); + +const PROVIDER_KEY_ENV: Readonly> = { + "ant-ling": "ANT_LING_API_KEY", + anthropic: "ANTHROPIC_API_KEY", + cerebras: "CEREBRAS_API_KEY", + deepseek: "DEEPSEEK_API_KEY", fireworks: "FIREWORKS_API_KEY", + "github-copilot": "COPILOT_GITHUB_TOKEN", google: "GEMINI_API_KEY", + "google-vertex": "GOOGLE_CLOUD_API_KEY", groq: "GROQ_API_KEY", huggingface: "HF_TOKEN", + "kimi-coding": "KIMI_API_KEY", minimax: "MINIMAX_API_KEY", "minimax-cn": "MINIMAX_CN_API_KEY", + mistral: "MISTRAL_API_KEY", moonshotai: "MOONSHOT_API_KEY", "moonshotai-cn": "MOONSHOT_API_KEY", + nvidia: "NVIDIA_API_KEY", openai: "OPENAI_API_KEY", opencode: "OPENCODE_API_KEY", + "opencode-go": "OPENCODE_API_KEY", openrouter: "OPENROUTER_API_KEY", together: "TOGETHER_API_KEY", + "vercel-ai-gateway": "AI_GATEWAY_API_KEY", xai: "XAI_API_KEY", xiaomi: "XIAOMI_API_KEY", + "xiaomi-token-plan-ams": "XIAOMI_TOKEN_PLAN_AMS_API_KEY", + "xiaomi-token-plan-cn": "XIAOMI_TOKEN_PLAN_CN_API_KEY", + "xiaomi-token-plan-sgp": "XIAOMI_TOKEN_PLAN_SGP_API_KEY", zai: "ZAI_API_KEY", + "zai-coding-cn": "ZAI_CODING_CN_API_KEY", +}; +const COMPOUND_PROVIDERS = new Set([ + "amazon-bedrock", "azure-openai-responses", "cloudflare-ai-gateway", "cloudflare-workers-ai", +]); +const LOCAL_PROVIDERS = new Set(["ollama", "lmstudio", "local", "aritmolab", "faux"]); + +export function canonicalPiProvider(provider: string | undefined): string | undefined { + const value = provider?.trim().toLowerCase(); + if (!value) return undefined; + if (value === "gemini") return "google"; + return value; +} + +export interface CredentialFsOps { + lstat(path: string): Stats; + open(path: string, flags: number): number; + fstat(fd: number): Stats; + read(fd: number): string; + close(fd: number): void; +} +const realFs: CredentialFsOps = { + lstat: lstatSync, open: openSync, fstat: fstatSync, + read: (fd) => readFileSync(fd, "utf8"), close: closeSync, +}; + +function validSecretStat(info: Stats, docker: boolean): boolean { + const mode = info.mode & 0o777; + if (!info.isFile() || info.isSymbolicLink() || info.nlink !== 1 || info.size > 16_384) return false; + if (docker) return info.uid === 0 && mode === 0o444; + return info.uid === process.getuid?.() && (mode === 0o400 || mode === 0o600); +} + +function readCredential(file: string, fs: CredentialFsOps): string { + let fd: number | undefined; + try { + const docker = file.startsWith("/run/secrets/") && !file.slice("/run/secrets/".length).includes("/"); + if (file.startsWith("/run/secrets/") && !docker) throw new Error(); + if (docker) { + const parent = fs.lstat("/run/secrets"); + if (!parent.isDirectory() || parent.uid !== 0 || (parent.mode & 0o022) !== 0) throw new Error(); + } + const before = fs.lstat(file); + if (!validSecretStat(before, docker)) throw new Error(); + fd = fs.open(file, constants.O_RDONLY | constants.O_NOFOLLOW); + const opened = fs.fstat(fd); + if (!validSecretStat(opened, docker) || before.dev !== opened.dev || before.ino !== opened.ino) throw new Error(); + const value = fs.read(fd); + if (!value || /\s/.test(value)) throw new Error(); + return value; + } catch { + throw new Error("model provider credential is unavailable"); + } finally { + if (fd !== undefined) { + try { fs.close(fd); } catch { /* sanitized by design */ } + } + } +} + +export function buildPiChildEnv(opts: { + ambient?: NodeJS.ProcessEnv; + provider?: string; + credentialFile?: string; + additions?: NodeJS.ProcessEnv; + credentialValue?: string; + fsOps?: CredentialFsOps; +}): NodeJS.ProcessEnv { + const env = { ...(opts.ambient ?? process.env), ...opts.additions }; + delete env.PI_PROVIDER_API_KEY; + delete env.THT_SECRETS_FILE; + delete env.THT_DWH_API_KEY; + delete env.THT_VEC_API_KEY; + delete env.THT_VEC_WRITE_API_KEY; + delete env.THT_MODEL_API_KEY; + delete env.THT_SSL_CA; + delete env.THT_CA; + delete env.THT_VECTOR_BOOTSTRAP_PASSWORD; + delete env.THT_VECTOR_MIGRATOR_PASSWORD; + delete env.THT_VECTOR_READER_PASSWORD; + delete env.THT_VECTOR_WRITER_PASSWORD; + delete env.THT_MODEL_API_KEY_FILE; + delete env.THT_DWH_API_KEY_SECRET_FILE; + delete env.THT_VEC_API_KEY_SECRET_FILE; + delete env.THT_VEC_WRITE_API_KEY_SECRET_FILE; + delete env.THT_CA_SECRET_FILE; + delete env.THT_VECTOR_BOOTSTRAP_PASSWORD_SECRET_FILE; + delete env.THT_VECTOR_MIGRATOR_PASSWORD_SECRET_FILE; + delete env.THT_VECTOR_READER_PASSWORD_SECRET_FILE; + delete env.THT_VECTOR_WRITER_PASSWORD_SECRET_FILE; + delete env.THT_VECTOR_BOOTSTRAP_PASSWORD_FILE; + delete env.THT_VECTOR_MIGRATOR_PASSWORD_FILE; + delete env.THT_VECTOR_READER_PASSWORD_FILE; + delete env.THT_VECTOR_WRITER_PASSWORD_FILE; + delete env.THT_DWH_API_KEY_FILE; + delete env.THT_VEC_API_KEY_FILE; + delete env.THT_VEC_WRITE_API_KEY_FILE; + delete env.THT_SSL_CA_FILE; + for (const name of PI_0803_CREDENTIAL_ENV_NAMES) delete env[name]; + const provider = canonicalPiProvider(opts.provider); + if (provider && COMPOUND_PROVIDERS.has(provider)) { + throw new Error( + "compound credential bundles are unsupported by THT_MODEL_API_KEY_FILE; " + + "dedicated provider configuration is required", + ); + } + if (provider && !LOCAL_PROVIDERS.has(provider)) { + const envName = PROVIDER_KEY_ENV[provider]; + if (!envName || (!opts.credentialFile && opts.credentialValue === undefined)) { + throw new Error("model provider credential is unavailable"); + } + if (opts.credentialValue !== undefined) { + if (!opts.credentialValue || /\s/.test(opts.credentialValue)) { + throw new Error("model provider credential is unavailable"); + } + env[envName] = opts.credentialValue; + } + else if (opts.credentialFile) env[envName] = readCredential(opts.credentialFile, opts.fsOps ?? realFs); + else throw new Error("model provider credential is unavailable"); + } else if (opts.credentialFile && !provider) { + throw new Error("model provider credential is unavailable"); + } + return env; +} diff --git a/backend/src/routes/sessions.ts b/backend/src/routes/sessions.ts index cfc29212..ca85acd2 100644 --- a/backend/src/routes/sessions.ts +++ b/backend/src/routes/sessions.ts @@ -29,12 +29,13 @@ export function sessionRoutes( model: s.model, thinking: s.thinking, author: getUser(req).id, + question: b.question, }); rt.bridge.onClientEvent((e) => d.hub.publish(id, e.type, e)); return { id }; }); - app.get("/sessions", async () => d.tht.sessionList()); - app.get("/sessions/:id", async (req) => d.tht.sessionShow((req.params as any).id)); + app.get("/sessions", async () => d.tht.sessionList(d.getSettings().workspace)); + app.get("/sessions/:id", async (req) => d.tht.sessionShow((req.params as any).id, d.getSettings().workspace)); app.post("/sessions/:id/response", async (req, reply) => { const id = (req.params as any).id; const rt = d.mgr.get(id); @@ -50,7 +51,7 @@ export function sessionRoutes( }); app.post("/sessions/:id/resume", async (req, reply) => { const id = (req.params as any).id; - const manifest = (await d.tht.sessionShow(id)) as { status?: string; archived?: boolean } | null; + const manifest = (await d.tht.sessionShow(id, d.getSettings().workspace)) as { status?: string; archived?: boolean } | null; if (manifest?.status === "finalized" || manifest?.archived) { return reply.code(409).send({ error: "sessione in sola lettura (finalizzata o archiviata)" }); } @@ -77,6 +78,9 @@ export function sessionRoutes( "Access-Control-Allow-Origin": origin, "Access-Control-Allow-Credentials": "true", }); + // Send the handshake immediately. Without this, Node waits for the first event body and + // proxies/clients cannot establish an idle SSE subscription or inspect its headers. + reply.raw.flushHeaders(); const send = (event: string, data: object) => reply.raw.write(`event: ${event}\ndata: ${JSON.stringify(data)}\n\n`); const off = d.hub.subscribe(id, send, rt?.bridge.pendingWidget() ?? null); req.raw.on("close", off); @@ -100,7 +104,7 @@ export function sessionRoutes( app.delete("/sessions/:id", async (req, reply) => { const id = (req.params as any).id; d.mgr.teardown(id); // drop any live runtime before deleting on disk - await d.tht.deleteSession(id); + await d.tht.deleteSession(id, d.getSettings().workspace); return reply.code(204).send(); }); app.get("/sessions/:id/documents", async (req) => d.tht.documents((req.params as any).id)); diff --git a/backend/src/tht/tht-runner.ts b/backend/src/tht/tht-runner.ts index c93409b6..7bf1dd95 100644 --- a/backend/src/tht/tht-runner.ts +++ b/backend/src/tht/tht-runner.ts @@ -6,6 +6,7 @@ export interface ThtConfig { thtBin: string; harnessDir: string; configPath: string; + dataRoot?: string; } export interface SessionRow { @@ -61,8 +62,12 @@ export class ThtRunner { run(args: string[], workspace?: string): Promise<{ code: number; stdout: string; stderr: string }> { return new Promise((resolve) => { + const env: NodeJS.ProcessEnv = { ...process.env }; + delete env.THT_DATA_ROOT; + if (this.cfg.dataRoot !== undefined) env.THT_DATA_ROOT = this.cfg.dataRoot; const ch = spawn(this.cfg.thtBin, this.buildArgv(args, workspace), { cwd: this.cfg.harnessDir, + env, }); let stdout = ""; let stderr = ""; @@ -104,12 +109,12 @@ export class ThtRunner { return this.json<{ id: string }>(a, o.workspace); } - sessionList() { - return this.json(["session", "list", "--json"]); + sessionList(workspace?: string) { + return this.json(["session", "list", "--json"], workspace); } - sessionShow(id: string) { - return this.json(["session", "show", id, "--json"]); + sessionShow(id: string, workspace?: string) { + return this.json(["session", "show", id, "--json"], workspace); } sqlPreview(id: string, p: { limit?: number; offset?: number }) { @@ -136,7 +141,10 @@ export class ThtRunner { setGroup(id: string, group: string) { return this.ok(["session", "set-group", id, "--group", group]); } archive(id: string) { return this.ok(["session", "archive", id]); } unarchive(id: string) { return this.ok(["session", "unarchive", id]); } - deleteSession(id: string) { return this.ok(["session", "delete", id]); } + async deleteSession(id: string, workspace?: string) { + const { code, stderr } = await this.run(["session", "delete", id], workspace); + if (code !== 0) throw new Error(`tht session delete exit ${code}: ${stderr.trim()}`); + } documents(id: string) { return this.json(["session", "documents", id, "--json"]); } async ollamaEnsure(workspace: string, timeoutSec: number): Promise { diff --git a/backend/test/auth.test.ts b/backend/test/auth.test.ts index 7a68ac3a..0702549c 100644 --- a/backend/test/auth.test.ts +++ b/backend/test/auth.test.ts @@ -22,3 +22,17 @@ test("mode mock legge l'header", async () => { }); expect(res.json()).toEqual({ id: "alice" }); }); + +test("upstream mode requires the authenticated proxy identity header", async () => { + const app = Fastify(); + app.addHook("preHandler", authPreHandler("upstream")); + app.get("/me", async (req) => getUser(req)); + + expect((await app.inject({ method: "GET", url: "/me" })).statusCode).toBe(401); + const authenticated = await app.inject({ + method: "GET", + url: "/me", + headers: { "x-authenticated-user": "alice@example.test" }, + }); + expect(authenticated.json()).toEqual({ id: "alice@example.test" }); +}); diff --git a/backend/test/config.test.ts b/backend/test/config.test.ts index 9786f513..f10d0b5c 100644 --- a/backend/test/config.test.ts +++ b/backend/test/config.test.ts @@ -1,20 +1,57 @@ -import { test, expect } from "vitest"; +import { expect, test } from "vitest"; import { loadConfig } from "../src/config.js"; -test("loadConfig espone host con default dev-safe 127.0.0.1", () => { - expect(loadConfig({})).toMatchObject({ host: "127.0.0.1" }); -}); - -test("loadConfig rispetta HOST esplicito (necessario in container)", () => { - expect(loadConfig({ HOST: "0.0.0.0" })).toMatchObject({ host: "0.0.0.0" }); -}); - -test("loadConfig merge altri campi con l'host", () => { - expect( - loadConfig({ HOST: "0.0.0.0", THT_BIN: "/opt/venv/bin/tht", PORT: "9000" }) - ).toMatchObject({ +test("loadConfig accepts container listening and runtime paths", () => { + expect(loadConfig({ + HOST: "0.0.0.0", + PORT: "9000", + THT_HARNESS_DIR: "/app/harness", + THT_BIN: "/opt/venv/bin/tht", + PI_BIN: "/usr/local/bin/pi", + SETTINGS_FILE: "/data/settings/settings.json", + THT_DATA_ROOT: "/data", + })).toMatchObject({ host: "0.0.0.0", - thtBin: "/opt/venv/bin/tht", port: 9000, + harnessDir: "/app/harness", + thtBin: "/opt/venv/bin/tht", + piBin: "/usr/local/bin/pi", + settingsFile: "/data/settings/settings.json", + dataRoot: "/data", }); }); + +test("loadConfig keeps local development defaults", () => { + expect(loadConfig({})).toMatchObject({ + host: "127.0.0.1", + port: 8787, + harnessDir: "../harness", + thtBin: "tht", + piBin: "pi", + settingsFile: "data/settings.json", + }); + expect(loadConfig({}).dataRoot).toBeUndefined(); +}); + +test("loadConfig rejects unauthenticated public exposure", () => { + expect(() => loadConfig({ + THOTH_PUBLIC_EXPOSURE: "true", + AUTH_MODE: "none", + })).toThrow(/public exposure requires AUTH_MODE=upstream/); +}); + +test("loadConfig accepts an authenticated upstream trust boundary", () => { + expect(loadConfig({ + THOTH_PUBLIC_EXPOSURE: "true", + AUTH_MODE: "upstream", + }).authMode).toBe("upstream"); +}); + +test("loadConfig accepts only an absolute generic model key file", () => { + expect(loadConfig({ THT_MODEL_API_KEY_FILE: "/run/secrets/model_api_key" }).modelApiKeyFile) + .toBe("/run/secrets/model_api_key"); + expect(() => loadConfig({ THT_MODEL_API_KEY_FILE: "relative/key" })) + .toThrow(/model credential configuration is invalid/); + expect(() => loadConfig({ THT_MODEL_API_KEY_FILE: " /run/secrets/key" })) + .toThrow(/model credential configuration is invalid/); +}); diff --git a/backend/test/health.test.ts b/backend/test/health.test.ts index d94c24c4..b02fa401 100644 --- a/backend/test/health.test.ts +++ b/backend/test/health.test.ts @@ -2,9 +2,44 @@ import { test, expect } from "vitest"; import { buildApp } from "../src/app.js"; import { loadConfig } from "../src/config.js"; -test("GET /health ritorna ok", async () => { +test("GET /health reports process readiness without external services", async () => { const app = buildApp(loadConfig({ THT_HARNESS_DIR: "/tmp/h" })); const res = await app.inject({ method: "GET", url: "/health" }); expect(res.statusCode).toBe(200); + expect(res.headers["content-type"]).toContain("application/json"); + expect(res.json()).toEqual({ status: "ok" }); +}); + +test("GET /health remains available to container probes in upstream auth mode", async () => { + const app = buildApp(loadConfig({ + THT_HARNESS_DIR: "/tmp/h", + AUTH_MODE: "upstream", + THOTH_PUBLIC_EXPOSURE: "true", + })); + const res = await app.inject({ method: "GET", url: "/health" }); + expect(res.statusCode).toBe(200); expect(res.json()).toEqual({ status: "ok" }); }); + +test("SSE response headers are flushed before the first event", async () => { + const app = buildApp(loadConfig({ THT_HARNESS_DIR: "/tmp/h" })); + await app.listen({ port: 0, host: "127.0.0.1" }); + const port = (app.server.address() as { port: number }).port; + const controller = new AbortController(); + + try { + const response = await Promise.race([ + fetch(`http://127.0.0.1:${port}/sessions/header-probe/events`, { + signal: controller.signal, + }), + new Promise((_, reject) => + setTimeout(() => reject(new Error("SSE headers were not flushed")), 250), + ), + ]); + expect(response.headers.get("content-type")).toContain("text/event-stream"); + expect(response.headers.get("cache-control")).toBe("no-cache"); + } finally { + controller.abort(); + await app.close(); + } +}); diff --git a/backend/test/list-models.test.ts b/backend/test/list-models.test.ts index 6175cacd..f43532e4 100644 --- a/backend/test/list-models.test.ts +++ b/backend/test/list-models.test.ts @@ -1,6 +1,6 @@ import { test, expect } from "vitest"; import { spawn } from "node:child_process"; -import { mkdtempSync, writeFileSync, rmSync } from "node:fs"; +import { chmodSync, mkdtempSync, writeFileSync, rmSync } from "node:fs"; import { tmpdir } from "node:os"; import { join } from "node:path"; import path from "node:path"; @@ -51,3 +51,137 @@ test("createPiModelLister caches within ttl (spawns once for two calls)", async rmSync(path.dirname(script), { recursive: true, force: true }); } }); + +test("production model-list spawn preserves PATH and passes the portable data root", async () => { + const script = scriptWith([]); + const calls: any[][] = []; + const previousPath = process.env.PATH; + process.env.PATH = "/usr/local/bin:/usr/bin"; + try { + const lister = createPiModelLister(loadConfig({ + THT_HARNESS_DIR: "/app/harness", + PI_BIN: "/usr/local/bin/pi", + THT_DATA_ROOT: "/data", + }), { + spawnFn: (...args: any[]) => { + calls.push(args); + return spawn("node", [FAKE, script]) as any; + }, + }); + await lister(); + expect(calls[0][0]).toBe("/usr/local/bin/pi"); + expect(calls[0][1]).toEqual(["--mode", "rpc"]); + expect(calls[0][2]).toMatchObject({ + cwd: "/app/harness", + env: expect.objectContaining({ PATH: "/usr/local/bin:/usr/bin", THT_DATA_ROOT: "/data" }), + }); + } finally { + if (previousPath === undefined) delete process.env.PATH; + else process.env.PATH = previousPath; + rmSync(path.dirname(script), { recursive: true, force: true }); + } +}); + +test("model-list spawn scrubs ambient provider credentials and generic secret metadata", async () => { + const script = scriptWith([]); + const calls: any[][] = []; + const previousDataRoot = process.env.THT_DATA_ROOT; + const previousCredential = process.env.PI_PROVIDER_API_KEY; + process.env.THT_DATA_ROOT = "/ambient-must-not-leak"; + process.env.PI_PROVIDER_API_KEY = "must-not-leak"; + process.env.OPENAI_API_KEY = "must-not-leak"; + process.env.AWS_SECRET_ACCESS_KEY = "must-not-leak"; + process.env.CLOUDFLARE_ACCOUNT_ID = "must-not-leak"; + try { + const lister = createPiModelLister(loadConfig({ PI_BIN: "/usr/local/bin/pi" }), { + spawnFn: (...args: any[]) => { + calls.push(args); + return spawn("node", [FAKE, script]) as any; + }, + }); + await lister(); + expect(calls[0][2].env).not.toHaveProperty("THT_DATA_ROOT"); + expect(calls[0][2].env).not.toHaveProperty("PI_PROVIDER_API_KEY"); + expect(calls[0][2].env).not.toHaveProperty("OPENAI_API_KEY"); + expect(calls[0][2].env).not.toHaveProperty("AWS_SECRET_ACCESS_KEY"); + expect(calls[0][2].env).not.toHaveProperty("CLOUDFLARE_ACCOUNT_ID"); + } finally { + if (previousDataRoot === undefined) delete process.env.THT_DATA_ROOT; + else process.env.THT_DATA_ROOT = previousDataRoot; + if (previousCredential === undefined) delete process.env.PI_PROVIDER_API_KEY; + else process.env.PI_PROVIDER_API_KEY = previousCredential; + delete process.env.OPENAI_API_KEY; + delete process.env.AWS_SECRET_ACCESS_KEY; + delete process.env.CLOUDFLARE_ACCOUNT_ID; + rmSync(path.dirname(script), { recursive: true, force: true }); + } +}); + +test("model-list spawn loads only the selected canonical provider credential", async () => { + const script = scriptWith([]); + const secret = join(path.dirname(script), "model-key"); + writeFileSync(secret, "selected-secret", { mode: 0o600 }); + chmodSync(secret, 0o600); + const calls: any[][] = []; + const lister = createPiModelLister(loadConfig({ + PI_PROVIDER: "Gemini", + THT_MODEL_API_KEY_FILE: secret, + }), { + spawnFn: (...args: any[]) => { + calls.push(args); + return spawn("node", [FAKE, script]) as any; + }, + }); + try { + await lister(); + expect(calls[0][2].env.GEMINI_API_KEY).toBe("selected-secret"); + expect(calls[0][2].env).not.toHaveProperty("THT_MODEL_API_KEY_FILE"); + expect(JSON.stringify(calls[0].slice(0, 2))).not.toContain("selected-secret"); + } finally { + rmSync(path.dirname(script), { recursive: true, force: true }); + } +}); + +test("model-list spawn uses the same single secret bundle as sessions", async () => { + const script = scriptWith([]); + const bundle = join(path.dirname(script), "bundle"); + writeFileSync(bundle, "THT_MODEL_API_KEY=selected-bundle-secret\n", { mode: 0o600 }); + const calls: any[][] = []; + const lister = createPiModelLister(loadConfig({ + PI_PROVIDER: "openai", THT_SECRETS_FILE: bundle, + }), { + spawnFn: (...args: any[]) => { + calls.push(args); + return spawn("node", [FAKE, script]) as any; + }, + }); + try { + await lister(); + expect(calls[0][2].env.OPENAI_API_KEY).toBe("selected-bundle-secret"); + expect(calls[0][2].env).not.toHaveProperty("THT_SECRETS_FILE"); + } finally { + rmSync(path.dirname(script), { recursive: true, force: true }); + } +}); + +test.each(["amazon-bedrock", "azure-openai-responses", "cloudflare-workers-ai", "cloudflare-ai-gateway"])( + "model listing rejects compound provider %s before spawning Pi", async (provider) => { + const script = scriptWith([]); + const secret = join(path.dirname(script), "model-key"); + writeFileSync(secret, "selected-secret", { mode: 0o600 }); + let spawns = 0; + const lister = createPiModelLister(loadConfig({ + PI_PROVIDER: provider, THT_MODEL_API_KEY_FILE: secret, + }), { + spawnFn: () => { spawns += 1; throw new Error("must not spawn"); }, + }); + try { + await expect(lister()).rejects.toThrow( + "compound credential bundles are unsupported by THT_MODEL_API_KEY_FILE; dedicated provider configuration is required", + ); + expect(spawns).toBe(0); + } finally { + rmSync(path.dirname(script), { recursive: true, force: true }); + } + }, +); diff --git a/backend/test/pi-process-manager.test.ts b/backend/test/pi-process-manager.test.ts index 53e574e2..413e2ee4 100644 --- a/backend/test/pi-process-manager.test.ts +++ b/backend/test/pi-process-manager.test.ts @@ -1,8 +1,9 @@ -import { test, expect } from "vitest"; +import { test, expect, vi } from "vitest"; import { spawn } from "node:child_process"; import path from "node:path"; import { fileURLToPath } from "node:url"; import { EventEmitter } from "node:events"; +import { chmodSync, writeFileSync } from "node:fs"; import { PiProcessManager } from "../src/pi/pi-process-manager.js"; import { loadConfig } from "../src/config.js"; @@ -125,3 +126,228 @@ test("spawnFor default (new) mode sends /nuova-domanda", async () => { expect(child._writes.join("")).toContain("/nuova-domanda"); mgr.teardown("sid-10"); }); + +test("spawnFor new mode forwards the real question instead of kickoff", async () => { + const cfg = loadConfig({}); + const child = recordingChild(); + const mgr = new PiProcessManager(cfg, { spawnFn: () => child as any }); + await mgr.spawnFor("sid-question", { question: "pazienti con cardioversione e ILR" }); + const prompt = JSON.parse(child._writes.at(-1)!); + expect(prompt.message).toBe('/nuova-domanda "pazienti con cardioversione e ILR"'); + expect(prompt.message).not.toContain("kickoff"); + mgr.teardown("sid-question"); +}); + +test("production spawn uses explicit Pi path and passes portable data root without rewriting PATH", async () => { + vi.stubEnv("PATH", "/usr/local/bin:/usr/bin"); + vi.stubEnv("PI_PROVIDER_API_KEY", "provider-secret"); + vi.stubEnv("NODE_EXTRA_CA_CERTS", "/certs/company-ca.pem"); + const calls: any[][] = []; + const child = recordingChild(); + child.stderr.resume = () => {}; + const spawnFn = (...args: any[]) => { calls.push(args); return child as any; }; + const cfg = loadConfig({ + THT_HARNESS_DIR: "/app/harness", + PI_BIN: "/usr/local/bin/pi", + THT_DATA_ROOT: "/data", + }); + const mgr = new PiProcessManager(cfg, { spawnFn }); + + try { + await mgr.spawnFor("portable-session", { author: "user@example.test" }); + const [bin, args, options] = calls[0]; + expect(bin).toBe("/usr/local/bin/pi"); + expect(args).toEqual(["--mode", "rpc"]); + expect(options.cwd).toBe("/app/harness"); + expect(options.env).toMatchObject({ + PATH: "/usr/local/bin:/usr/bin", + NODE_EXTRA_CA_CERTS: "/certs/company-ca.pem", + THT_DATA_ROOT: "/data", + THT_SESSION: "portable-session", + THT_AUTHOR: "user@example.test", + }); + expect(options.env).not.toHaveProperty("PI_PROVIDER_API_KEY"); + } finally { + mgr.teardown("portable-session"); + vi.unstubAllEnvs(); + } +}); + +test("session Pi spawn omits ambient THT_DATA_ROOT when config does not provide one", async () => { + vi.stubEnv("THT_DATA_ROOT", "/ambient-must-not-leak"); + vi.stubEnv("PI_PROVIDER_API_KEY", "still-inherited"); + const calls: any[][] = []; + const child = recordingChild(); + child.stderr.resume = () => {}; + const mgr = new PiProcessManager(loadConfig({ PI_BIN: "/usr/local/bin/pi" }), { + spawnFn: (...args: any[]) => { calls.push(args); return child as any; }, + }); + + try { + await mgr.spawnFor("no-data-root", {}); + expect(calls[0][2].env).not.toHaveProperty("THT_DATA_ROOT"); + expect(calls[0][2].env).not.toHaveProperty("PI_PROVIDER_API_KEY"); + } finally { + mgr.teardown("no-data-root"); + vi.unstubAllEnvs(); + } +}); + +test.each([ + ["anthropic", "ANTHROPIC_API_KEY"], + ["OpenAI", "OPENAI_API_KEY"], + ["gemini", "GEMINI_API_KEY"], + ["google", "GEMINI_API_KEY"], + ["deepseek", "DEEPSEEK_API_KEY"], + ["zai", "ZAI_API_KEY"], + ["openrouter", "OPENROUTER_API_KEY"], +])("injects the generic file credential only as %s provider env", async (provider, expectedName) => { + const secret = path.resolve(__dirname, `.model-key-${process.pid}-${provider}`); + writeFileSync(secret, "provider-secret", { mode: 0o600 }); + const calls: any[][] = []; + const child = recordingChild(); + child.stderr.resume = () => {}; + const mgr = new PiProcessManager(loadConfig({ + PI_BIN: "/usr/local/bin/pi", THT_MODEL_API_KEY_FILE: secret, + }), { spawnFn: (...args: any[]) => { calls.push(args); return child as any; } }); + try { + await mgr.spawnFor("credential-session", { provider }); + const env = calls[0][2].env; + expect(env[expectedName]).toBe("provider-secret"); + expect(env).not.toHaveProperty("PI_PROVIDER_API_KEY"); + expect(env).not.toHaveProperty("THT_MODEL_API_KEY_FILE"); + expect(JSON.stringify(calls[0].slice(0, 2))).not.toContain("provider-secret"); + } finally { + mgr.teardown("credential-session"); + await import("node:fs/promises").then((fs) => fs.unlink(secret)); + } +}); + +test("session Pi spawn reads the single secret bundle and scrubs its path", async () => { + const secret = path.resolve(__dirname, `.bundle-${process.pid}`); + writeFileSync(secret, "THT_MODEL_API_KEY=bundle-secret\n", { mode: 0o600 }); + chmodSync(secret, 0o600); + const calls: any[][] = []; + const child = recordingChild(); + child.stderr.resume = () => {}; + const mgr = new PiProcessManager(loadConfig({ + PI_BIN: "/usr/local/bin/pi", THT_SECRETS_FILE: secret, + }), { spawnFn: (...args: any[]) => { calls.push(args); return child as any; } }); + try { + await mgr.spawnFor("bundle-session", { provider: "openai" }); + expect(calls[0][2].env.OPENAI_API_KEY).toBe("bundle-secret"); + expect(calls[0][2].env).not.toHaveProperty("THT_SECRETS_FILE"); + } finally { + mgr.teardown("bundle-session"); + await import("node:fs/promises").then((fs) => fs.unlink(secret)); + } +}); + +test.each([["OpenAI", "openai"], ["gemini", "google"]])( + "set_model uses canonical packaged provider ID for %s", async (provider, canonical) => { + const secret = path.resolve(__dirname, `.canonical-key-${process.pid}-${provider}`); + writeFileSync(secret, "provider-secret", { mode: 0o600 }); + const child = recordingChild(); + child.stderr.resume = () => {}; + child.stdin.write = (data: unknown) => { + const request = JSON.parse(String(data)); + child._writes.push(String(data)); + if (request.id) { + queueMicrotask(() => child.stdout.emit("data", `${JSON.stringify({ + type: "response", id: request.id, success: true, + })}\n`)); + } + return true; + }; + const mgr = new PiProcessManager(loadConfig({ THT_MODEL_API_KEY_FILE: secret }), { + spawnFn: () => child as any, + }); + try { + await mgr.spawnFor("canonical-provider", { provider, model: "model-id" }); + expect(child._writes.join("")).toContain(`\"provider\":\"${canonical}\"`); + } finally { + mgr.teardown("canonical-provider"); + await import("node:fs/promises").then((fs) => fs.unlink(secret)); + } + }, +); + +test("local providers spawn without a model key and scrub ambient generic credentials", async () => { + vi.stubEnv("PI_PROVIDER_API_KEY", "ambient-secret"); + vi.stubEnv("THT_MODEL_API_KEY_FILE", "/ambient/secret-path"); + vi.stubEnv("OPENAI_API_KEY", "unselected-provider-secret"); + const calls: any[][] = []; + const child = recordingChild(); + child.stderr.resume = () => {}; + const mgr = new PiProcessManager(loadConfig({ PI_BIN: "/usr/local/bin/pi" }), { + spawnFn: (...args: any[]) => { calls.push(args); return child as any; }, + }); + try { + await mgr.spawnFor("local-session", { provider: "ollama" }); + expect(calls[0][2].env).not.toHaveProperty("PI_PROVIDER_API_KEY"); + expect(calls[0][2].env).not.toHaveProperty("THT_MODEL_API_KEY_FILE"); + expect(calls[0][2].env).not.toHaveProperty("OPENAI_API_KEY"); + } finally { + mgr.teardown("local-session"); + vi.unstubAllEnvs(); + } +}); + +test.each(["amazon-bedrock", "azure-openai-responses", "cloudflare-workers-ai", "cloudflare-ai-gateway"])( + "session spawn rejects compound provider %s before spawning Pi", async (provider) => { + const secret = path.resolve(__dirname, `.compound-key-${process.pid}-${provider}`); + writeFileSync(secret, "provider-secret", { mode: 0o600 }); + let spawns = 0; + const mgr = new PiProcessManager(loadConfig({ THT_MODEL_API_KEY_FILE: secret }), { + spawnFn: () => { spawns += 1; throw new Error("must not spawn"); }, + }); + try { + await expect(mgr.spawnFor("compound-provider", { provider })).rejects.toThrow( + "compound credential bundles are unsupported by THT_MODEL_API_KEY_FILE; dedicated provider configuration is required", + ); + expect(spawns).toBe(0); + } finally { + await import("node:fs/promises").then((fs) => fs.unlink(secret)); + } + }, +); + +test.each(["missing", "permissive", "unreadable", "directory", "symlink", "unsupported"])( + "hosted provider credential failure is sanitized: %s", async (kind) => { + const target = path.resolve(__dirname, `.bad-model-key-${process.pid}-${kind}`); + if (kind === "permissive") { + writeFileSync(target, "DO_NOT_LEAK", { mode: 0o644 }); + chmodSync(target, 0o644); + } else if (kind === "unreadable") { + writeFileSync(target, "DO_NOT_LEAK", { mode: 0o000 }); + } else if (kind === "directory") { + await import("node:fs/promises").then((fs) => fs.mkdir(target)); + } else if (kind === "symlink") { + const source = `${target}-source`; + writeFileSync(source, "DO_NOT_LEAK", { mode: 0o600 }); + await import("node:fs/promises").then((fs) => fs.symlink(source, target)); + } + const cfg = loadConfig({ THT_MODEL_API_KEY_FILE: target }); + const mgr = new PiProcessManager(cfg, { spawnFn: () => { + throw new Error("spawn must not occur"); + } }); + try { + const provider = kind === "unsupported" ? "unknown-hosted" : "anthropic"; + await expect(mgr.spawnFor("bad-secret", { provider })) + .rejects.toThrow("model provider credential is unavailable"); + } finally { + if (kind === "permissive") await import("node:fs/promises").then((fs) => fs.unlink(target)); + if (kind === "unreadable") { + chmodSync(target, 0o600); + await import("node:fs/promises").then((fs) => fs.unlink(target)); + } + if (kind === "directory") await import("node:fs/promises").then((fs) => fs.rmdir(target)); + if (kind === "symlink") { + await import("node:fs/promises").then(async (fs) => { + await fs.unlink(target); + await fs.unlink(`${target}-source`); + }); + } + } + }, +); diff --git a/backend/test/provider-credentials.test.ts b/backend/test/provider-credentials.test.ts new file mode 100644 index 00000000..e8352845 --- /dev/null +++ b/backend/test/provider-credentials.test.ts @@ -0,0 +1,148 @@ +import { expect, test } from "vitest"; +import { readFileSync } from "node:fs"; +import path from "node:path"; +import { + PI_0803_CREDENTIAL_ENV_NAMES, + buildPiChildEnv, + canonicalPiProvider, +} from "../src/pi/provider-credentials.js"; + +test("canonical provider aliases resolve to packaged Pi 0.80.3 IDs", () => { + expect(canonicalPiProvider(" OpenAI ")).toBe("openai"); + expect(canonicalPiProvider("gemini")).toBe("google"); + expect(canonicalPiProvider("Google")).toBe("google"); + expect(() => buildPiChildEnv({ + ambient: {}, provider: "cohere", credentialFile: "/unused", + })).toThrow("model provider credential is unavailable"); +}); + +test("credential scrub list matches the audited Pi AI 0.80.3 provider definitions", () => { + const lock = JSON.parse(readFileSync(path.resolve("../docker/pi-runtime/package-lock.json"), "utf8")); + expect(lock.packages["node_modules/@earendil-works/pi-coding-agent/node_modules/@earendil-works/pi-ai"].version) + .toBe("0.80.3"); + expect(PI_0803_CREDENTIAL_ENV_NAMES).toEqual([ + "AI_GATEWAY_API_KEY", "ANTHROPIC_API_KEY", "ANTHROPIC_OAUTH_TOKEN", "ANT_LING_API_KEY", + "AWS_ACCESS_KEY_ID", "AWS_BEARER_TOKEN_BEDROCK", "AWS_CONFIG_FILE", + "AWS_CONTAINER_AUTHORIZATION_TOKEN", "AWS_CONTAINER_AUTHORIZATION_TOKEN_FILE", + "AWS_CONTAINER_CREDENTIALS_FULL_URI", "AWS_CONTAINER_CREDENTIALS_RELATIVE_URI", "AWS_PROFILE", + "AWS_ROLE_ARN", "AWS_ROLE_SESSION_NAME", "AWS_SECRET_ACCESS_KEY", "AWS_SESSION_TOKEN", + "AWS_SHARED_CREDENTIALS_FILE", "AWS_WEB_IDENTITY_TOKEN_FILE", "AZURE_OPENAI_API_KEY", + "CEREBRAS_API_KEY", "CLOUDFLARE_ACCOUNT_ID", "CLOUDFLARE_API_KEY", + "CLOUDFLARE_GATEWAY_ID", "COPILOT_GITHUB_TOKEN", "DEEPSEEK_API_KEY", "FIREWORKS_API_KEY", + "GCLOUD_PROJECT", "GEMINI_API_KEY", "GOOGLE_APPLICATION_CREDENTIALS", "GOOGLE_CLOUD_API_KEY", + "GOOGLE_CLOUD_LOCATION", "GOOGLE_CLOUD_PROJECT", "GROQ_API_KEY", "HF_TOKEN", + "KIMI_API_KEY", "MINIMAX_API_KEY", "MINIMAX_CN_API_KEY", "MISTRAL_API_KEY", + "MOONSHOT_API_KEY", "NVIDIA_API_KEY", "OPENCODE_API_KEY", "OPENAI_API_KEY", + "OPENROUTER_API_KEY", "TOGETHER_API_KEY", "XAI_API_KEY", "XIAOMI_API_KEY", + "XIAOMI_TOKEN_PLAN_AMS_API_KEY", "XIAOMI_TOKEN_PLAN_CN_API_KEY", + "XIAOMI_TOKEN_PLAN_SGP_API_KEY", "ZAI_API_KEY", "ZAI_CODING_CN_API_KEY", + ]); + expect(PI_0803_CREDENTIAL_ENV_NAMES).not.toContain("COHERE_API_KEY"); +}); + +test("credential file replacement between lstat and open is rejected before read", () => { + let reads = 0; + const stat = (ino: number) => ({ + dev: 7, ino, uid: process.getuid?.() ?? 0, mode: 0o100600, nlink: 1, size: 6, + isFile: () => true, isDirectory: () => false, isSymbolicLink: () => false, + }); + expect(() => buildPiChildEnv({ + ambient: {}, provider: "openai", credentialFile: "/safe/key", + fsOps: { + lstat: () => stat(1) as any, + open: () => 9, + fstat: () => stat(2) as any, + read: () => { reads += 1; return "secret"; }, + close: () => undefined, + }, + })).toThrow("model provider credential is unavailable"); + expect(reads).toBe(0); +}); + +test("Docker secrets require a secure root-owned /run/secrets parent and 0444 file", () => { + const stat = (kind: "parent" | "file") => ({ + dev: 7, ino: 1, uid: kind === "parent" ? 1000 : 0, + mode: kind === "parent" ? 0o40755 : 0o100444, nlink: 1, size: 6, + isFile: () => kind === "file", isDirectory: () => kind === "parent", + isSymbolicLink: () => false, + }); + expect(() => buildPiChildEnv({ + ambient: {}, provider: "openai", credentialFile: "/run/secrets/model-key", + fsOps: { + lstat: (file) => stat(file === "/run/secrets" ? "parent" : "file") as any, + open: () => 9, fstat: () => stat("file") as any, read: () => "secret", close: () => undefined, + }, + })).toThrow("model provider credential is unavailable"); +}); + +test.each(["amazon-bedrock", "azure-openai-responses", "cloudflare-workers-ai", "cloudflare-ai-gateway"])( + "compound provider %s fails closed before opening the generic secret", (provider) => { + let opens = 0; + expect(() => buildPiChildEnv({ + ambient: {}, provider, credentialFile: "/unused", + fsOps: { + lstat: () => { throw new Error("must not inspect file"); }, + open: () => { opens += 1; return 9; }, + fstat: () => { throw new Error("must not inspect file"); }, + read: () => "secret", close: () => undefined, + }, + })).toThrow( + "compound credential bundles are unsupported by THT_MODEL_API_KEY_FILE; dedicated provider configuration is required", + ); + expect(opens).toBe(0); + }, +); + +test("single-key providers scrub ambient compound companions before injecting their key", () => { + const stat = { + dev: 7, ino: 1, uid: process.getuid?.() ?? 0, mode: 0o100600, nlink: 1, size: 6, + isFile: () => true, isDirectory: () => false, isSymbolicLink: () => false, + }; + const env = buildPiChildEnv({ + ambient: { + AWS_ACCESS_KEY_ID: "ambient", AWS_SECRET_ACCESS_KEY: "ambient", + CLOUDFLARE_ACCOUNT_ID: "ambient", CLOUDFLARE_GATEWAY_ID: "ambient", + }, + provider: "openai", credentialFile: "/safe/key", + fsOps: { + lstat: () => stat as any, open: () => 9, fstat: () => stat as any, + read: () => "secret", close: () => undefined, + }, + }); + expect(env.OPENAI_API_KEY).toBe("secret"); + expect(env).not.toHaveProperty("AWS_ACCESS_KEY_ID"); + expect(env).not.toHaveProperty("AWS_SECRET_ACCESS_KEY"); + expect(env).not.toHaveProperty("CLOUDFLARE_ACCOUNT_ID"); + expect(env).not.toHaveProperty("CLOUDFLARE_GATEWAY_ID"); +}); + +test("bundle value is injected without exposing bundle metadata to Pi", () => { + const env = buildPiChildEnv({ + ambient: { + THT_SECRETS_FILE: "/run/secrets/thothii.secrets", THT_MODEL_API_KEY_FILE: "/run/secrets/model", + THT_DWH_API_KEY: "dwh-secret", THT_VEC_API_KEY: "vec-reader-secret", + THT_VEC_WRITE_API_KEY: "vec-writer-secret", THT_SSL_CA: "/run/secrets/ca.pem", + THT_CA: "/run/secrets/ca.pem", THT_DWH_API_KEY_SECRET_FILE: "/run/secrets/dwh", + THT_MODEL_API_KEY: "model-secret", THT_VECTOR_READER_PASSWORD: "reader-secret", + THT_VECTOR_READER_PASSWORD_FILE: "/tmp/thothii-secrets/reader", + THT_VECTOR_WRITER_PASSWORD_FILE: "/tmp/thothii-secrets/writer", + THT_DWH_API_KEY_FILE: "/run/secrets/dwh", THT_VEC_API_KEY_FILE: "/run/secrets/vec", + }, + provider: "openai", credentialValue: "bundle-secret", + }); + expect(env.OPENAI_API_KEY).toBe("bundle-secret"); + expect(env).not.toHaveProperty("THT_SECRETS_FILE"); + expect(env).not.toHaveProperty("THT_MODEL_API_KEY_FILE"); + expect(env).not.toHaveProperty("THT_DWH_API_KEY"); + expect(env).not.toHaveProperty("THT_VEC_API_KEY"); + expect(env).not.toHaveProperty("THT_VEC_WRITE_API_KEY"); + expect(env).not.toHaveProperty("THT_MODEL_API_KEY"); + expect(env).not.toHaveProperty("THT_SSL_CA"); + expect(env).not.toHaveProperty("THT_CA"); + expect(env).not.toHaveProperty("THT_VECTOR_READER_PASSWORD"); + expect(env).not.toHaveProperty("THT_DWH_API_KEY_SECRET_FILE"); + expect(env).not.toHaveProperty("THT_VECTOR_READER_PASSWORD_FILE"); + expect(env).not.toHaveProperty("THT_VECTOR_WRITER_PASSWORD_FILE"); + expect(env).not.toHaveProperty("THT_DWH_API_KEY_FILE"); + expect(env).not.toHaveProperty("THT_VEC_API_KEY_FILE"); +}); diff --git a/backend/test/routes-sessions.test.ts b/backend/test/routes-sessions.test.ts index e54bc278..e0122449 100644 --- a/backend/test/routes-sessions.test.ts +++ b/backend/test/routes-sessions.test.ts @@ -1,6 +1,8 @@ import { test, expect } from "vitest"; import { spawn as nodeSpawn } from "node:child_process"; import path from "node:path"; +import os from "node:os"; +import { chmodSync, unlinkSync, writeFileSync } from "node:fs"; import { buildApp } from "../src/app.js"; import { loadConfig } from "../src/config.js"; @@ -16,9 +18,14 @@ function mutApp(thtRunner: any) { } test("POST /sessions usa i settings (workspace/provider/model/thinking) e crea+avvia", async () => { + const modelKey = path.join(os.tmpdir(), `thoth-model-key-${process.pid}`); + writeFileSync(modelKey, "test-model-key", { mode: 0o600 }); + chmodSync(modelKey, 0o600); let sessionNewArg: any; let spawnArg: any; - const app = buildApp(loadConfig({ THT_HARNESS_DIR: "../harness" }), { + const app = buildApp(loadConfig({ + THT_HARNESS_DIR: "../harness", THT_MODEL_API_KEY_FILE: modelKey, + }), { thtRunner: { ollamaEnsure: async () => ({ ok: true }), sessionNew: async (o: any) => { sessionNewArg = o; return { id: "s1" }; }, @@ -36,6 +43,7 @@ test("POST /sessions usa i settings (workspace/provider/model/thinking) e crea+a expect(sessionNewArg.question).toBe("q"); const list = await app.inject({ method: "GET", url: "/sessions" }); expect(list.json()).toEqual([{ id: "s1" }]); + unlinkSync(modelKey); }); test("POST /sessions/:id/response inoltra al bridge (no error)", async () => { diff --git a/backend/test/secret-bundle.test.ts b/backend/test/secret-bundle.test.ts new file mode 100644 index 00000000..a3f2b550 --- /dev/null +++ b/backend/test/secret-bundle.test.ts @@ -0,0 +1,74 @@ +import { afterEach, expect, test } from "vitest"; +import { chmodSync, mkdtempSync, renameSync, rmSync, writeFileSync } from "node:fs"; +import { join } from "node:path"; +import { tmpdir } from "node:os"; +import { loadSecretBundle, loadSecretBundleWithFs, secretValue } from "../src/config/secret-bundle.js"; + +const dirs: string[] = []; +afterEach(() => { for (const dir of dirs.splice(0)) rmSync(dir, { recursive: true, force: true }); }); +function bundle(contents: string, mode = 0o600): string { + const dir = mkdtempSync(join(tmpdir(), "thothii-secret-bundle-")); + dirs.push(dir); + const file = join(dir, "bundle"); + writeFileSync(file, contents, { mode }); + chmodSync(file, mode); + return file; +} + +test("parses comments, blank lines and values containing equals", () => { + const file = bundle("# comment\n\nTHT_MODEL_API_KEY=abc=123\nTHT_DWH_API_KEY=dwh\n"); + expect(loadSecretBundle(file)).toEqual(new Map([ + ["THT_MODEL_API_KEY", "abc=123"], ["THT_DWH_API_KEY", "dwh"], + ])); +}); + +test.each([ + ["duplicate", "THT_MODEL_API_KEY=a\nTHT_MODEL_API_KEY=b\n"], + ["unknown", "UNKNOWN_KEY=x\n"], + ["empty", "THT_MODEL_API_KEY=\n"], + ["syntax", "THT_MODEL_API_KEY\n"], +])("rejects %s bundle lines without exposing values", (_name, contents) => { + expect(() => loadSecretBundle(bundle(contents))).toThrow("secret bundle is unavailable"); + expect(() => loadSecretBundle(bundle(contents))).not.toThrow(/abc|dwh/); +}); + +test("rejects missing and insecure files", () => { + const file = bundle("THT_MODEL_API_KEY=secret\n", 0o644); + expect(() => loadSecretBundle(file)).toThrow("secret bundle is unavailable"); + expect(() => loadSecretBundle(join(file, "missing"))).toThrow("secret bundle is unavailable"); +}); + +test("checks inode identity before parsing", () => { + const file = bundle("THT_MODEL_API_KEY=secret\n"); + const replacement = `${file}.replacement`; + writeFileSync(replacement, "THT_MODEL_API_KEY=replaced\n", { mode: 0o600 }); + // A real replacement is safe because the loader's open/fstat check is the invariant; + // this also ensures the normal post-replacement file remains parseable. + renameSync(replacement, file); + expect(loadSecretBundle(file).get("THT_MODEL_API_KEY")).toBe("replaced"); +}); + +test("rejects inode replacement between lstat and open without reading", () => { + let reads = 0; + const stat = (ino: number) => ({ + dev: 7, ino, uid: process.getuid?.() ?? 0, mode: 0o100600, nlink: 1, size: 24, + isFile: () => true, isDirectory: () => false, isSymbolicLink: () => false, + }); + expect(() => loadSecretBundleWithFs("/safe/bundle", { + lstat: () => stat(1) as any, + open: () => 9, + fstat: () => stat(2) as any, + read: () => { reads += 1; return "THT_MODEL_API_KEY=secret\n"; }, + close: () => undefined, + })).toThrow("secret bundle is unavailable"); + expect(reads).toBe(0); +}); + +test("secretValue prefers bundle and supports the legacy file fallback", () => { + const file = bundle("THT_MODEL_API_KEY=from-bundle\n"); + const legacy = bundle("from-legacy"); + expect(secretValue({ secretsFile: file, secretFiles: { THT_MODEL_API_KEY_SECRET_FILE: legacy } }, "THT_MODEL_API_KEY")) + .toBe("from-bundle"); + expect(secretValue({ secretFiles: { THT_MODEL_API_KEY_SECRET_FILE: legacy } }, "THT_MODEL_API_KEY")) + .toBe("from-legacy"); +}); diff --git a/backend/test/tht-runner.test.ts b/backend/test/tht-runner.test.ts index 39c3f709..753160aa 100644 --- a/backend/test/tht-runner.test.ts +++ b/backend/test/tht-runner.test.ts @@ -24,6 +24,54 @@ test("sessionNew parses id from JSON", async () => { expect(await r.sessionNew({ question: "q" })).toEqual({ id: "2026-06-27-100000-x" }); }); +test("run passes configured THT_DATA_ROOT and preserves the remaining environment", async () => { + const previousDataRoot = process.env.THT_DATA_ROOT; + const previousCa = process.env.NODE_EXTRA_CA_CERTS; + process.env.THT_DATA_ROOT = "/ambient"; + process.env.NODE_EXTRA_CA_CERTS = "/certs/company-ca.pem"; + try { + (spawn as any).mockClear(); + const r = new ThtRunner({ + thtBin: "/opt/venv/bin/tht", + harnessDir: "/app/harness", + configPath: "config/tht.yaml", + dataRoot: "/configured", + }); + await r.run(["session", "list", "--json"]); + const [bin, , options] = (spawn as any).mock.calls[0]; + expect(bin).toBe("/opt/venv/bin/tht"); + expect(options.env).toMatchObject({ + THT_DATA_ROOT: "/configured", + NODE_EXTRA_CA_CERTS: "/certs/company-ca.pem", + }); + } finally { + if (previousDataRoot === undefined) delete process.env.THT_DATA_ROOT; + else process.env.THT_DATA_ROOT = previousDataRoot; + if (previousCa === undefined) delete process.env.NODE_EXTRA_CA_CERTS; + else process.env.NODE_EXTRA_CA_CERTS = previousCa; + } +}); + +test("run omits ambient THT_DATA_ROOT when config does not provide one", async () => { + const previousDataRoot = process.env.THT_DATA_ROOT; + const previousCredential = process.env.PI_PROVIDER_API_KEY; + process.env.THT_DATA_ROOT = "/ambient-must-not-leak"; + process.env.PI_PROVIDER_API_KEY = "still-inherited"; + try { + (spawn as any).mockClear(); + const r = new ThtRunner({ thtBin: "tht", harnessDir: "/h", configPath: "config/tht.yaml" }); + await r.run(["session", "list", "--json"]); + const options = (spawn as any).mock.calls[0][2]; + expect(options.env).not.toHaveProperty("THT_DATA_ROOT"); + expect(options.env.PI_PROVIDER_API_KEY).toBe("still-inherited"); + } finally { + if (previousDataRoot === undefined) delete process.env.THT_DATA_ROOT; + else process.env.THT_DATA_ROOT = previousDataRoot; + if (previousCredential === undefined) delete process.env.PI_PROVIDER_API_KEY; + else process.env.PI_PROVIDER_API_KEY = previousCredential; + } +}); + test("run with exit != 0 propagates error with stderr", async () => { const r = new ThtRunner({ thtBin: "tht", harnessDir: "/h", configPath: "config/tht.yaml" }); r.run = async () => ({ code: 1, stdout: "", stderr: "ERRORE: boom" }); diff --git a/deploy/compose.local-vector.yaml b/deploy/compose.local-vector.yaml new file mode 100644 index 00000000..0bf41337 --- /dev/null +++ b/deploy/compose.local-vector.yaml @@ -0,0 +1,90 @@ +services: + core: + profiles: [local-vector] + environment: + THT_VECTOR_DATABASE: "${THT_VECTOR_DATABASE:-thoth}" + THT_VECTOR_BOOTSTRAP_USER: "${THT_VECTOR_BOOTSTRAP_USER:-postgres}" + THT_VECTOR_READER_USER: "${THT_VECTOR_READER_USER:-thoth_vector_reader}" + THT_VECTOR_WRITER_USER: "${THT_VECTOR_WRITER_USER:-thoth_vector_writer}" + THT_SECRETS_FILE: /run/secrets/thothii.secrets + secrets: [{source: thothii_secrets, target: thothii.secrets}] + depends_on: + vector-migrate: + condition: service_completed_successfully + + frontend: + profiles: [local-vector] + + vector-db: + image: pgvector/pgvector:0.8.5-pg16@sha256:1d533553fefe4f12e5d80c7b80622ba0c382abb5758856f52983d8789179f0fb + profiles: [local-vector] + labels: {io.thothii.smoke-owner: "${THOTH_SMOKE_OWNER:-operator}"} + environment: + POSTGRES_DB: "${THT_VECTOR_DATABASE:-thoth}" + POSTGRES_USER: "${THT_VECTOR_BOOTSTRAP_USER:-postgres}" + THT_VECTOR_MIGRATOR_USER: "${THT_VECTOR_MIGRATOR_USER:-thoth_vector_migrator}" + THT_VECTOR_READER_USER: "${THT_VECTOR_READER_USER:-thoth_vector_reader}" + THT_VECTOR_WRITER_USER: "${THT_VECTOR_WRITER_USER:-thoth_vector_writer}" + THT_SECRETS_FILE: /run/secrets/thothii.secrets + secrets: [{source: thothii_secrets, target: thothii.secrets}] + entrypoint: [/opt/thoth/vector-db-entrypoint.sh] + volumes: + - vector_data:/var/lib/postgresql/data + - ./deploy/vector/vector-db-entrypoint.sh:/opt/thoth/vector-db-entrypoint.sh:ro + - ./deploy/vector/secret-policy.sh:/opt/thoth/secret-policy.sh:ro + healthcheck: + test: [CMD-SHELL, "pg_isready -U $$POSTGRES_USER -d $$POSTGRES_DB"] + interval: 5s + timeout: 3s + retries: 20 + start_period: 10s + restart: unless-stopped + + vector-reconcile: + image: pgvector/pgvector:0.8.5-pg16@sha256:1d533553fefe4f12e5d80c7b80622ba0c382abb5758856f52983d8789179f0fb + profiles: [local-vector] + labels: {io.thothii.smoke-owner: "${THOTH_SMOKE_OWNER:-operator}"} + environment: + PGHOST: vector-db + PGPORT: 5432 + PGDATABASE: "${THT_VECTOR_DATABASE:-thoth}" + PGUSER: "${THT_VECTOR_BOOTSTRAP_USER:-postgres}" + THT_VECTOR_BOOTSTRAP_USER: "${THT_VECTOR_BOOTSTRAP_USER:-postgres}" + THT_VECTOR_MIGRATOR_USER: "${THT_VECTOR_MIGRATOR_USER:-thoth_vector_migrator}" + THT_VECTOR_READER_USER: "${THT_VECTOR_READER_USER:-thoth_vector_reader}" + THT_VECTOR_WRITER_USER: "${THT_VECTOR_WRITER_USER:-thoth_vector_writer}" + THT_SECRETS_FILE: /run/secrets/thothii.secrets + entrypoint: [/opt/thoth/reconcile-roles.sh] + secrets: [{source: thothii_secrets, target: thothii.secrets}] + volumes: + - ./deploy/vector/reconcile-roles.sh:/opt/thoth/reconcile-roles.sh:ro + - ./deploy/vector/secret-policy.sh:/opt/thoth/secret-policy.sh:ro + depends_on: + vector-db: {condition: service_healthy} + restart: "no" + + vector-migrate: + image: thothii-core:local + profiles: [local-vector] + labels: {io.thothii.smoke-owner: "${THOTH_SMOKE_OWNER:-operator}"} + build: + context: . + dockerfile: docker/core.Dockerfile + entrypoint: [sh, -ec] + command: + - | + . /opt/thoth/secret-policy.sh + export PGPASSWORD=$$(read_bundle_secret /run/secrets/thothii.secrets THT_VECTOR_MIGRATOR_PASSWORD) + exec /opt/venv/bin/tht vector migrate --database-url "postgresql+psycopg2://${THT_VECTOR_MIGRATOR_USER:-thoth_vector_migrator}@vector-db:5432/${THT_VECTOR_DATABASE:-thoth}" --json + secrets: [{source: thothii_secrets, target: thothii.secrets}] + environment: + THT_SECRETS_FILE: /run/secrets/thothii.secrets + volumes: + - ./deploy/vector/secret-policy.sh:/opt/thoth/secret-policy.sh:ro + depends_on: + vector-reconcile: {condition: service_completed_successfully} + restart: "no" + +volumes: + vector_data: + labels: {io.thothii.smoke-owner: "${THOTH_SMOKE_OWNER:-operator}"} diff --git a/deploy/compose.local.yaml b/deploy/compose.local.yaml new file mode 100644 index 00000000..9c4ebbf9 --- /dev/null +++ b/deploy/compose.local.yaml @@ -0,0 +1,6 @@ +services: + core: + environment: + # Non-secret settings come from the root .env interpolation file. + AUTH_MODE: "${AUTH_MODE:-none}" + THT_SECRETS_FILE: /run/secrets/thothii.secrets diff --git a/deploy/compose.preprocess-local-vector.yaml b/deploy/compose.preprocess-local-vector.yaml new file mode 100644 index 00000000..8ce48d61 --- /dev/null +++ b/deploy/compose.preprocess-local-vector.yaml @@ -0,0 +1,14 @@ +services: + preprocess-evidence: + environment: + THT_SECRETS_FILE: /run/secrets/thothii.secrets + secrets: [{source: thothii_secrets, target: thothii.secrets}] + depends_on: + vector-migrate: {condition: service_completed_successfully} + + preprocess-dwh: + environment: + THT_SECRETS_FILE: /run/secrets/thothii.secrets + secrets: [{source: thothii_secrets, target: thothii.secrets}] + depends_on: + vector-migrate: {condition: service_completed_successfully} diff --git a/deploy/compose.preprocess.yaml b/deploy/compose.preprocess.yaml new file mode 100644 index 00000000..0b18e140 --- /dev/null +++ b/deploy/compose.preprocess.yaml @@ -0,0 +1,35 @@ +services: + preprocess-evidence: + image: thothii-core:local + profiles: [preprocess] + build: + context: . + dockerfile: docker/core.Dockerfile + entrypoint: [sh, -ec] + command: ["mkdir -p /data/workspaces/preprocess-evidence && exec /app/docker/core-entrypoint.sh preprocess evidence --json -c /app/harness/workspaces/preprocess-evidence.yaml"] + environment: + THT_DATA_ROOT: /data + THT_OLLAMA_URL: "${THT_OLLAMA_URL:-http://host.docker.internal:11434}" + THT_SECRETS_FILE: /run/secrets/thothii.secrets + secrets: [{source: thothii_secrets, target: thothii.secrets}] + volumes: + - thoth_data:/data + - ./deploy/workspaces:/app/harness/workspaces:ro + restart: "no" + + preprocess-dwh: + image: thothii-core:local + profiles: [preprocess] + build: + context: . + dockerfile: docker/core.Dockerfile + entrypoint: [sh, -ec] + command: ["mkdir -p /data/workspaces/preprocess-dwh && exec /app/docker/core-entrypoint.sh preprocess dwh --steps introspect --json -c /app/harness/workspaces/preprocess-dwh.yaml"] + environment: + THT_DATA_ROOT: /data + THT_SECRETS_FILE: /run/secrets/thothii.secrets + secrets: [{source: thothii_secrets, target: thothii.secrets}] + volumes: + - thoth_data:/data + - ./deploy/workspaces:/app/harness/workspaces:ro + restart: "no" diff --git a/deploy/compose.production.yaml b/deploy/compose.production.yaml new file mode 100644 index 00000000..ab419743 --- /dev/null +++ b/deploy/compose.production.yaml @@ -0,0 +1,11 @@ +services: + core: + environment: + AUTH_MODE: upstream + THOTH_PUBLIC_EXPOSURE: "true" + THT_DB_NAME: ${THT_DB_NAME:?set THT_DB_NAME} + THT_DWH_REST_URL: ${THT_DWH_REST_URL:?set THT_DWH_REST_URL} + THT_VEC_REST_URL: ${THT_VEC_REST_URL:?set THT_VEC_REST_URL} + THT_OLLAMA_URL: ${THT_OLLAMA_URL:?set THT_OLLAMA_URL} + THT_DOCS_ROOT: ${THT_DOCS_ROOT:-/data/workspaces/example/evidence-source} + THT_SECRETS_FILE: /run/secrets/thothii.secrets diff --git a/deploy/compose.psd-local.yaml.example b/deploy/compose.psd-local.yaml.example new file mode 100644 index 00000000..0325db47 --- /dev/null +++ b/deploy/compose.psd-local.yaml.example @@ -0,0 +1,11 @@ +services: + core: + environment: + THT_DOCS_ROOT: /data/workspaces/psd + extra_hosts: + - host.docker.internal:host-gateway + volumes: + - ./deploy/workspaces/psd.yaml:/app/harness/config/tht.yaml:ro + - type: bind + source: ${THT_PSD_WORKSPACE_HOST_PATH:?set THT_PSD_WORKSPACE_HOST_PATH} + target: /data/workspaces/psd diff --git a/deploy/env.example b/deploy/env.example new file mode 100644 index 00000000..1a1da078 --- /dev/null +++ b/deploy/env.example @@ -0,0 +1,27 @@ +# Deprecated compatibility template; it is not loaded by Docker Compose automatically. +# New installations must copy ../.env.example to ../.env and run +# `docker compose up --build -d` from the repository root. Keep this file only for +# staged upgrades that still invoke `--env-file deploy/env.example` explicitly. +# Never put secret values in this file. + +COMPOSE_FILE=compose.yaml +COMPOSE_PROFILES= +THT_SECRETS_FILE=deploy/secrets/thothii.secrets + +PI_PROVIDER= +PI_MODEL= +PI_THINKING= +MAX_PI_PROCESSES=4 +AUTH_MODE=none + +THT_DB_NAME= +THT_DWH_REST_URL= +THT_VEC_REST_URL= +THT_OLLAMA_URL= +THT_DOCS_ROOT=/data/workspaces/example/evidence-source + +THT_VECTOR_DATABASE=thoth +THT_VECTOR_BOOTSTRAP_USER=postgres +THT_VECTOR_MIGRATOR_USER=thoth_vector_migrator +THT_VECTOR_READER_USER=thoth_vector_reader +THT_VECTOR_WRITER_USER=thoth_vector_writer diff --git a/deploy/nginx-authenticated-proxy.conf.example b/deploy/nginx-authenticated-proxy.conf.example new file mode 100644 index 00000000..86c9b502 --- /dev/null +++ b/deploy/nginx-authenticated-proxy.conf.example @@ -0,0 +1,26 @@ +# Host nginx example. The auth service MUST authenticate every request and return a stable +# identity in X-Authenticated-User. ThothII itself remains on 127.0.0.1:8080. +server { + listen 443 ssl; + server_name thoth.example.test; + + ssl_certificate /etc/nginx/tls/fullchain.pem; + ssl_certificate_key /etc/nginx/tls/privkey.pem; + + location = /_authenticate { + internal; + proxy_pass http://authentication-gateway/verify; + proxy_pass_request_body off; + proxy_set_header Content-Length ""; + proxy_set_header X-Original-URI $request_uri; + } + + location / { + auth_request /_authenticate; + auth_request_set $authenticated_user $upstream_http_x_authenticated_user; + proxy_set_header X-Authenticated-User $authenticated_user; + proxy_set_header X-Forwarded-Proto https; + proxy_set_header Host $host; + proxy_pass http://127.0.0.1:8080; + } +} diff --git a/deploy/pi/models.json b/deploy/pi/models.json new file mode 100644 index 00000000..1b49dac8 --- /dev/null +++ b/deploy/pi/models.json @@ -0,0 +1,18 @@ +{ + "providers": { + "zai": { + "baseUrl": "https://api.z.ai/api/coding/paas/v4", + "api": "openai-completions", + "apiKey": "$ZAI_API_KEY", + "models": [ + { + "id": "glm-5.2", + "name": "GLM-5.2", + "reasoning": true, + "contextWindow": 200000, + "maxTokens": 131072 + } + ] + } + } +} diff --git a/deploy/pi/settings.json b/deploy/pi/settings.json new file mode 100644 index 00000000..bc2fe370 --- /dev/null +++ b/deploy/pi/settings.json @@ -0,0 +1,3 @@ +{ + "defaultProjectTrust": "always" +} diff --git a/deploy/secrets/README.md b/deploy/secrets/README.md new file mode 100644 index 00000000..58898e8b --- /dev/null +++ b/deploy/secrets/README.md @@ -0,0 +1,45 @@ +# Runtime secrets + +The canonical deployment secret is the single local file +`deploy/secrets/thothii.secrets`. Copy the tracked template and protect the copy: + +```sh +cp deploy/secrets/thothii.secrets.example deploy/secrets/thothii.secrets +chmod 600 deploy/secrets/thothii.secrets +``` + +The file uses strict `KEY=VALUE` lines (comments and blank lines are allowed). The supported +keys are `THT_MODEL_API_KEY`, `THT_DWH_API_KEY`, `THT_VEC_API_KEY`, +`THT_VEC_WRITE_API_KEY`, and the four `THT_VECTOR_*_PASSWORD` role passwords. Values must be +non-empty and contain no whitespace. Do not put secrets in the root `.env`, workspace YAML, +URLs, logs, or `docker compose config` output. + +Compose mounts the bundle read-only as `/run/secrets/thothii.secrets`. The host file must be a +regular non-symlink file with mode `0600` or `0400`; Docker's normal `0444` mode is accepted +only for the runtime mount beneath `/run/secrets`. The core runs as UID 10001. Verify the mount +without printing its contents: + +```sh +docker compose run --rm core sh -c 'id && test -r /run/secrets/thothii.secrets' +``` + +A private CA PEM chain is not a bundle value: PEM whitespace is rejected by the strict parser. +Keep it in the host or secret manager and add a reviewed Compose override that mounts it at +`/run/secrets/ca-chain.pem` and sets `THT_SSL_CA` (or the adapter-specific setting). The base +Compose files intentionally do not create this mount. + +## Migration from separate secret files + +Older installations used `THT_*_SECRET_FILE` variables and one file per value. Migrate by +copying each value to its bundle key, validating with `docker compose config --quiet`, and only +then deleting the old files. The old variables remain a compatibility path for staged upgrades, +but the documented and tested default is `THT_SECRETS_FILE=deploy/secrets/thothii.secrets`. + +The local-vector bootstrap rotation helper still accepts an old/new password file as its +maintenance interface. Run it only with files protected by `0600`, then copy the resulting +password into `THT_VECTOR_BOOTSTRAP_PASSWORD` in the bundle before restarting +`vector-reconcile`/the application. The helper never prints password contents. + +Hosted Pi providers must use a single provider key. Compound providers (Bedrock, Azure OpenAI +Responses, Cloudflare Workers AI/Gateway) fail closed until a provider-specific credential +adapter is implemented. diff --git a/deploy/secrets/thothii.secrets.example b/deploy/secrets/thothii.secrets.example new file mode 100644 index 00000000..c6598060 --- /dev/null +++ b/deploy/secrets/thothii.secrets.example @@ -0,0 +1,20 @@ +# Copy to deploy/secrets/thothii.secrets and chmod 600. +# Values are read as literal strings (no shell expansion or command substitution). +# Leave unused keys out of the file. + +# Hosted model provider (single-key providers only). +# THT_MODEL_API_KEY=replace-me + +# External DWH and vector adapters. +# THT_DWH_API_KEY=replace-me +# THT_VEC_API_KEY=replace-me +# THT_VEC_WRITE_API_KEY=replace-me + +# Optional local-vector roles. +# THT_VECTOR_BOOTSTRAP_PASSWORD=replace-me +# THT_VECTOR_MIGRATOR_PASSWORD=replace-me +# THT_VECTOR_READER_PASSWORD=replace-me +# THT_VECTOR_WRITER_PASSWORD=replace-me + +# Optional CA material/path understood by the configured adapter. +# THT_CA=/run/secrets/ca-chain.pem diff --git a/deploy/vector/reconcile-roles.sh b/deploy/vector/reconcile-roles.sh new file mode 100755 index 00000000..74f14afe --- /dev/null +++ b/deploy/vector/reconcile-roles.sh @@ -0,0 +1,54 @@ +#!/bin/sh +set -eu + +. /opt/thoth/secret-policy.sh + +bundle=${THT_SECRETS_FILE:-/run/secrets/thothii.secrets} +export PGPASSWORD=$(read_bundle_secret "$bundle" THT_VECTOR_BOOTSTRAP_PASSWORD) +migrator_password=$(read_bundle_secret "$bundle" THT_VECTOR_MIGRATOR_PASSWORD) +reader_password=$(read_bundle_secret "$bundle" THT_VECTOR_READER_PASSWORD) +writer_password=$(read_bundle_secret "$bundle" THT_VECTOR_WRITER_PASSWORD) + +psql --set=ON_ERROR_STOP=1 \ + --set=migrator_user="$THT_VECTOR_MIGRATOR_USER" \ + --set=migrator_password="$migrator_password" \ + --set=reader_user="$THT_VECTOR_READER_USER" \ + --set=reader_password="$reader_password" \ + --set=writer_user="$THT_VECTOR_WRITER_USER" \ + --set=writer_password="$writer_password" <<'SQL' +SELECT 'CREATE ROLE vector_reader NOLOGIN' +WHERE NOT EXISTS (SELECT FROM pg_catalog.pg_roles WHERE rolname = 'vector_reader') \gexec +SELECT 'CREATE ROLE vector_writer NOLOGIN' +WHERE NOT EXISTS (SELECT FROM pg_catalog.pg_roles WHERE rolname = 'vector_writer') \gexec +ALTER ROLE vector_reader NOLOGIN NOSUPERUSER NOCREATEDB NOCREATEROLE NOREPLICATION; +ALTER ROLE vector_writer NOLOGIN NOSUPERUSER NOCREATEDB NOCREATEROLE NOREPLICATION; + +SELECT format('CREATE ROLE %I LOGIN', :'migrator_user') +WHERE NOT EXISTS (SELECT FROM pg_catalog.pg_roles WHERE rolname = :'migrator_user') \gexec +SELECT format('CREATE ROLE %I LOGIN', :'reader_user') +WHERE NOT EXISTS (SELECT FROM pg_catalog.pg_roles WHERE rolname = :'reader_user') \gexec +SELECT format('CREATE ROLE %I LOGIN', :'writer_user') +WHERE NOT EXISTS (SELECT FROM pg_catalog.pg_roles WHERE rolname = :'writer_user') \gexec + +SELECT format( + 'ALTER ROLE %I LOGIN PASSWORD %L NOSUPERUSER NOCREATEDB NOCREATEROLE NOREPLICATION', + :'migrator_user', :'migrator_password' +) \gexec +SELECT format( + 'ALTER ROLE %I LOGIN PASSWORD %L NOSUPERUSER NOCREATEDB NOCREATEROLE NOREPLICATION', + :'reader_user', :'reader_password' +) \gexec +SELECT format( + 'ALTER ROLE %I LOGIN PASSWORD %L NOSUPERUSER NOCREATEDB NOCREATEROLE NOREPLICATION', + :'writer_user', :'writer_password' +) \gexec + +SELECT format('GRANT vector_reader TO %I', :'reader_user') \gexec +SELECT format('GRANT vector_writer TO %I', :'writer_user') \gexec +SELECT format('ALTER DATABASE %I OWNER TO %I', current_database(), :'migrator_user') \gexec +SELECT format('CREATE SCHEMA IF NOT EXISTS vectors AUTHORIZATION %I', :'migrator_user') \gexec +SELECT format('ALTER SCHEMA vectors OWNER TO %I', :'migrator_user') \gexec +REVOKE ALL ON SCHEMA vectors FROM PUBLIC; +GRANT USAGE ON SCHEMA vectors TO vector_reader, vector_writer; +CREATE EXTENSION IF NOT EXISTS vector WITH SCHEMA vectors; +SQL diff --git a/deploy/vector/rotate-bootstrap-password.py b/deploy/vector/rotate-bootstrap-password.py new file mode 100755 index 00000000..042c49d7 --- /dev/null +++ b/deploy/vector/rotate-bootstrap-password.py @@ -0,0 +1,93 @@ +#!/usr/bin/env python3 +"""Rotate the initialized PostgreSQL bootstrap role and verify before returning success.""" + +from __future__ import annotations + +import os +import sys +from pathlib import Path + +import psycopg2 +from psycopg2 import sql + + +def read_secret(path: str) -> str: + value = Path(path).read_text() + if not value or "\x00" in value or any(character.isspace() for character in value): + raise ValueError("secret must be non-empty and contain no whitespace or NUL bytes") + return value + + +def connect(password: str): + return psycopg2.connect( + host=os.environ.get("THT_VECTOR_HOST", "vector-db"), + port=int(os.environ.get("THT_VECTOR_PORT", "5432")), + dbname=os.environ.get("THT_VECTOR_DATABASE", "thoth"), + user=os.environ.get("THT_VECTOR_BOOTSTRAP_USER", "postgres"), + password=password, + connect_timeout=5, + ) + + +def alter_current_role(connection, password: str) -> None: + with connection.cursor() as cursor: + cursor.execute("SELECT current_user") + current_user = cursor.fetchone()[0] + expected = os.environ.get("THT_VECTOR_BOOTSTRAP_USER", "postgres") + if current_user != expected: + raise RuntimeError("authenticated role does not match THT_VECTOR_BOOTSTRAP_USER") + cursor.execute( + sql.SQL("ALTER ROLE {} PASSWORD {}").format( + sql.Identifier(current_user), sql.Literal(password) + ) + ) + connection.commit() + + +def main() -> int: + if len(sys.argv) != 3: + print("usage: rotate-bootstrap-password.py OLD_SECRET NEW_SECRET", file=sys.stderr) + return 2 + try: + old_password = read_secret(sys.argv[1]) + new_password = read_secret(sys.argv[2]) + if old_password == new_password: + raise ValueError("old and new bootstrap passwords must differ") + old_connection = connect(old_password) + except Exception as exc: + print(f"bootstrap rotation refused before change: {type(exc).__name__}", file=sys.stderr) + return 1 + + try: + alter_current_role(old_connection, new_password) + try: + verification = connect(new_password) + verification.close() + except Exception as verify_exc: + try: + alter_current_role(old_connection, old_password) + except Exception as restore_exc: + print( + "bootstrap rotation verification failed and password restore failed: " + f"{type(verify_exc).__name__}/{type(restore_exc).__name__}", + file=sys.stderr, + ) + return 3 + print( + f"bootstrap rotation verification failed; old password restored: " + f"{type(verify_exc).__name__}", + file=sys.stderr, + ) + return 1 + except Exception as exc: + print(f"bootstrap rotation failed: {type(exc).__name__}", file=sys.stderr) + return 1 + finally: + old_connection.close() + + print("bootstrap database password rotated and new login verified") + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/deploy/vector/secret-policy.sh b/deploy/vector/secret-policy.sh new file mode 100755 index 00000000..01de8eab --- /dev/null +++ b/deploy/vector/secret-policy.sh @@ -0,0 +1,95 @@ +#!/bin/sh + +validate_secret_file() { + secret_path=$1 + secret_name=$2 + if [ -L "$secret_path" ] || [ ! -f "$secret_path" ] || [ ! -r "$secret_path" ] || [ ! -s "$secret_path" ]; then + echo "$secret_name must be a readable, non-empty regular file" >&2 + return 2 + fi + if LC_ALL=C grep -q '[[:space:]]' "$secret_path"; then + echo "$secret_name must contain no whitespace" >&2 + return 2 + fi + mode=$(stat -c '%a' "$secret_path" 2>/dev/null || stat -f '%Lp' "$secret_path" 2>/dev/null) || return 2 + case "$secret_path:$mode" in + /run/secrets/*:444|/run/secrets/*:400|/run/secrets/*:600|*:600|*:400) ;; + *) echo "$secret_name must have mode 0600 or stricter (Docker secrets may be 0444)" >&2; return 2 ;; + esac +} + +read_secret_file() { + validate_secret_file "$1" "$2" || return + cat "$1" +} + +# Validate the bundle without printing any value. Keep this parser aligned with +# the backend loader: comments/blank lines are allowed, while syntax, allowlist, +# duplicates, empty values, file size, and line size are fail-closed. +validate_bundle() { + bundle_path=$1 + if [ -L "$bundle_path" ] || [ ! -f "$bundle_path" ] || [ ! -r "$bundle_path" ] || [ ! -s "$bundle_path" ]; then + echo "secret bundle must be a readable, non-empty regular file" >&2 + return 2 + fi + mode=$(stat -c '%a' "$bundle_path" 2>/dev/null || stat -f '%Lp' "$bundle_path" 2>/dev/null) || return 2 + case "$bundle_path:$mode" in + /run/secrets/*:444|/run/secrets/*:400|/run/secrets/*:600|*:600|*:400) ;; + *) echo "secret bundle must have mode 0600 or stricter (Docker secrets may be 0444)" >&2; return 2 ;; + esac + size=$(stat -c '%s' "$bundle_path" 2>/dev/null || stat -f '%z' "$bundle_path" 2>/dev/null) || return 2 + if [ "$size" -gt 65536 ]; then + echo "secret bundle exceeds the 64KiB limit" >&2 + return 2 + fi + awk ' + { sub(/\r$/, "", $0) } + length($0) > 16384 { exit 9 } + /^[[:space:]]*$/ || /^[[:space:]]*#/ { next } + /^[A-Z][A-Z0-9_]*=/ { + key=$0; sub(/=.*/, "", key) + val=$0; sub(/^[^=]*=/, "", val) + if (key !~ /^(THT_MODEL_API_KEY|THT_DWH_API_KEY|THT_VEC_API_KEY|THT_VEC_WRITE_API_KEY|THT_CA|THT_SSL_CA|THT_VECTOR_BOOTSTRAP_PASSWORD|THT_VECTOR_MIGRATOR_PASSWORD|THT_VECTOR_READER_PASSWORD|THT_VECTOR_WRITER_PASSWORD|PI_PROVIDER_API_KEY)$/) exit 6 + if (val == "" || ++seen[key] > 1) exit 7 + next + } + { exit 4 } + ' "$bundle_path" || { + echo "secret bundle syntax is invalid" >&2 + return 2 + } +} + +# Read one value from the deployment bundle without putting the bundle itself in +# a service environment. Values selected for credentials must contain no spaces. +read_bundle_secret() { + bundle_path=$1 + bundle_key=$2 + validate_bundle "$bundle_path" || return + case "$bundle_key" in + THT_[A-Z0-9_]*|PI_PROVIDER_API_KEY) ;; + *) echo "invalid secret bundle key" >&2; return 2 ;; + esac + value=$(awk -v wanted="$bundle_key" ' + { sub(/\r$/, "", $0) } + /^[[:space:]]*$/ || /^[[:space:]]*#/ { next } + /^[A-Z][A-Z0-9_]*=/ { + key=$0; sub(/=.*/, "", key) + val=$0; sub(/^[^=]*=/, "", val) + if (key == wanted) { + found=1; print val + } + next + } + { exit 4 } + END { if (!found) exit 5 } + ' "$bundle_path") || { + echo "$bundle_key is unavailable in secret bundle" >&2 + return 3 + } + if [ -z "$value" ] || printf '%s' "$value" | LC_ALL=C grep -q '[[:space:]]'; then + echo "$bundle_key must contain no whitespace" >&2 + return 2 + fi + printf '%s' "$value" +} diff --git a/deploy/vector/vector-db-entrypoint.sh b/deploy/vector/vector-db-entrypoint.sh new file mode 100755 index 00000000..75cd0e40 --- /dev/null +++ b/deploy/vector/vector-db-entrypoint.sh @@ -0,0 +1,9 @@ +#!/bin/sh +set -eu + +. /opt/thoth/secret-policy.sh + +bundle=${THT_SECRETS_FILE:-/run/secrets/thothii.secrets} +export POSTGRES_PASSWORD=$(read_bundle_secret "$bundle" THT_VECTOR_BOOTSTRAP_PASSWORD) +unset THT_SECRETS_FILE +exec /usr/local/bin/docker-entrypoint.sh postgres diff --git a/deploy/workspaces/example.yaml b/deploy/workspaces/example.yaml new file mode 100644 index 00000000..10c1a186 --- /dev/null +++ b/deploy/workspaces/example.yaml @@ -0,0 +1,69 @@ +language: en + +dwh: + type: thoth_rest + database: + database: ${THT_DB_NAME} + schema: datawarehouse + endpoint: + base_url: ${THT_DWH_REST_URL} + api_key: ${THT_DWH_API_KEY} + ssl_ca: ${THT_SSL_CA} + +# Relative logical roots are resolved beneath /data/workspaces/example. +roots: + artifacts: artifacts + indexes: indexes + sessions: sessions + +examples: + max_per_column: 10 + +lsh: + signature_size: 64 + n_gram: 3 + threshold: 0.5 + max_values_per_column: 1000 + +eligibility: + max_declared_len: 128 + max_avg_length: 40 + max_sampled_len: 200 + ignore_columns: [etl_last_update] + +evidence: + source_root: ${THT_DOCS_ROOT} + evidence_dir: evidence + +embeddings: + base_url: ${THT_OLLAMA_URL} + model: nomic-embed-text-v2-moe + dim: 768 + batch_size: 32 + +vectors: + type: thoth_vector_http + reader: + base_url: ${THT_VEC_REST_URL} + api_key: ${THT_VEC_API_KEY} + ssl_ca: ${THT_SSL_CA} + writer: + base_url: ${THT_VEC_REST_URL} + api_key: ${THT_VEC_WRITE_API_KEY} + ssl_ca: ${THT_SSL_CA} + +vector: + max_chunk_chars: 4000 + +search: + rrf_k: 60 + top_schema_tables: 12 + schema_chunk_pool: 150 + +execution: + allow: [cte_test, explain, preview, aggregate, export] + max_preview_rows: 10 + max_export_rows: 100000 + statement_timeout_ms: 30000 + warn_execution_ms: 5000 + max_aggregate_cells: 20 diff --git a/deploy/workspaces/local-vector.yaml b/deploy/workspaces/local-vector.yaml new file mode 100644 index 00000000..b8a46556 --- /dev/null +++ b/deploy/workspaces/local-vector.yaml @@ -0,0 +1,48 @@ +language: en + +dwh: + type: thoth_rest + database: + database: ${THT_DB_NAME} + schema: datawarehouse + endpoint: + base_url: ${THT_DWH_REST_URL} + api_key: ${THT_DWH_API_KEY} + +vectors: + type: pgvector_direct + reader: + host: vector-db + port: 5432 + database: ${THT_VECTOR_DATABASE} + schema: vectors + user: ${THT_VECTOR_READER_USER} + password_file: ${THT_VECTOR_READER_PASSWORD_FILE} + writer: + host: vector-db + port: 5432 + database: ${THT_VECTOR_DATABASE} + schema: vectors + user: ${THT_VECTOR_WRITER_USER} + password_file: ${THT_VECTOR_WRITER_PASSWORD_FILE} + +roots: + artifacts: artifacts + indexes: indexes + sessions: sessions + +evidence: + source_root: ${THT_DOCS_ROOT} + evidence_dir: evidence + +embeddings: + base_url: ${THT_OLLAMA_URL} + model: nomic-embed-text-v2-moe + dim: 768 + batch_size: 32 + +execution: + allow: [cte_test, explain, preview, aggregate, export] + max_preview_rows: 10 + max_export_rows: 100000 + statement_timeout_ms: 30000 diff --git a/deploy/workspaces/preprocess-dwh.yaml b/deploy/workspaces/preprocess-dwh.yaml new file mode 100644 index 00000000..59079e7c --- /dev/null +++ b/deploy/workspaces/preprocess-dwh.yaml @@ -0,0 +1,7 @@ +language: en +dwh: + type: postgres_direct + connection: + {host: vector-db, database: thoth, schema: vectors, user: thoth_vector_reader, + password_file: "${THT_VECTOR_READER_PASSWORD_FILE}"} +roots: {artifacts: artifacts, indexes: indexes, sessions: sessions} diff --git a/deploy/workspaces/preprocess-evidence.yaml b/deploy/workspaces/preprocess-evidence.yaml new file mode 100644 index 00000000..638f547d --- /dev/null +++ b/deploy/workspaces/preprocess-evidence.yaml @@ -0,0 +1,15 @@ +language: en +dwh: + type: postgres_direct + connection: {host: unused, database: unused, schema: public, user: unused, password: unused} +vectors: + type: pgvector_direct + reader: + {host: vector-db, database: thoth, schema: vectors, user: thoth_vector_reader, + password_file: "${THT_VECTOR_READER_PASSWORD_FILE}"} + writer: + {host: vector-db, database: thoth, schema: vectors, user: thoth_vector_writer, + password_file: "${THT_VECTOR_WRITER_PASSWORD_FILE}"} +roots: {artifacts: artifacts, indexes: indexes, sessions: sessions} +evidence: {source_root: /data/source, evidence_dir: evidence} +embeddings: {base_url: "${THT_OLLAMA_URL}", model: smoke, dim: 768, batch_size: 32} diff --git a/deploy/workspaces/psd.yaml.example b/deploy/workspaces/psd.yaml.example new file mode 100644 index 00000000..4b9e52f3 --- /dev/null +++ b/deploy/workspaces/psd.yaml.example @@ -0,0 +1,43 @@ +language: it + +dwh: + type: thoth_rest + database: + database: ${THT_DB_NAME} + schema: datawarehouse + endpoint: + base_url: ${THT_DWH_REST_URL} + api_key: ${THT_DWH_API_KEY} + ssl_ca: ${THT_SSL_CA} + +vectors: + type: thoth_vector_http + reader: + base_url: ${THT_VEC_REST_URL} + api_key: ${THT_VEC_API_KEY} + ssl_ca: ${THT_SSL_CA} + writer: + base_url: ${THT_VEC_WRITE_REST_URL} + api_key: ${THT_VEC_WRITE_API_KEY} + ssl_ca: ${THT_SSL_CA} + +roots: + artifacts: /data/workspaces/psd/runtime-v2/artifacts + indexes: /data/workspaces/psd/runtime-v2/indexes + sessions: /data/workspaces/psd/sessions + +evidence: + source_root: ${THT_DOCS_ROOT} + evidence_dir: evidence + +embeddings: + base_url: ${THT_OLLAMA_URL} + model: nomic-embed-text-v2-moe + dim: 768 + batch_size: 32 + +execution: + allow: [cte_test, explain, preview, aggregate, export] + max_preview_rows: 10 + max_export_rows: 100000 + statement_timeout_ms: 30000 diff --git a/docker/LOCKS.md b/docker/LOCKS.md new file mode 100644 index 00000000..eb85bf76 --- /dev/null +++ b/docker/LOCKS.md @@ -0,0 +1,52 @@ +# Core runtime dependency locks + +The core image consumes committed, production-only locks for Pi and the Python harness. Refresh +them from the repository root after intentionally changing the corresponding direct dependencies. + +## Pi runtime + +Keep the exact Pi version in `docker/pi-runtime/package.json`, then regenerate its npm lock: + +```sh +npm install --package-lock-only --ignore-scripts --no-audit --no-fund \ + --prefix docker/pi-runtime +``` + +The image installs this tree with `npm ci --omit=dev`; do not replace it with an unpinned global +install. + +## Python runtime + +Install [uv](https://docs.astral.sh/uv/) and compile the harness's production dependencies for +Python 3.12. `--universal` retains platform markers and hashes for a cross-platform resolution; +the `dev` extra is deliberately absent. + +```sh +uv pip compile harness/pyproject.toml docker/python-runtime/build-requirements.in \ + --universal \ + --python-version 3.12 \ + --no-emit-package tht \ + --generate-hashes \ + --custom-compile-command \ + 'uv pip compile harness/pyproject.toml docker/python-runtime/build-requirements.in --universal --python-version 3.12 --no-emit-package tht --generate-hashes --output-file docker/python-runtime/requirements.lock' \ + --output-file docker/python-runtime/requirements.lock +``` + +The small input file pins the harness's PEP 517 build backend as well; it is not derived from a +host environment. The image installs the resulting lock with pip's `--require-hashes`, then +installs the local `tht` project with `--no-deps --no-build-isolation`. This prevents both project +metadata and an isolated build environment from resolving unpinned packages. + +## Base images + +Every `FROM` uses an exact tag plus a multi-platform manifest-list digest. To update one: + +1. Choose an exact patch tag that publishes both `linux/amd64` and `linux/arm64`. +2. Inspect it with `docker buildx imagetools inspect `. +3. Replace both the human-readable tag and `@sha256:...` digest. +4. Run `./scripts/verify-container-images.sh` and the Compose smoke. +5. Review the generated inventory under `.artifacts/container-images/`. + +The digest freezes image layers, but `apt-get update` and `apk add` still consume mutable package +repositories during a no-cache rebuild. Full OS-package immutability would require Debian/Alpine +snapshot repositories and is not claimed by this deployment. diff --git a/docker/frontend-entrypoint.sh b/docker/frontend-entrypoint.sh new file mode 100644 index 00000000..15554e4a --- /dev/null +++ b/docker/frontend-entrypoint.sh @@ -0,0 +1,18 @@ +#!/bin/sh +set -eu + +backend_base_url=${BACKEND_BASE_URL-/api} +if ! /usr/local/bin/validate-backend-url "$backend_base_url"; then + echo "Invalid BACKEND_BASE_URL: use empty/root, /api, or a valid http(s) base without credentials, query, or fragment" >&2 + exit 2 +fi +runtime_config=$(jq -cn --arg backend_base_url "$backend_base_url" \ + '{backendBaseUrl: $backend_base_url}') +printf 'window.__THOTHII_CONFIG__ = %s;\n' "$runtime_config" \ + > /usr/share/nginx/html/config.js + +if [ "$#" -gt 0 ]; then + exec "$@" +fi + +exec nginx -g 'daemon off;' diff --git a/docker/nginx.conf.template b/docker/nginx.conf.template new file mode 100644 index 00000000..83226fe6 --- /dev/null +++ b/docker/nginx.conf.template @@ -0,0 +1,36 @@ +server { + listen 8080; + server_name _; + root /usr/share/nginx/html; + + location = /config.js { + add_header Cache-Control "no-store"; + try_files $uri =404; + } + + location = /health { + proxy_pass http://core:8787/health; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_cache off; + } + + location /api/ { + proxy_pass http://core:8787/; + proxy_http_version 1.1; + proxy_set_header Host $host; + proxy_set_header X-Real-IP $remote_addr; + proxy_set_header X-Forwarded-For $proxy_add_x_forwarded_for; + proxy_set_header X-Forwarded-Proto $scheme; + # Trusted only when AUTH_MODE=upstream and this frontend port is reachable solely + # from the authenticated host proxy documented in deploy/. + proxy_set_header X-Authenticated-User $http_x_authenticated_user; + proxy_buffering off; + proxy_cache off; + proxy_read_timeout 1h; + } + + location / { + try_files $uri $uri/ /index.html; + } +} diff --git a/docker/pi-runtime/package-lock.json b/docker/pi-runtime/package-lock.json new file mode 100644 index 00000000..5143b147 --- /dev/null +++ b/docker/pi-runtime/package-lock.json @@ -0,0 +1,1833 @@ +{ + "name": "thothii-pi-runtime", + "version": "1.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "thothii-pi-runtime", + "version": "1.0.0", + "dependencies": { + "@earendil-works/pi-coding-agent": "0.80.3" + } + }, + "node_modules/@earendil-works/pi-coding-agent": { + "version": "0.80.3", + "resolved": "https://registry.npmjs.org/@earendil-works/pi-coding-agent/-/pi-coding-agent-0.80.3.tgz", + "integrity": "sha512-TIggw9gCXpA+Ph7OjdTA7ka2NPwTVuPmy39KDSyUzaKq8VvHfMGR7vtRz4JB7Um/RMRblmzhu4p9tUCk6MTgGA==", + "hasShrinkwrap": true, + "license": "MIT", + "dependencies": { + "@earendil-works/pi-agent-core": "^0.80.3", + "@earendil-works/pi-ai": "^0.80.3", + "@earendil-works/pi-tui": "^0.80.3", + "@silvia-odwyer/photon-node": "0.3.4", + "chalk": "5.6.2", + "cross-spawn": "7.0.6", + "diff": "8.0.4", + "glob": "13.0.6", + "highlight.js": "10.7.3", + "hosted-git-info": "9.0.3", + "ignore": "7.0.5", + "jiti": "2.7.0", + "minimatch": "10.2.5", + "proper-lockfile": "4.1.2", + "semver": "7.8.0", + "typebox": "1.1.38", + "undici": "8.5.0", + "yaml": "2.9.0" + }, + "bin": { + "pi": "dist/cli.js" + }, + "engines": { + "node": ">=22.19.0" + }, + "optionalDependencies": { + "@mariozechner/clipboard": "0.3.9" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@anthropic-ai/sdk": { + "version": "0.91.1", + "resolved": "https://registry.npmjs.org/@anthropic-ai/sdk/-/sdk-0.91.1.tgz", + "integrity": "sha512-LAmu761tSN9r66ixvmciswUj/ZC+1Q4iAfpedTfSVLeswRwnY3n2Nb6Tsk+cLPP28aLOPWeMgIuTuCcMC6W/iw==", + "license": "MIT", + "dependencies": { + "json-schema-to-ts": "^3.1.1" + }, + "bin": { + "anthropic-ai-sdk": "bin/cli" + }, + "peerDependencies": { + "zod": "^3.25.0 || ^4.0.0" + }, + "peerDependenciesMeta": { + "zod": { + "optional": true + } + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-crypto/crc32": { + "version": "5.2.0", + "resolved": "https://registry.npmjs.org/@aws-crypto/crc32/-/crc32-5.2.0.tgz", + "integrity": "sha512-nLbCWqQNgUiwwtFsen1AdzAtvuLRsQS8rYgMuxCrdKf9kOssamGLuPwyTY9wyYblNr9+1XM8v6zoDTPPSIeANg==", + "license": "Apache-2.0", + "dependencies": { + "@aws-crypto/util": "^5.2.0", + "@aws-sdk/types": "^3.222.0", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=16.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-crypto/sha256-browser": { + "version": "5.2.0", + "resolved": "https://registry.npmjs.org/@aws-crypto/sha256-browser/-/sha256-browser-5.2.0.tgz", + "integrity": "sha512-AXfN/lGotSQwu6HNcEsIASo7kWXZ5HYWvfOmSNKDsEqC4OashTp8alTmaz+F7TC2L083SFv5RdB+qU3Vs1kZqw==", + "license": "Apache-2.0", + "dependencies": { + "@aws-crypto/sha256-js": "^5.2.0", + "@aws-crypto/supports-web-crypto": "^5.2.0", + "@aws-crypto/util": "^5.2.0", + "@aws-sdk/types": "^3.222.0", + "@aws-sdk/util-locate-window": "^3.0.0", + "@smithy/util-utf8": "^2.0.0", + "tslib": "^2.6.2" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-crypto/sha256-js": { + "version": "5.2.0", + "resolved": "https://registry.npmjs.org/@aws-crypto/sha256-js/-/sha256-js-5.2.0.tgz", + "integrity": "sha512-FFQQyu7edu4ufvIZ+OadFpHHOt+eSTBaYaki44c+akjg7qZg9oOQeLlk77F6tSYqjDAFClrHJk9tMf0HdVyOvA==", + "license": "Apache-2.0", + "dependencies": { + "@aws-crypto/util": "^5.2.0", + "@aws-sdk/types": "^3.222.0", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=16.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-crypto/supports-web-crypto": { + "version": "5.2.0", + "resolved": "https://registry.npmjs.org/@aws-crypto/supports-web-crypto/-/supports-web-crypto-5.2.0.tgz", + "integrity": "sha512-iAvUotm021kM33eCdNfwIN//F77/IADDSs58i+MDaOqFrVjZo9bAal0NK7HurRuWLLpF1iLX7gbWrjHjeo+YFg==", + "license": "Apache-2.0", + "dependencies": { + "tslib": "^2.6.2" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-crypto/util": { + "version": "5.2.0", + "resolved": "https://registry.npmjs.org/@aws-crypto/util/-/util-5.2.0.tgz", + "integrity": "sha512-4RkU9EsI6ZpBve5fseQlGNUWKMa1RLPQ1dnjnQoe07ldfIzcsGb5hC5W0Dm7u423KWzawlrpbjXBrXCEv9zazQ==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/types": "^3.222.0", + "@smithy/util-utf8": "^2.0.0", + "tslib": "^2.6.2" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/client-bedrock-runtime": { + "version": "3.1048.0", + "resolved": "https://registry.npmjs.org/@aws-sdk/client-bedrock-runtime/-/client-bedrock-runtime-3.1048.0.tgz", + "integrity": "sha512-u+NT61JZEkRFtpL0CAw1N1dwxnaLgwVXQl/zjJxTGgLyS/jTIdg2SdoEoCTHxgDyCnqa1HEi9QOoE9/pYRNpOQ==", + "license": "Apache-2.0", + "dependencies": { + "@aws-crypto/sha256-browser": "5.2.0", + "@aws-crypto/sha256-js": "5.2.0", + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/credential-provider-node": "^3.972.42", + "@aws-sdk/eventstream-handler-node": "^3.972.16", + "@aws-sdk/middleware-eventstream": "^3.972.12", + "@aws-sdk/middleware-websocket": "^3.972.19", + "@aws-sdk/token-providers": "3.1048.0", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/fetch-http-handler": "^5.4.2", + "@smithy/node-http-handler": "^4.7.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/core": { + "version": "3.974.11", + "resolved": "https://registry.npmjs.org/@aws-sdk/core/-/core-3.974.11.tgz", + "integrity": "sha512-QpnINq5FZH6EOaDEkmHdT7eUunbvD27pDNQypaWjFyYz7Zl1q3UCMQErBZxpmfGfI7MvI2TlK8KTkgNpv8b1ug==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/types": "^3.973.8", + "@aws-sdk/xml-builder": "^3.972.24", + "@aws/lambda-invoke-store": "^0.2.2", + "@smithy/core": "^3.24.2", + "@smithy/signature-v4": "^5.4.2", + "@smithy/types": "^4.14.1", + "bowser": "^2.11.0", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-env": { + "version": "3.972.37", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-env/-/credential-provider-env-3.972.37.tgz", + "integrity": "sha512-/jpPvEh6f7ntmIzf7dNxoNX6Q8vt8UpesCjbW6mFfk4V1NW6bIy9qxcQ6WbA8As5yQhsZOe+xeNd4xHX8kdY2Q==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-http": { + "version": "3.972.39", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-http/-/credential-provider-http-3.972.39.tgz", + "integrity": "sha512-pIgTpisWyWg7X1bUbzSjuUYosYTD0Ghz2M0hkSTmb3a6i3qV3uU+NYJPI/E2XSC0HcsZh5rsLPzeXrkb2DS0Cg==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/fetch-http-handler": "^5.4.2", + "@smithy/node-http-handler": "^4.7.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-ini": { + "version": "3.972.41", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-ini/-/credential-provider-ini-3.972.41.tgz", + "integrity": "sha512-u2tyjaxJJzW8UtW4SM1ZcPMDwO6y+kV+llvou+Adts0FAKyzes5jG4izQN+KX3yE8ZROpS5y1LJ//xL2iSf76w==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/credential-provider-env": "^3.972.37", + "@aws-sdk/credential-provider-http": "^3.972.39", + "@aws-sdk/credential-provider-login": "^3.972.41", + "@aws-sdk/credential-provider-process": "^3.972.37", + "@aws-sdk/credential-provider-sso": "^3.972.41", + "@aws-sdk/credential-provider-web-identity": "^3.972.41", + "@aws-sdk/nested-clients": "^3.997.9", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/credential-provider-imds": "^4.3.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-login": { + "version": "3.972.41", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-login/-/credential-provider-login-3.972.41.tgz", + "integrity": "sha512-0LBitxXiAiaE5nlFPfpNIww/8FRY/I7WIndWsc9GmNFOM7cE1wNpVNQEGEk9Outg5l8xl+3vybxFyUy4l9q/LQ==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/nested-clients": "^3.997.9", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-node": { + "version": "3.972.42", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-node/-/credential-provider-node-3.972.42.tgz", + "integrity": "sha512-D4oon2zbqqsWOJUM99Gm3/ZyJ0IJvTXVN3PyloGb3kQEyI36fjCZheZj422lAgTWWd6TSHgiImLt3RIaLdv3dQ==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/credential-provider-env": "^3.972.37", + "@aws-sdk/credential-provider-http": "^3.972.39", + "@aws-sdk/credential-provider-ini": "^3.972.41", + "@aws-sdk/credential-provider-process": "^3.972.37", + "@aws-sdk/credential-provider-sso": "^3.972.41", + "@aws-sdk/credential-provider-web-identity": "^3.972.41", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/credential-provider-imds": "^4.3.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-process": { + "version": "3.972.37", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-process/-/credential-provider-process-3.972.37.tgz", + "integrity": "sha512-7nVaHBUaWIddASYfVaA9O4D5ZVjewU3sCol9WqZPGfW0nR+0WqE0xHZnD/U2L33PlOB8KNXGKZ6wOES/QijKzg==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-sso": { + "version": "3.972.41", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-sso/-/credential-provider-sso-3.972.41.tgz", + "integrity": "sha512-IOWAWEHe5LkjSKkkUUX9ciV6Y1scHTsnfEkdt5yyC4Slrc7AGbkLPrpntjqh18ksJAMOaVhoBsO8p2WyTcY2wQ==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/nested-clients": "^3.997.9", + "@aws-sdk/token-providers": "3.1048.0", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/credential-provider-web-identity": { + "version": "3.972.41", + "resolved": "https://registry.npmjs.org/@aws-sdk/credential-provider-web-identity/-/credential-provider-web-identity-3.972.41.tgz", + "integrity": "sha512-mbACk9Yypa8nm4iGZLs0PofOXEcTDOUw6wDnsPXNDNSd2WNXs1tSo+6nc/fh0jLYdfVZThhBL98PHW4aXFsG5A==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/nested-clients": "^3.997.9", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/eventstream-handler-node": { + "version": "3.972.16", + "resolved": "https://registry.npmjs.org/@aws-sdk/eventstream-handler-node/-/eventstream-handler-node-3.972.16.tgz", + "integrity": "sha512-yedpPgKftqjU5SlPFHfqWpOw6xSCRieWRG1euWOlXn4WJxt2VX92VprCa2PpSOXjVCAeK6dTjW9eJRXVig9yGA==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/middleware-eventstream": { + "version": "3.972.12", + "resolved": "https://registry.npmjs.org/@aws-sdk/middleware-eventstream/-/middleware-eventstream-3.972.12.tgz", + "integrity": "sha512-tHTHHCHNrq6XklQvlzHBDJG4Iuhh7NVPRdtmvP+nHFA+5sxPlIDzlAHHgfoYHGvT3NXP1yVP/L5c3opUn6T3Qg==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/middleware-websocket": { + "version": "3.972.19", + "resolved": "https://registry.npmjs.org/@aws-sdk/middleware-websocket/-/middleware-websocket-3.972.19.tgz", + "integrity": "sha512-mkEhOGYozqKQkbFaVrjwr0faiwwZza1v5/jSY6Tucm3bD+uKTazIUH/4Yo6aMnQD2ua2W9cMP6s8mvwTcjtqHw==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/fetch-http-handler": "^5.4.2", + "@smithy/signature-v4": "^5.4.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">= 14.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/nested-clients": { + "version": "3.997.9", + "resolved": "https://registry.npmjs.org/@aws-sdk/nested-clients/-/nested-clients-3.997.9.tgz", + "integrity": "sha512-jPR3rnmRI4hWYyzfmTGBr7NblMp8QYYeflHXba1H6+7CGrWVqWKQzaXFQ4qbExqPRsXN3T3L3JxFhr6aouXUGQ==", + "license": "Apache-2.0", + "dependencies": { + "@aws-crypto/sha256-browser": "5.2.0", + "@aws-crypto/sha256-js": "5.2.0", + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/signature-v4-multi-region": "^3.996.27", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/fetch-http-handler": "^5.4.2", + "@smithy/node-http-handler": "^4.7.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/signature-v4-multi-region": { + "version": "3.996.27", + "resolved": "https://registry.npmjs.org/@aws-sdk/signature-v4-multi-region/-/signature-v4-multi-region-3.996.27.tgz", + "integrity": "sha512-0Phbz4t6HI3D3skxvG2uI+VWU034/nSIw1T8d+FPzzQG9EQTrw94o9mOKO2Gv3n3Oc8P7JD7RAUxkoneLWv5Eg==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/signature-v4": "^5.4.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/token-providers": { + "version": "3.1048.0", + "resolved": "https://registry.npmjs.org/@aws-sdk/token-providers/-/token-providers-3.1048.0.tgz", + "integrity": "sha512-k0y/GcuesuSfWyUM0WamrGyeZmltRYaPbHO82UDA6mZ/doB+FOHKutikPAtSXMn/hDz970cF+iRuuiYO9VEbAA==", + "license": "Apache-2.0", + "dependencies": { + "@aws-sdk/core": "^3.974.11", + "@aws-sdk/nested-clients": "^3.997.9", + "@aws-sdk/types": "^3.973.8", + "@smithy/core": "^3.24.2", + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/types": { + "version": "3.973.8", + "resolved": "https://registry.npmjs.org/@aws-sdk/types/-/types-3.973.8.tgz", + "integrity": "sha512-gjlAdtHMbtR9X5iIhVUvbVcy55KnznpC6bkDUWW9z915bi0ckdUr5cjf16Kp6xq0bP5HBD2xzgbL9F9Quv5vUw==", + "license": "Apache-2.0", + "dependencies": { + "@smithy/types": "^4.14.1", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/util-locate-window": { + "version": "3.965.5", + "resolved": "https://registry.npmjs.org/@aws-sdk/util-locate-window/-/util-locate-window-3.965.5.tgz", + "integrity": "sha512-WhlJNNINQB+9qtLtZJcpQdgZw3SCDCpXdUJP7cToGwHbCWCnRckGlc6Bx/OhWwIYFNAn+FIydY8SZ0QmVu3xTQ==", + "license": "Apache-2.0", + "dependencies": { + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws-sdk/xml-builder": { + "version": "3.972.24", + "resolved": "https://registry.npmjs.org/@aws-sdk/xml-builder/-/xml-builder-3.972.24.tgz", + "integrity": "sha512-V8z5YcDPfsvzrBlj0xR1vhRtocblhYbqdreCJB/voGd4Sr5zjNAeWxexbnqVtskTJe0vFb5KMqbSL++ePl+zRw==", + "license": "Apache-2.0", + "dependencies": { + "@nodable/entities": "2.1.0", + "@smithy/types": "^4.14.1", + "fast-xml-parser": "5.7.3", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=20.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@aws/lambda-invoke-store": { + "version": "0.2.4", + "resolved": "https://registry.npmjs.org/@aws/lambda-invoke-store/-/lambda-invoke-store-0.2.4.tgz", + "integrity": "sha512-iY8yvjE0y651BixKNPgmv1WrQc+GZ142sb0z4gYnChDDY2YqI4P/jsSopBWrKfAt7LOJAkOXt7rC/hms+WclQQ==", + "license": "Apache-2.0", + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@babel/runtime": { + "version": "7.29.2", + "resolved": "https://registry.npmjs.org/@babel/runtime/-/runtime-7.29.2.tgz", + "integrity": "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g==", + "license": "MIT", + "engines": { + "node": ">=6.9.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@earendil-works/pi-agent-core": { + "version": "0.80.3", + "resolved": "https://registry.npmjs.org/@earendil-works/pi-agent-core/-/pi-agent-core-0.80.3.tgz", + "license": "MIT", + "dependencies": { + "@earendil-works/pi-ai": "^0.80.3", + "ignore": "7.0.5", + "typebox": "1.1.38", + "yaml": "2.9.0" + }, + "engines": { + "node": ">=22.19.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@earendil-works/pi-ai": { + "version": "0.80.3", + "resolved": "https://registry.npmjs.org/@earendil-works/pi-ai/-/pi-ai-0.80.3.tgz", + "license": "MIT", + "dependencies": { + "@anthropic-ai/sdk": "0.91.1", + "@aws-sdk/client-bedrock-runtime": "3.1048.0", + "@google/genai": "1.52.0", + "@mistralai/mistralai": "2.2.6", + "@opentelemetry/api": "1.9.0", + "@smithy/node-http-handler": "4.7.3", + "http-proxy-agent": "7.0.2", + "https-proxy-agent": "7.0.6", + "openai": "6.26.0", + "partial-json": "0.1.7", + "typebox": "1.1.38" + }, + "bin": { + "pi-ai": "./dist/cli.js" + }, + "engines": { + "node": ">=22.19.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@earendil-works/pi-tui": { + "version": "0.80.3", + "resolved": "https://registry.npmjs.org/@earendil-works/pi-tui/-/pi-tui-0.80.3.tgz", + "license": "MIT", + "dependencies": { + "get-east-asian-width": "1.6.0", + "marked": "18.0.5" + }, + "engines": { + "node": ">=22.19.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@google/genai": { + "version": "1.52.0", + "resolved": "https://registry.npmjs.org/@google/genai/-/genai-1.52.0.tgz", + "integrity": "sha512-gwSvbpiN/17O9TbsqSsE/OzZcpv5Fo4RQjdngGgogtuB9RsyJ8ZHhX5KjHj1bp5N9snN2eK8LDGXSaWW2hof8Q==", + "hasInstallScript": true, + "license": "Apache-2.0", + "dependencies": { + "google-auth-library": "^10.3.0", + "p-retry": "^4.6.2", + "protobufjs": "^7.5.4", + "ws": "^8.18.0" + }, + "engines": { + "node": ">=20.0.0" + }, + "peerDependencies": { + "@modelcontextprotocol/sdk": "^1.25.2" + }, + "peerDependenciesMeta": { + "@modelcontextprotocol/sdk": { + "optional": true + } + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard/-/clipboard-0.3.9.tgz", + "integrity": "sha512-ABnA53mdfkGZwOFUdZNv2S0CWGO/EIuPj8Vv9xmBFmSYg/qFc7ihO6q5FcQjvoE67kZpWkEc4AhD6B/os04yuA==", + "license": "MIT", + "optional": true, + "engines": { + "node": ">= 10" + }, + "optionalDependencies": { + "@mariozechner/clipboard-darwin-arm64": "0.3.9", + "@mariozechner/clipboard-darwin-universal": "0.3.9", + "@mariozechner/clipboard-darwin-x64": "0.3.9", + "@mariozechner/clipboard-linux-arm64-gnu": "0.3.9", + "@mariozechner/clipboard-linux-arm64-musl": "0.3.9", + "@mariozechner/clipboard-linux-riscv64-gnu": "0.3.9", + "@mariozechner/clipboard-linux-x64-gnu": "0.3.9", + "@mariozechner/clipboard-linux-x64-musl": "0.3.9", + "@mariozechner/clipboard-win32-arm64-msvc": "0.3.9", + "@mariozechner/clipboard-win32-x64-msvc": "0.3.9" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-darwin-arm64": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-darwin-arm64/-/clipboard-darwin-arm64-0.3.9.tgz", + "integrity": "sha512-BfgV7vCEWZwJwZJw03r6bP5+tf0iI/ANuQYCxi9RNn7FrWB3yzGuMKCrNLRl6V761vXRdL8+OqZ0wd4TqlsNOQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-darwin-universal": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-darwin-universal/-/clipboard-darwin-universal-0.3.9.tgz", + "integrity": "sha512-BGGR4iA9Z2shAjI65eI5xtyb3LYNlDW9X3gxKxDbqtbnREohsrqznov6zpKoIrsRWpzlYVEdKphS7ksJ0/ndSQ==", + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-darwin-x64": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-darwin-x64/-/clipboard-darwin-x64-0.3.9.tgz", + "integrity": "sha512-4kURmCbS6nt8uYhtmWpUcJWyPHfmAr5dTpXD1nO3pIfa+TSQ9DbrGOYCKH+aEFW47XhQ4Vp8ZTszie+wfFvDKg==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "darwin" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-linux-arm64-gnu": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-linux-arm64-gnu/-/clipboard-linux-arm64-gnu-0.3.9.tgz", + "integrity": "sha512-g59OkUGP2DDfCOIKypHeYgv2M55u/cKvXa5dSxFbEJ34XvIQMdcVmpKCkGUro3ZgefXiGVdwguvTMQGpHWzIXw==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-linux-arm64-musl": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-linux-arm64-musl/-/clipboard-linux-arm64-musl-0.3.9.tgz", + "integrity": "sha512-AGuJdgKsmJdm4Pych7kv3sqe591ERRaAHW3xjLooiFzn8J+PxUyof++7YZrB5Y5tpnTO+K18Og3taj2NpluCRQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-linux-riscv64-gnu": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-linux-riscv64-gnu/-/clipboard-linux-riscv64-gnu-0.3.9.tgz", + "integrity": "sha512-DXBEAiuMpk7dhS1a9NzNxVAFi1vaKoPu7rQNgY8LIDLGrK3lnIp3nT10DUum+PKVJoJppIP+NAA8IZe4DMNDPw==", + "cpu": [ + "riscv64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-linux-x64-gnu": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-linux-x64-gnu/-/clipboard-linux-x64-gnu-0.3.9.tgz", + "integrity": "sha512-WORrMLd6EpElEME7JRKfSaY34nW1P5LbdgK5YNCS1ncG2LqmITsSMEJ8nh2mpvxb3TxqbOOKgY7k9eMJYlW9Mw==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-linux-x64-musl": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-linux-x64-musl/-/clipboard-linux-x64-musl-0.3.9.tgz", + "integrity": "sha512-/DHn+1DrfL6oRaPPWXaOKvonFFrni666fxd+zFqiQEfvBH0tsHVWjq9iqBk0oDp0qaPA72lIMy5BptxISBEhZQ==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "linux" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-win32-arm64-msvc": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-win32-arm64-msvc/-/clipboard-win32-arm64-msvc-0.3.9.tgz", + "integrity": "sha512-O5FHD3ErkMwMhNzAfu3ggy0ug4z7btZuoQgwwxlzPrwV2bxlD6WDpqBY4NCgICAgZdDKdp+loUEKVAVt8aYnhQ==", + "cpu": [ + "arm64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mariozechner/clipboard-win32-x64-msvc": { + "version": "0.3.9", + "resolved": "https://registry.npmjs.org/@mariozechner/clipboard-win32-x64-msvc/-/clipboard-win32-x64-msvc-0.3.9.tgz", + "integrity": "sha512-ihQC3EufqEY81vhXBgVBtK4prL+wc62zJsSvxrgz7K1hsdt6OObz6v9p3Rn1OG3GJksTTKMJF0u/guMISHPhSA==", + "cpu": [ + "x64" + ], + "license": "MIT", + "optional": true, + "os": [ + "win32" + ], + "engines": { + "node": ">= 10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@mistralai/mistralai": { + "version": "2.2.6", + "resolved": "https://registry.npmjs.org/@mistralai/mistralai/-/mistralai-2.2.6.tgz", + "integrity": "sha512-W8pX7zHxjJvMIpw8JMxeJEleapXX0Q9NPszdNzqkM3MIEoIGPObdodujj+WHteXEvGfaP/AMwlNyRfEzSY6dQQ==", + "license": "Apache-2.0", + "dependencies": { + "@opentelemetry/semantic-conventions": "^1.40.0", + "ws": "^8.18.0", + "zod": "^3.25.0 || ^4.0.0", + "zod-to-json-schema": "^3.25.0" + }, + "peerDependencies": { + "@opentelemetry/api": "^1.9.0" + }, + "peerDependenciesMeta": { + "@opentelemetry/api": { + "optional": true + } + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@nodable/entities": { + "version": "2.1.0", + "resolved": "https://registry.npmjs.org/@nodable/entities/-/entities-2.1.0.tgz", + "integrity": "sha512-nyT7T3nbMyBI/lvr6L5TyWbFJAI9FTgVRakNoBqCD+PmID8DzFrrNdLLtHMwMszOtqZa8PAOV24ZqDnQrhQINA==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/nodable" + } + ], + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@opentelemetry/api": { + "version": "1.9.0", + "resolved": "https://registry.npmjs.org/@opentelemetry/api/-/api-1.9.0.tgz", + "integrity": "sha512-3giAOQvZiH5F9bMlMiv8+GSPMeqg0dbaeo58/0SlA9sxSqZhnUtxzX9/2FzyhS9sWQf5S0GJE0AKBrFqjpeYcg==", + "license": "Apache-2.0", + "engines": { + "node": ">=8.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@opentelemetry/semantic-conventions": { + "version": "1.41.1", + "resolved": "https://registry.npmjs.org/@opentelemetry/semantic-conventions/-/semantic-conventions-1.41.1.tgz", + "integrity": "sha512-/UhIkaZgPutTFmQ7RnIJGgDXZmtEJ7Dvi86xNTFWcnRxVRNk/aotsqDJYeEvDP+FSMB2SdW+pQzNMcWP0rwuNA==", + "license": "Apache-2.0", + "engines": { + "node": ">=14" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/aspromise": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/@protobufjs/aspromise/-/aspromise-1.1.2.tgz", + "integrity": "sha512-j+gKExEuLmKwvz3OgROXtrJ2UG2x8Ch2YZUxahh+s1F2HZ+wAceUNLkvy6zKCPVRkU++ZWQrdxsUeQXmcg4uoQ==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/base64": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/@protobufjs/base64/-/base64-1.1.2.tgz", + "integrity": "sha512-AZkcAA5vnN/v4PDqKyMR5lx7hZttPDgClv83E//FMNhR2TMcLUhfRUBHCmSl0oi9zMgDDqRUJkSxO3wm85+XLg==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/codegen": { + "version": "2.0.5", + "resolved": "https://registry.npmjs.org/@protobufjs/codegen/-/codegen-2.0.5.tgz", + "integrity": "sha512-zgXFLzW3Ap33e6d0Wlj4MGIm6Ce8O89n/apUaGNB/jx+hw+ruWEp7EwGUshdLKVRCxZW12fp9r40E1mQrf/34g==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/eventemitter": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/eventemitter/-/eventemitter-1.1.1.tgz", + "integrity": "sha512-vW1GmwMZNnL+gMRaovlh9yZX74kc+TTU3FObkkurpMaRtBfLP3ldjS9KQWlwZgraRE0+dheEEoAxdzcJQ8eXZg==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/fetch": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/fetch/-/fetch-1.1.1.tgz", + "integrity": "sha512-GpptLrs57adMSuHi3VNj0mAF8dwh36LMaYF6XyJ6JMWlVsc+t42tm1HSEDmOs3A8fC9yyeisgLhsTVQokOZ0zw==", + "license": "BSD-3-Clause", + "dependencies": { + "@protobufjs/aspromise": "^1.1.1" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/float": { + "version": "1.0.2", + "resolved": "https://registry.npmjs.org/@protobufjs/float/-/float-1.0.2.tgz", + "integrity": "sha512-Ddb+kVXlXst9d+R9PfTIxh1EdNkgoRe5tOX6t01f1lYWOvJnSPDBlG241QLzcyPdoNTsblLUdujGSE4RzrTZGQ==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/path": { + "version": "1.1.2", + "resolved": "https://registry.npmjs.org/@protobufjs/path/-/path-1.1.2.tgz", + "integrity": "sha512-6JOcJ5Tm08dOHAbdR3GrvP+yUUfkjG5ePsHYczMFLq3ZmMkAD98cDgcT2iA1lJ9NVwFd4tH/iSSoe44YWkltEA==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/pool": { + "version": "1.1.0", + "resolved": "https://registry.npmjs.org/@protobufjs/pool/-/pool-1.1.0.tgz", + "integrity": "sha512-0kELaGSIDBKvcgS4zkjz1PeddatrjYcmMWOlAuAPwAeccUrPHdUqo/J6LiymHHEiJT5NrF1UVwxY14f+fy4WQw==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@protobufjs/utf8": { + "version": "1.1.1", + "resolved": "https://registry.npmjs.org/@protobufjs/utf8/-/utf8-1.1.1.tgz", + "integrity": "sha512-oOAWABowe8EAbMyWKM0tYDKi8Yaox52D+HWZhAIJqQXbqe0xI/GV7FhLWqlEKreMkfDjshR5FKgi3mnle0h6Eg==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@silvia-odwyer/photon-node": { + "version": "0.3.4", + "resolved": "https://registry.npmjs.org/@silvia-odwyer/photon-node/-/photon-node-0.3.4.tgz", + "integrity": "sha512-bnly4BKB3KDTFxrUIcgCLbaeVVS8lrAkri1pEzskpmxu9MdfGQTy8b8EgcD83ywD3RPMsIulY8xJH5Awa+t9fA==", + "license": "Apache-2.0" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/core": { + "version": "3.24.3", + "resolved": "https://registry.npmjs.org/@smithy/core/-/core-3.24.3.tgz", + "integrity": "sha512-Ep/7tPamGY8mgESE3LyLKtxJyy6U52WWAqr/3wial47Sj4u3PiIF73AOGI27UyLy9duTkhZbgzodOfLV4TduZg==", + "license": "Apache-2.0", + "dependencies": { + "@aws-crypto/crc32": "5.2.0", + "@smithy/types": "^4.14.2", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/credential-provider-imds": { + "version": "4.3.3", + "resolved": "https://registry.npmjs.org/@smithy/credential-provider-imds/-/credential-provider-imds-4.3.3.tgz", + "integrity": "sha512-I2Bti0DKFo2IJyN28ijCsx51BAumEYR4/1yZ1FXyBygy9MqbnMqCev4JPth/MbpRfBSRAX35hITSnAdJRo1u5w==", + "license": "Apache-2.0", + "dependencies": { + "@smithy/core": "^3.24.3", + "@smithy/types": "^4.14.2", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/fetch-http-handler": { + "version": "5.4.3", + "resolved": "https://registry.npmjs.org/@smithy/fetch-http-handler/-/fetch-http-handler-5.4.3.tgz", + "integrity": "sha512-F+DRf8IJazRJgYog2A/yJK7eYVc0rqTlRzO+5ZxjJd4WkZoKz0IJRncf7G6t1pdVT3kryJcwuTFhN1c5m6N47A==", + "license": "Apache-2.0", + "dependencies": { + "@smithy/core": "^3.24.3", + "@smithy/types": "^4.14.2", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/is-array-buffer": { + "version": "2.2.0", + "resolved": "https://registry.npmjs.org/@smithy/is-array-buffer/-/is-array-buffer-2.2.0.tgz", + "integrity": "sha512-GGP3O9QFD24uGeAXYUjwSTXARoqpZykHadOmA8G5vfJPK0/DC67qa//0qvqrJzL1xc8WQWX7/yc7fwudjPHPhA==", + "license": "Apache-2.0", + "dependencies": { + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=14.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/node-http-handler": { + "version": "4.7.3", + "resolved": "https://registry.npmjs.org/@smithy/node-http-handler/-/node-http-handler-4.7.3.tgz", + "integrity": "sha512-/jPhevcTFPMVl6KNjbaI47iOg1zxC7IsnX4PQDGVZKMFceOXtB8IEYaB7a9VvkP/3oC60WzTeKocvSI7vLT0vA==", + "license": "Apache-2.0", + "dependencies": { + "@smithy/core": "^3.24.3", + "@smithy/types": "^4.14.2", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/signature-v4": { + "version": "5.4.3", + "resolved": "https://registry.npmjs.org/@smithy/signature-v4/-/signature-v4-5.4.3.tgz", + "integrity": "sha512-53+75QuPl6DL+ct6vVEB51FDO5oulXr20TPV46VvJZg76lIlXNWfxi8j+G2V/t0I2qxCBOa3vX/8bmjrpFVo9g==", + "license": "Apache-2.0", + "dependencies": { + "@smithy/core": "^3.24.3", + "@smithy/types": "^4.14.2", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/types": { + "version": "4.14.2", + "resolved": "https://registry.npmjs.org/@smithy/types/-/types-4.14.2.tgz", + "integrity": "sha512-P+otAxbV4CqBybp7EkcJCrig63yE2E7PuNVOmilVMRcx/O+QDzGULTrKsq4DV13gSfak9ObPrWaHl/9bL5YcWw==", + "license": "Apache-2.0", + "dependencies": { + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=18.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/util-buffer-from": { + "version": "2.2.0", + "resolved": "https://registry.npmjs.org/@smithy/util-buffer-from/-/util-buffer-from-2.2.0.tgz", + "integrity": "sha512-IJdWBbTcMQ6DA0gdNhh/BwrLkDR+ADW5Kr1aZmd4k3DIF6ezMV4R2NIAmT08wQJ3yUK82thHWmC/TnK/wpMMIA==", + "license": "Apache-2.0", + "dependencies": { + "@smithy/is-array-buffer": "^2.2.0", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=14.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@smithy/util-utf8": { + "version": "2.3.0", + "resolved": "https://registry.npmjs.org/@smithy/util-utf8/-/util-utf8-2.3.0.tgz", + "integrity": "sha512-R8Rdn8Hy72KKcebgLiv8jQcQkXoLMOGGv5uI1/k0l+snqkOzQ1R0ChUBCxWMlBsFMekWjq0wRudIweFs7sKT5A==", + "license": "Apache-2.0", + "dependencies": { + "@smithy/util-buffer-from": "^2.2.0", + "tslib": "^2.6.2" + }, + "engines": { + "node": ">=14.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/@types/node": { + "version": "22.19.19", + "resolved": "https://registry.npmjs.org/@types/node/-/node-22.19.19.tgz", + "integrity": "sha512-dyh/xO2Fh5bYrfWaaqGrRQQGkNdmYw6AmaAUvYeUMNTWQtvb796ikLdmTchRmOlOiIJ1TDXfWgVx1QkUlQ6Hew==", + "license": "MIT", + "dependencies": { + "undici-types": "~6.21.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/agent-base": { + "version": "7.1.4", + "resolved": "https://registry.npmjs.org/agent-base/-/agent-base-7.1.4.tgz", + "integrity": "sha512-MnA+YT8fwfJPgBx3m60MNqakm30XOkyIoH1y6huTQvC0PwZG7ki8NacLBcrPbNoo8vEZy7Jpuk7+jMO+CUovTQ==", + "license": "MIT", + "engines": { + "node": ">= 14" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/balanced-match": { + "version": "4.0.4", + "resolved": "https://registry.npmjs.org/balanced-match/-/balanced-match-4.0.4.tgz", + "integrity": "sha512-BLrgEcRTwX2o6gGxGOCNyMvGSp35YofuYzw9h1IMTRmKqttAZZVU67bdb9Pr2vUHA8+j3i2tJfjO6C6+4myGTA==", + "license": "MIT", + "engines": { + "node": "18 || 20 || >=22" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/base64-js": { + "version": "1.5.1", + "resolved": "https://registry.npmjs.org/base64-js/-/base64-js-1.5.1.tgz", + "integrity": "sha512-AKpaYlHn8t4SVbOHCy+b5+KKgvR4vrsD8vbvrbiQJps7fKDTkjkDry6ji0rUJjC0kzbNePLwzxq8iypo41qeWA==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/bignumber.js": { + "version": "9.3.1", + "resolved": "https://registry.npmjs.org/bignumber.js/-/bignumber.js-9.3.1.tgz", + "integrity": "sha512-Ko0uX15oIUS7wJ3Rb30Fs6SkVbLmPBAKdlm7q9+ak9bbIeFf0MwuBsQV6z7+X768/cHsfg+WlysDWJcmthjsjQ==", + "license": "MIT", + "engines": { + "node": "*" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/bowser": { + "version": "2.14.1", + "resolved": "https://registry.npmjs.org/bowser/-/bowser-2.14.1.tgz", + "integrity": "sha512-tzPjzCxygAKWFOJP011oxFHs57HzIhOEracIgAePE4pqB3LikALKnSzUyU4MGs9/iCEUuHlAJTjTc5M+u7YEGg==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/brace-expansion": { + "version": "5.0.6", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.6.tgz", + "integrity": "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g==", + "license": "MIT", + "dependencies": { + "balanced-match": "^4.0.2" + }, + "engines": { + "node": "18 || 20 || >=22" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/buffer-equal-constant-time": { + "version": "1.0.1", + "resolved": "https://registry.npmjs.org/buffer-equal-constant-time/-/buffer-equal-constant-time-1.0.1.tgz", + "integrity": "sha512-zRpUiDwd/xk6ADqPMATG8vc9VPrkck7T07OIx0gnjmJAnHnTVXNQG3vfvWNuiZIkwu9KrKdA1iJKfsfTVxE6NA==", + "license": "BSD-3-Clause" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/chalk": { + "version": "5.6.2", + "resolved": "https://registry.npmjs.org/chalk/-/chalk-5.6.2.tgz", + "integrity": "sha512-7NzBL0rN6fMUW+f7A6Io4h40qQlG+xGmtMxfbnH/K7TAtt8JQWVQK+6g0UXKMeVJoyV5EkkNsErQ8pVD3bLHbA==", + "license": "MIT", + "engines": { + "node": "^12.17.0 || ^14.13 || >=16.0.0" + }, + "funding": { + "url": "https://github.com/chalk/chalk?sponsor=1" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/cross-spawn": { + "version": "7.0.6", + "resolved": "https://registry.npmjs.org/cross-spawn/-/cross-spawn-7.0.6.tgz", + "integrity": "sha512-uV2QOWP2nWzsy2aMp8aRibhi9dlzF5Hgh5SHaB9OiTGEyDTiJJyx0uy51QXdyWbtAHNua4XJzUKca3OzKUd3vA==", + "license": "MIT", + "dependencies": { + "path-key": "^3.1.0", + "shebang-command": "^2.0.0", + "which": "^2.0.1" + }, + "engines": { + "node": ">= 8" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/data-uri-to-buffer": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/data-uri-to-buffer/-/data-uri-to-buffer-4.0.1.tgz", + "integrity": "sha512-0R9ikRb668HB7QDxT1vkpuUBtqc53YyAwMwGeUFKRojY/NWKvdZ+9UYtRfGmhqNbRkTSVpMbmyhXipFFv2cb/A==", + "license": "MIT", + "engines": { + "node": ">= 12" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/debug": { + "version": "4.4.3", + "resolved": "https://registry.npmjs.org/debug/-/debug-4.4.3.tgz", + "integrity": "sha512-RGwwWnwQvkVfavKVt22FGLw+xYSdzARwm0ru6DhTVA3umU5hZc28V3kO4stgYryrTlLpuvgI9GiijltAjNbcqA==", + "license": "MIT", + "dependencies": { + "ms": "^2.1.3" + }, + "engines": { + "node": ">=6.0" + }, + "peerDependenciesMeta": { + "supports-color": { + "optional": true + } + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/diff": { + "version": "8.0.4", + "resolved": "https://registry.npmjs.org/diff/-/diff-8.0.4.tgz", + "integrity": "sha512-DPi0FmjiSU5EvQV0++GFDOJ9ASQUVFh5kD+OzOnYdi7n3Wpm9hWWGfB/O2blfHcMVTL5WkQXSnRiK9makhrcnw==", + "license": "BSD-3-Clause", + "engines": { + "node": ">=0.3.1" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/ecdsa-sig-formatter": { + "version": "1.0.11", + "resolved": "https://registry.npmjs.org/ecdsa-sig-formatter/-/ecdsa-sig-formatter-1.0.11.tgz", + "integrity": "sha512-nagl3RYrbNv6kQkeJIpt6NJZy8twLB/2vtz6yN9Z4vRKHN4/QZJIEbqohALSgwKdnksuY3k5Addp5lg8sVoVcQ==", + "license": "Apache-2.0", + "dependencies": { + "safe-buffer": "^5.0.1" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/extend": { + "version": "3.0.2", + "resolved": "https://registry.npmjs.org/extend/-/extend-3.0.2.tgz", + "integrity": "sha512-fjquC59cD7CyW6urNXK0FBufkZcoiGG80wTuPujX590cB5Ttln20E2UB4S/WARVqhXffZl2LNgS+gQdPIIim/g==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/fast-xml-builder": { + "version": "1.2.0", + "resolved": "https://registry.npmjs.org/fast-xml-builder/-/fast-xml-builder-1.2.0.tgz", + "integrity": "sha512-00aAWieqff+ZJhsXA4g1g7M8k+7AYoMUUHF+/zFb5U6Uv/P0Vl4QZo84/IcufzYalLuEj9928bXN9PbbFzMF0Q==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "path-expression-matcher": "^1.5.0", + "xml-naming": "^0.1.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/fast-xml-parser": { + "version": "5.7.3", + "resolved": "https://registry.npmjs.org/fast-xml-parser/-/fast-xml-parser-5.7.3.tgz", + "integrity": "sha512-C0AaNuC+mscy6vrAQKAc/rMq+zAPHodfHGZu4sGVehvAQt/JLG1O5zEcYcXSY5zSqr4YVgxsB+pHXTq0i7eDlg==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "dependencies": { + "@nodable/entities": "^2.1.0", + "fast-xml-builder": "^1.1.7", + "path-expression-matcher": "^1.5.0", + "strnum": "^2.2.3" + }, + "bin": { + "fxparser": "src/cli/cli.js" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/fetch-blob": { + "version": "3.2.0", + "resolved": "https://registry.npmjs.org/fetch-blob/-/fetch-blob-3.2.0.tgz", + "integrity": "sha512-7yAQpD2UMJzLi1Dqv7qFYnPbaPx7ZfFK6PiIxQ4PfkGPyNyl2Ugx+a/umUonmKqjhM4DnfbMvdX6otXq83soQQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/jimmywarting" + }, + { + "type": "paypal", + "url": "https://paypal.me/jimmywarting" + } + ], + "license": "MIT", + "dependencies": { + "node-domexception": "^1.0.0", + "web-streams-polyfill": "^3.0.3" + }, + "engines": { + "node": "^12.20 || >= 14.13" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/formdata-polyfill": { + "version": "4.0.10", + "resolved": "https://registry.npmjs.org/formdata-polyfill/-/formdata-polyfill-4.0.10.tgz", + "integrity": "sha512-buewHzMvYL29jdeQTVILecSaZKnt/RJWjoZCF5OW60Z67/GmSLBkOFM7qh1PI3zFNtJbaZL5eQu1vLfazOwj4g==", + "license": "MIT", + "dependencies": { + "fetch-blob": "^3.1.2" + }, + "engines": { + "node": ">=12.20.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/gaxios": { + "version": "7.1.4", + "resolved": "https://registry.npmjs.org/gaxios/-/gaxios-7.1.4.tgz", + "integrity": "sha512-bTIgTsM2bWn3XklZISBTQX7ZSddGW+IO3bMdGaemHZ3tbqExMENHLx6kKZ/KlejgrMtj8q7wBItt51yegqalrA==", + "license": "Apache-2.0", + "dependencies": { + "extend": "^3.0.2", + "https-proxy-agent": "^7.0.1", + "node-fetch": "^3.3.2" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/gcp-metadata": { + "version": "8.1.2", + "resolved": "https://registry.npmjs.org/gcp-metadata/-/gcp-metadata-8.1.2.tgz", + "integrity": "sha512-zV/5HKTfCeKWnxG0Dmrw51hEWFGfcF2xiXqcA3+J90WDuP0SvoiSO5ORvcBsifmx/FoIjgQN3oNOGaQ5PhLFkg==", + "license": "Apache-2.0", + "dependencies": { + "gaxios": "^7.0.0", + "google-logging-utils": "^1.0.0", + "json-bigint": "^1.0.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/get-east-asian-width": { + "version": "1.6.0", + "resolved": "https://registry.npmjs.org/get-east-asian-width/-/get-east-asian-width-1.6.0.tgz", + "integrity": "sha512-QRbvDIbx6YklUe6RxeTeleMR0yv3cYH6PsPZHcnVn7xv7zO1BHN8r0XETu8n6Ye3Q+ahtSarc3WgtNWmehIBfA==", + "license": "MIT", + "engines": { + "node": ">=18" + }, + "funding": { + "url": "https://github.com/sponsors/sindresorhus" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/glob": { + "version": "13.0.6", + "resolved": "https://registry.npmjs.org/glob/-/glob-13.0.6.tgz", + "integrity": "sha512-Wjlyrolmm8uDpm/ogGyXZXb1Z+Ca2B8NbJwqBVg0axK9GbBeoS7yGV6vjXnYdGm6X53iehEuxxbyiKp8QmN4Vw==", + "license": "BlueOak-1.0.0", + "dependencies": { + "minimatch": "^10.2.2", + "minipass": "^7.1.3", + "path-scurry": "^2.0.2" + }, + "engines": { + "node": "18 || 20 || >=22" + }, + "funding": { + "url": "https://github.com/sponsors/isaacs" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/google-auth-library": { + "version": "10.6.2", + "resolved": "https://registry.npmjs.org/google-auth-library/-/google-auth-library-10.6.2.tgz", + "integrity": "sha512-e27Z6EThmVNNvtYASwQxose/G57rkRuaRbQyxM2bvYLLX/GqWZ5chWq2EBoUchJbCc57eC9ArzO5wMsEmWftCw==", + "license": "Apache-2.0", + "dependencies": { + "base64-js": "^1.3.0", + "ecdsa-sig-formatter": "^1.0.11", + "gaxios": "^7.1.4", + "gcp-metadata": "8.1.2", + "google-logging-utils": "1.1.3", + "jws": "^4.0.0" + }, + "engines": { + "node": ">=18" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/google-logging-utils": { + "version": "1.1.3", + "resolved": "https://registry.npmjs.org/google-logging-utils/-/google-logging-utils-1.1.3.tgz", + "integrity": "sha512-eAmLkjDjAFCVXg7A1unxHsLf961m6y17QFqXqAXGj/gVkKFrEICfStRfwUlGNfeCEjNRa32JEWOUTlYXPyyKvA==", + "license": "Apache-2.0", + "engines": { + "node": ">=14" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/graceful-fs": { + "version": "4.2.11", + "resolved": "https://registry.npmjs.org/graceful-fs/-/graceful-fs-4.2.11.tgz", + "integrity": "sha512-RbJ5/jmFcNNCcDV5o9eTnBLJ/HszWV0P73bc+Ff4nS/rJj+YaS6IGyiOL0VoBYX+l1Wrl3k63h/KrH+nhJ0XvQ==", + "license": "ISC" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/highlight.js": { + "version": "10.7.3", + "resolved": "https://registry.npmjs.org/highlight.js/-/highlight.js-10.7.3.tgz", + "integrity": "sha512-tzcUFauisWKNHaRkN4Wjl/ZA07gENAjFl3J/c480dprkGTg5EQstgaNFqBfUqCq54kZRIEcreTsAgF/m2quD7A==", + "license": "BSD-3-Clause", + "engines": { + "node": "*" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/hosted-git-info": { + "version": "9.0.3", + "resolved": "https://registry.npmjs.org/hosted-git-info/-/hosted-git-info-9.0.3.tgz", + "integrity": "sha512-Hc+ghLoSt6QaYZUv0WBiIvmMDZuZZ7oaDvdH8MbfOO4lOsxdXLEvuC6ePoGs9H1X9oCLyq6+NVN0MKqD+ydxyg==", + "license": "ISC", + "dependencies": { + "lru-cache": "^11.1.0" + }, + "engines": { + "node": "^20.17.0 || >=22.9.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/http-proxy-agent": { + "version": "7.0.2", + "resolved": "https://registry.npmjs.org/http-proxy-agent/-/http-proxy-agent-7.0.2.tgz", + "integrity": "sha512-T1gkAiYYDWYx3V5Bmyu7HcfcvL7mUrTWiM6yOfa3PIphViJ/gFPbvidQ+veqSOHci/PxBcDabeUNCzpOODJZig==", + "license": "MIT", + "dependencies": { + "agent-base": "^7.1.0", + "debug": "^4.3.4" + }, + "engines": { + "node": ">= 14" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/https-proxy-agent": { + "version": "7.0.6", + "resolved": "https://registry.npmjs.org/https-proxy-agent/-/https-proxy-agent-7.0.6.tgz", + "integrity": "sha512-vK9P5/iUfdl95AI+JVyUuIcVtd4ofvtrOr3HNtM2yxC9bnMbEdp3x01OhQNnjb8IJYi38VlTE3mBXwcfvywuSw==", + "license": "MIT", + "dependencies": { + "agent-base": "^7.1.2", + "debug": "4" + }, + "engines": { + "node": ">= 14" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/ignore": { + "version": "7.0.5", + "resolved": "https://registry.npmjs.org/ignore/-/ignore-7.0.5.tgz", + "integrity": "sha512-Hs59xBNfUIunMFgWAbGX5cq6893IbWg4KnrjbYwX3tx0ztorVgTDA6B2sxf8ejHJ4wz8BqGUMYlnzNBer5NvGg==", + "license": "MIT", + "engines": { + "node": ">= 4" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/isexe": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/isexe/-/isexe-2.0.0.tgz", + "integrity": "sha512-RHxMLp9lnKHGHRng9QFhRCMbYAcVpn69smSGcq3f36xjgVVWThj4qqLbTLlq7Ssj8B+fIQ1EuCEGI2lKsyQeIw==", + "license": "ISC" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/jiti": { + "version": "2.7.0", + "resolved": "https://registry.npmjs.org/jiti/-/jiti-2.7.0.tgz", + "integrity": "sha512-AC/7JofJvZGrrneWNaEnJeOLUx+JlGt7tNa0wZiRPT4MY1wmfKjt2+6O2p2uz2+skll8OZZmJMNqeke7kKbNgQ==", + "license": "MIT", + "bin": { + "jiti": "lib/jiti-cli.mjs" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/json-bigint": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/json-bigint/-/json-bigint-1.0.0.tgz", + "integrity": "sha512-SiPv/8VpZuWbvLSMtTDU8hEfrZWg/mH/nV/b4o0CYbSxu1UIQPLdwKOCIyLQX+VIPO5vrLX3i8qtqFyhdPSUSQ==", + "license": "MIT", + "dependencies": { + "bignumber.js": "^9.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/json-schema-to-ts": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/json-schema-to-ts/-/json-schema-to-ts-3.1.1.tgz", + "integrity": "sha512-+DWg8jCJG2TEnpy7kOm/7/AxaYoaRbjVB4LFZLySZlWn8exGs3A4OLJR966cVvU26N7X9TWxl+Jsw7dzAqKT6g==", + "license": "MIT", + "dependencies": { + "@babel/runtime": "^7.18.3", + "ts-algebra": "^2.0.0" + }, + "engines": { + "node": ">=16" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/jwa": { + "version": "2.0.1", + "resolved": "https://registry.npmjs.org/jwa/-/jwa-2.0.1.tgz", + "integrity": "sha512-hRF04fqJIP8Abbkq5NKGN0Bbr3JxlQ+qhZufXVr0DvujKy93ZCbXZMHDL4EOtodSbCWxOqR8MS1tXA5hwqCXDg==", + "license": "MIT", + "dependencies": { + "buffer-equal-constant-time": "^1.0.1", + "ecdsa-sig-formatter": "1.0.11", + "safe-buffer": "^5.0.1" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/jws": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/jws/-/jws-4.0.1.tgz", + "integrity": "sha512-EKI/M/yqPncGUUh44xz0PxSidXFr/+r0pA70+gIYhjv+et7yxM+s29Y+VGDkovRofQem0fs7Uvf4+YmAdyRduA==", + "license": "MIT", + "dependencies": { + "jwa": "^2.0.1", + "safe-buffer": "^5.0.1" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/long": { + "version": "5.3.2", + "resolved": "https://registry.npmjs.org/long/-/long-5.3.2.tgz", + "integrity": "sha512-mNAgZ1GmyNhD7AuqnTG3/VQ26o760+ZYBPKjPvugO8+nLbYfX6TVpJPseBvopbdY+qpZ/lKUnmEc1LeZYS3QAA==", + "license": "Apache-2.0" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/lru-cache": { + "version": "11.4.0", + "resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-11.4.0.tgz", + "integrity": "sha512-W+R+kFL4HgVxONq2bhXPi3bGpzGe/yEhVOp233qw9wCRtgncJ15P3bC+e4zZMu4Cq7d+WAJjXGW0uUkifhcatA==", + "license": "BlueOak-1.0.0", + "engines": { + "node": "20 || >=22" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/marked": { + "version": "18.0.5", + "resolved": "https://registry.npmjs.org/marked/-/marked-18.0.5.tgz", + "integrity": "sha512-S6GcvALHg6K4ohtu4E7x0a1AqhAjp6cV8KhLSyN9qVapnzJkusVBxZRcIU9AeYsbe6P1hKDusSbEOzGyyuce6w==", + "license": "MIT", + "bin": { + "marked": "bin/marked.js" + }, + "engines": { + "node": ">= 20" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/minimatch": { + "version": "10.2.5", + "resolved": "https://registry.npmjs.org/minimatch/-/minimatch-10.2.5.tgz", + "integrity": "sha512-MULkVLfKGYDFYejP07QOurDLLQpcjk7Fw+7jXS2R2czRQzR56yHRveU5NDJEOviH+hETZKSkIk5c+T23GjFUMg==", + "license": "BlueOak-1.0.0", + "dependencies": { + "brace-expansion": "^5.0.5" + }, + "engines": { + "node": "18 || 20 || >=22" + }, + "funding": { + "url": "https://github.com/sponsors/isaacs" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/minipass": { + "version": "7.1.3", + "resolved": "https://registry.npmjs.org/minipass/-/minipass-7.1.3.tgz", + "integrity": "sha512-tEBHqDnIoM/1rXME1zgka9g6Q2lcoCkxHLuc7ODJ5BxbP5d4c2Z5cGgtXAku59200Cx7diuHTOYfSBD8n6mm8A==", + "license": "BlueOak-1.0.0", + "engines": { + "node": ">=16 || 14 >=14.17" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/ms": { + "version": "2.1.3", + "resolved": "https://registry.npmjs.org/ms/-/ms-2.1.3.tgz", + "integrity": "sha512-6FlzubTLZG3J2a/NVCAleEhjzq5oxgHyaCU9yYXvcLsvoVaHJq/s5xXI6/XXP6tz7R9xAOtHnSO/tXtF3WRTlA==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/node-domexception": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/node-domexception/-/node-domexception-1.0.0.tgz", + "integrity": "sha512-/jKZoMpw0F8GRwl4/eLROPA3cfcXtLApP0QzLmUT/HuPCZWyB7IY9ZrMeKw2O/nFIqPQB3PVM9aYm0F312AXDQ==", + "deprecated": "Use your platform's native DOMException instead", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/jimmywarting" + }, + { + "type": "github", + "url": "https://paypal.me/jimmywarting" + } + ], + "license": "MIT", + "engines": { + "node": ">=10.5.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/node-fetch": { + "version": "3.3.2", + "resolved": "https://registry.npmjs.org/node-fetch/-/node-fetch-3.3.2.tgz", + "integrity": "sha512-dRB78srN/l6gqWulah9SrxeYnxeddIG30+GOqK/9OlLVyLg3HPnr6SqOWTWOXKRwC2eGYCkZ59NNuSgvSrpgOA==", + "license": "MIT", + "dependencies": { + "data-uri-to-buffer": "^4.0.0", + "fetch-blob": "^3.1.4", + "formdata-polyfill": "^4.0.10" + }, + "engines": { + "node": "^12.20.0 || ^14.13.1 || >=16.0.0" + }, + "funding": { + "type": "opencollective", + "url": "https://opencollective.com/node-fetch" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/openai": { + "version": "6.26.0", + "resolved": "https://registry.npmjs.org/openai/-/openai-6.26.0.tgz", + "integrity": "sha512-zd23dbWTjiJ6sSAX6s0HrCZi41JwTA1bQVs0wLQPZ2/5o2gxOJA5wh7yOAUgwYybfhDXyhwlpeQf7Mlgx8EOCA==", + "license": "Apache-2.0", + "bin": { + "openai": "bin/cli" + }, + "peerDependencies": { + "ws": "^8.18.0", + "zod": "^3.25 || ^4.0" + }, + "peerDependenciesMeta": { + "ws": { + "optional": true + }, + "zod": { + "optional": true + } + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/p-retry": { + "version": "4.6.2", + "resolved": "https://registry.npmjs.org/p-retry/-/p-retry-4.6.2.tgz", + "integrity": "sha512-312Id396EbJdvRONlngUx0NydfrIQ5lsYu0znKVUzVvArzEIt08V1qhtyESbGVd1FGX7UKtiFp5uwKZdM8wIuQ==", + "license": "MIT", + "dependencies": { + "@types/retry": "0.12.0", + "retry": "^0.13.1" + }, + "engines": { + "node": ">=8" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/p-retry/node_modules/@types/retry": { + "version": "0.12.0", + "resolved": "https://registry.npmjs.org/@types/retry/-/retry-0.12.0.tgz", + "integrity": "sha512-wWKOClTTiizcZhXnPY4wikVAwmdYHp8q6DmC+EJUzAMsycb7HB32Kh9RN4+0gExjmPmZSAQjgURXIGATPegAvA==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/partial-json": { + "version": "0.1.7", + "resolved": "https://registry.npmjs.org/partial-json/-/partial-json-0.1.7.tgz", + "integrity": "sha512-Njv/59hHaokb/hRUjce3Hdv12wd60MtM9Z5Olmn+nehe0QDAsRtRbJPvJ0Z91TusF0SuZRIvnM+S4l6EIP8leA==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/path-expression-matcher": { + "version": "1.5.0", + "resolved": "https://registry.npmjs.org/path-expression-matcher/-/path-expression-matcher-1.5.0.tgz", + "integrity": "sha512-cbrerZV+6rvdQrrD+iGMcZFEiiSrbv9Tfdkvnusy6y0x0GKBXREFg/Y65GhIfm0tnLntThhzCnfKwp1WRjeCyQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "engines": { + "node": ">=14.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/path-key": { + "version": "3.1.1", + "resolved": "https://registry.npmjs.org/path-key/-/path-key-3.1.1.tgz", + "integrity": "sha512-ojmeN0qd+y0jszEtoY48r0Peq5dwMEkIlCOu6Q5f41lfkswXuKtYrhgoTpLnyIcHm24Uhqx+5Tqm2InSwLhE6Q==", + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/path-scurry": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/path-scurry/-/path-scurry-2.0.2.tgz", + "integrity": "sha512-3O/iVVsJAPsOnpwWIeD+d6z/7PmqApyQePUtCndjatj/9I5LylHvt5qluFaBT3I5h3r1ejfR056c+FCv+NnNXg==", + "license": "BlueOak-1.0.0", + "dependencies": { + "lru-cache": "^11.0.0", + "minipass": "^7.1.2" + }, + "engines": { + "node": "18 || 20 || >=22" + }, + "funding": { + "url": "https://github.com/sponsors/isaacs" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/proper-lockfile": { + "version": "4.1.2", + "resolved": "https://registry.npmjs.org/proper-lockfile/-/proper-lockfile-4.1.2.tgz", + "integrity": "sha512-TjNPblN4BwAWMXU8s9AEz4JmQxnD1NNL7bNOY/AKUzyamc379FWASUhc/K1pL2noVb+XmZKLL68cjzLsiOAMaA==", + "license": "MIT", + "dependencies": { + "graceful-fs": "^4.2.4", + "retry": "^0.12.0", + "signal-exit": "^3.0.2" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/proper-lockfile/node_modules/retry": { + "version": "0.12.0", + "resolved": "https://registry.npmjs.org/retry/-/retry-0.12.0.tgz", + "integrity": "sha512-9LkiTwjUh6rT555DtE9rTX+BKByPfrMzEAtnlEtdEwr3Nkffwiihqe2bWADg+OQRjt9gl6ICdmB/ZFDCGAtSow==", + "license": "MIT", + "engines": { + "node": ">= 4" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/protobufjs": { + "version": "7.6.4", + "resolved": "https://registry.npmjs.org/protobufjs/-/protobufjs-7.6.4.tgz", + "integrity": "sha512-RJJPTTpvFfHcWLkIa2JFWK4XvtSzS0yEWDmunqHXli1h3JlkbcQZXDZdcWxv+JK3Xsl5/UFDPZ0iGm7DAengYw==", + "hasInstallScript": true, + "license": "BSD-3-Clause", + "dependencies": { + "@protobufjs/aspromise": "^1.1.2", + "@protobufjs/base64": "^1.1.2", + "@protobufjs/codegen": "^2.0.5", + "@protobufjs/eventemitter": "^1.1.1", + "@protobufjs/fetch": "^1.1.1", + "@protobufjs/float": "^1.0.2", + "@protobufjs/path": "^1.1.2", + "@protobufjs/pool": "^1.1.0", + "@protobufjs/utf8": "^1.1.1", + "@types/node": ">=13.7.0", + "long": "^5.3.2" + }, + "engines": { + "node": ">=12.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/retry": { + "version": "0.13.1", + "resolved": "https://registry.npmjs.org/retry/-/retry-0.13.1.tgz", + "integrity": "sha512-XQBQ3I8W1Cge0Seh+6gjj03LbmRFWuoszgK9ooCpwYIrhhoO80pfq4cUkU5DkknwfOfFteRwlZ56PYOGYyFWdg==", + "license": "MIT", + "engines": { + "node": ">= 4" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/safe-buffer": { + "version": "5.2.1", + "resolved": "https://registry.npmjs.org/safe-buffer/-/safe-buffer-5.2.1.tgz", + "integrity": "sha512-rp3So07KcdmmKbGvgaNxQSJr7bGVSVk5S9Eq1F+ppbRo70+YeaDxkw5Dd8NPN+GD6bjnYm2VuPuCXmpuYvmCXQ==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/feross" + }, + { + "type": "patreon", + "url": "https://www.patreon.com/feross" + }, + { + "type": "consulting", + "url": "https://feross.org/support" + } + ], + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/semver": { + "version": "7.8.0", + "resolved": "https://registry.npmjs.org/semver/-/semver-7.8.0.tgz", + "integrity": "sha512-AcM7dV/5ul4EekoQ29Agm5vri8JNqRyj39o0qpX6vDF2GZrtutZl5RwgD1XnZjiTAfncsJhMI48QQH3sN87YNA==", + "license": "ISC", + "bin": { + "semver": "bin/semver.js" + }, + "engines": { + "node": ">=10" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/shebang-command": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/shebang-command/-/shebang-command-2.0.0.tgz", + "integrity": "sha512-kHxr2zZpYtdmrN1qDjrrX/Z1rR1kG8Dx+gkpK1G4eXmvXswmcE1hTWBWYUzlraYw1/yZp6YuDY77YtvbN0dmDA==", + "license": "MIT", + "dependencies": { + "shebang-regex": "^3.0.0" + }, + "engines": { + "node": ">=8" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/shebang-regex": { + "version": "3.0.0", + "resolved": "https://registry.npmjs.org/shebang-regex/-/shebang-regex-3.0.0.tgz", + "integrity": "sha512-7++dFhtcx3353uBaq8DDR4NuxBetBzC7ZQOhmTQInHEd6bSrXdiEyzCvG07Z44UYdLShWUyXt5M/yhz8ekcb1A==", + "license": "MIT", + "engines": { + "node": ">=8" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/signal-exit": { + "version": "3.0.7", + "resolved": "https://registry.npmjs.org/signal-exit/-/signal-exit-3.0.7.tgz", + "integrity": "sha512-wnD2ZE+l+SPC/uoS0vXeE9L1+0wuaMqKlfz9AMUo38JsyLSBWSFcHR1Rri62LZc12vLr1gb3jl7iwQhgwpAbGQ==", + "license": "ISC" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/strnum": { + "version": "2.3.0", + "resolved": "https://registry.npmjs.org/strnum/-/strnum-2.3.0.tgz", + "integrity": "sha512-ums3KNd42PGyx5xaoVTO1mjU1bH3NpY4vsrVlnv9PNGqQj8wd7rJ6nEypLrJ7z5vxK5RP0yMLo6J/Gsm62DI5Q==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/ts-algebra": { + "version": "2.0.0", + "resolved": "https://registry.npmjs.org/ts-algebra/-/ts-algebra-2.0.0.tgz", + "integrity": "sha512-FPAhNPFMrkwz76P7cdjdmiShwMynZYN6SgOujD1urY4oNm80Ou9oMdmbR45LotcKOXoy7wSmHkRFE6Mxbrhefw==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/tslib": { + "version": "2.8.1", + "resolved": "https://registry.npmjs.org/tslib/-/tslib-2.8.1.tgz", + "integrity": "sha512-oJFu94HQb+KVduSUQL7wnpmqnfmLsOA/nAh6b6EH0wCEoK0/mPeXU6c3wKDV83MkOuHPRHtSXKKU99IBazS/2w==", + "license": "0BSD" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/typebox": { + "version": "1.1.38", + "resolved": "https://registry.npmjs.org/typebox/-/typebox-1.1.38.tgz", + "integrity": "sha512-pZ0aQPmMmXoUvSbeuWf/Hzsc+avNw/Zd6VeE8CFgkVGWyuHPJvqeJJDeJqLve+K70LvjYIoleGcoJHPT17cWoA==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/undici": { + "version": "8.5.0", + "resolved": "https://registry.npmjs.org/undici/-/undici-8.5.0.tgz", + "integrity": "sha512-xamtWoB1EshgjpmlXd7GGm2VfdDtw1+rD8uhry8pSNW3If6S8E0m2T2+orSKeZXEn/aPJMviCpDBA65WJt8zhg==", + "license": "MIT", + "engines": { + "node": ">=22.19.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/undici-types": { + "version": "6.21.0", + "resolved": "https://registry.npmjs.org/undici-types/-/undici-types-6.21.0.tgz", + "integrity": "sha512-iwDZqg0QAGrg9Rav5H4n0M64c3mkR59cJ6wQp+7C4nI0gsmExaedaYLNO44eT4AtBBwjbTiGPMlt2Md0T9H9JQ==", + "license": "MIT" + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/web-streams-polyfill": { + "version": "3.3.3", + "resolved": "https://registry.npmjs.org/web-streams-polyfill/-/web-streams-polyfill-3.3.3.tgz", + "integrity": "sha512-d2JWLCivmZYTSIoge9MsgFCZrt571BikcWGYkjC1khllbTeDlGqZ2D8vD8E/lJa8WGWbb7Plm8/XJYV7IJHZZw==", + "license": "MIT", + "engines": { + "node": ">= 8" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/which": { + "version": "2.0.2", + "resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz", + "integrity": "sha512-BLI3Tl1TW3Pvl70l3yq3Y64i+awpwXqsGBYWkkqMtnbXgrMD+yj7rhW0kuEDxzJaYXGjEW5ogapKNMEKNMjibA==", + "license": "ISC", + "dependencies": { + "isexe": "^2.0.0" + }, + "bin": { + "node-which": "bin/node-which" + }, + "engines": { + "node": ">= 8" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/ws": { + "version": "8.21.0", + "resolved": "https://registry.npmjs.org/ws/-/ws-8.21.0.tgz", + "integrity": "sha512-Vsp28b7DRcimFQvrqu2Wek3z1iYxDCWqHYB8Qsnk/S4RfaCQzPGPyBNuVjJV3cd6UiKtUtp6sNM77gWvzcCH+g==", + "license": "MIT", + "engines": { + "node": ">=10.0.0" + }, + "peerDependencies": { + "bufferutil": "^4.0.1", + "utf-8-validate": ">=5.0.2" + }, + "peerDependenciesMeta": { + "bufferutil": { + "optional": true + }, + "utf-8-validate": { + "optional": true + } + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/xml-naming": { + "version": "0.1.0", + "resolved": "https://registry.npmjs.org/xml-naming/-/xml-naming-0.1.0.tgz", + "integrity": "sha512-k8KO9hrMyNk6tUWqUfkTEZbezRRpONVOzUTnc97VnCvyj6Tf9lyUR9EDAIeiVLv56jsMcoXEwjW8Kv5yPY52lw==", + "funding": [ + { + "type": "github", + "url": "https://github.com/sponsors/NaturalIntelligence" + } + ], + "license": "MIT", + "engines": { + "node": ">=16.0.0" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/yaml": { + "version": "2.9.0", + "resolved": "https://registry.npmjs.org/yaml/-/yaml-2.9.0.tgz", + "integrity": "sha512-2AvhNX3mb8zd6Zy7INTtSpl1F15HW6Wnqj0srWlkKLcpYl/gMIMJiyuGq2KeI2YFxUPjdlB+3Lc10seMLtL4cA==", + "license": "ISC", + "bin": { + "yaml": "bin.mjs" + }, + "engines": { + "node": ">= 14.6" + }, + "funding": { + "url": "https://github.com/sponsors/eemeli" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/zod": { + "version": "3.25.76", + "resolved": "https://registry.npmjs.org/zod/-/zod-3.25.76.tgz", + "integrity": "sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==", + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/colinhacks" + } + }, + "node_modules/@earendil-works/pi-coding-agent/node_modules/zod-to-json-schema": { + "version": "3.25.2", + "resolved": "https://registry.npmjs.org/zod-to-json-schema/-/zod-to-json-schema-3.25.2.tgz", + "integrity": "sha512-O/PgfnpT1xKSDeQYSCfRI5Gy3hPf91mKVDuYLUHZJMiDFptvP41MSnWofm8dnCm0256ZNfZIM7DSzuSMAFnjHA==", + "license": "ISC", + "peerDependencies": { + "zod": "^3.25.28 || ^4" + } + } + } +} diff --git a/docker/pi-runtime/package.json b/docker/pi-runtime/package.json new file mode 100644 index 00000000..7bd4812e --- /dev/null +++ b/docker/pi-runtime/package.json @@ -0,0 +1,9 @@ +{ + "name": "thothii-pi-runtime", + "version": "1.0.0", + "private": true, + "description": "Locked Pi runtime dependency for the ThothII core image", + "dependencies": { + "@earendil-works/pi-coding-agent": "0.80.3" + } +} diff --git a/docker/python-runtime/build-requirements.in b/docker/python-runtime/build-requirements.in new file mode 100644 index 00000000..1d611690 --- /dev/null +++ b/docker/python-runtime/build-requirements.in @@ -0,0 +1,2 @@ +# PEP 517 backend used to install the local harness without network-isolated resolution. +setuptools==80.9.0 diff --git a/docker/python-runtime/requirements.lock b/docker/python-runtime/requirements.lock new file mode 100644 index 00000000..215c2757 --- /dev/null +++ b/docker/python-runtime/requirements.lock @@ -0,0 +1,942 @@ +# This file was autogenerated by uv via the following command: +# uv pip compile harness/pyproject.toml docker/python-runtime/build-requirements.in --extra s3 --universal --python-version 3.12 --no-emit-package tht --generate-hashes --output-file docker/python-runtime/requirements.lock +annotated-doc==0.0.4 \ + --hash=sha256:571ac1dc6991c450b25a9c2d84a3705e2ae7a53467b5d111c24fa8baabbed320 \ + --hash=sha256:fbcda96e87e9c92ad167c2e53839e57503ecfda18804ea28102353485033faa4 + # via typer +annotated-types==0.7.0 \ + --hash=sha256:1f02e8b43a8fbbc3f3e0d4f0f4bfc8131bcb4eebe8849b8e5c773f3a1c582a53 \ + --hash=sha256:aff07c09a53a08bc8cfccb9c85b05f1aa9a2a6f23728d790723543408344ce89 + # via pydantic +boto3==1.43.46 \ + --hash=sha256:66c0d943b049a46a492ec4ec2ebe73c930b1842c7137bee83aad6d93e95d4d96 \ + --hash=sha256:69453e2c1bcb9fd9806527ab99950cacfc2826cb0dce9a3a0414d19270c06c3c + # via tht (harness/pyproject.toml) +botocore==1.43.46 \ + --hash=sha256:59f2e1ac3cdc66d191cae91c0804bc41847ce817dc8147cf43eaada8f76a5533 \ + --hash=sha256:cb673891e623ae6e6a1bf24d94ef169504f3eb02584adb5d5bee2f6aae819b60 + # via + # boto3 + # s3transfer +certifi==2026.6.17 \ + --hash=sha256:024c88eeec92ca068db80f02b8b07c9cef7b9fe261d1d535abfd5abd6f6af432 \ + --hash=sha256:2227dcbaafe0d2f59279d1762ddddc37783ed4354594f194ffc31d20f41fc3db + # via requests +charset-normalizer==3.4.9 \ + --hash=sha256:0327fcd59a935777d83410750c50600ee9571af2846f71ce40f25b13da1ef380 \ + --hash=sha256:03d07803992c6c7bbc976327f34b18b6160327fc81cb82c9d504720ac0be3b62 \ + --hash=sha256:04ce310cb89c15df659582aee80a0603788732a5e017d5bd5c81158106ce249c \ + --hash=sha256:0d861473f743244d349b50f850d10eb87aeb22bbdcc8e64f79273c94af5a8226 \ + --hash=sha256:0e94703ec9684807f20cfb5eed95c70f67f2a8f21ad620146d7b5a13677b93e5 \ + --hash=sha256:0fa1aec2d32bcc03c8fa0f6f1712caad1adc38509f31142112e5c9daf5b9c833 \ + --hash=sha256:16b65ea0f2465b6fb52aa22de5eca612aa964ddfec00a912e26f4656cbef890b \ + --hash=sha256:16d10d789dd9bcca1173c95af82c58433122564b7bc39385124be735a35cbe99 \ + --hash=sha256:19ac87f93086ce37b86e098888555c4b4bc48102279bae3350098c0ed664b501 \ + --hash=sha256:1d22856ffbe153a602df38e4a5464f0b748a54002e0d69ac6d2ad0a197cc99ec \ + --hash=sha256:21e764fd1e70b6a3e205a0e46f3051701f98a8cb3fad66eeb80e48bb502f8698 \ + --hash=sha256:231ddcbb35e2ff8973e1365db41fe0572662893b99a05deb183b68ad4c0c8bd4 \ + --hash=sha256:253a4a220747e8b5faf57ec320c4f5efb0cef05f647420bf267143ec15dba10a \ + --hash=sha256:280081916dc341820640489a66e4696049401ef1cf6dd672f672e70ad915aca3 \ + --hash=sha256:2a441ea71902098ffe78c5abe6c494f44160b4af614ed16c3d9a3b1d17fd8ee2 \ + --hash=sha256:304b13570067b2547562e308af560b3963857b1fa90bd6afd978130130fe2d6a \ + --hash=sha256:32286a2c8d167e897177b673176c1e3e00d4057caf5d2b64eef9a3666b03018e \ + --hash=sha256:33bdcc2a32c0a0e861f60841a512c8acc658c87c2ac59d89e3a46dacf7d866e4 \ + --hash=sha256:375b83ed0aecfce76c16d198fbc21f3b11b337d68662bea0a995046682a11419 \ + --hash=sha256:3c09a49d6cde137258beb3d551994a2927fd35ad5cf96aed573f61bbd67c5f84 \ + --hash=sha256:3d92613ec25e43b05f042302531ec0f00b8445190e43325880cbd6ab7c2581da \ + --hash=sha256:40a126142a56b2dfc0aacbad1de8310cbf60da7656db0e6b16eebd48e3e93519 \ + --hash=sha256:416c229f77e5ea25b3dfd4b582f8d73d7e43c22320302b9ab128a2d3a0b38efe \ + --hash=sha256:432786d3561e69aeeae6c7e8648964ce0ad05736120135601f87ac26b9c83381 \ + --hash=sha256:43b9e366a31fdd1c87d0eb08f579b4a82b723ea54338f040d6b4e518a026ea29 \ + --hash=sha256:440eede837960000d74978f0eba527be106b5b9aee0daf779d395276ed0b0614 \ + --hash=sha256:45b0cc4e3556cd875e09102988d1ab8356c998b596c9fced84547c8138b487a0 \ + --hash=sha256:476743fe6dfe14a2da12e3ac79125dc84a3b2cf8094369a47a1529b0cd8549fe \ + --hash=sha256:4773092f8019072343a7447203308b176e10199920eb02d6195e81bbb3274c29 \ + --hash=sha256:4b3dac63058cc36820b0dd072f89898604e2d39686fe05321729d00d8ac185a0 \ + --hash=sha256:4d1c96a7a18b9690a4d46df09e3e3382406ae3213727cd1019ebade1c4a81917 \ + --hash=sha256:51307f5c71007673a2bf8232ad973483d281e74cb99c8c5a990af1eefa6277d9 \ + --hash=sha256:51447e9aa2684679af07ca5021c3db526e0284347ebf4ffcec1154c3350cfe32 \ + --hash=sha256:58150c9f9b9a552505912d182ccdf26f6396fb6094816ceebcbb20eecabaed94 \ + --hash=sha256:5b10cd92fc5c498b35a8635df6d5a100207f88b63a4dc1de7ef9a548e1e2cd63 \ + --hash=sha256:5e226f6218febc71f6c1fc2fafb91c226f75bdc1d8fb12d66823716e891608fd \ + --hash=sha256:609b3ba8fcc0fb5ab7af00719d0fb6ad0cb518e48e7712d12fd68f1327951198 \ + --hash=sha256:60f44ade2cf573dad7a277e6f8ca9a51a21dda572b13bd7d8539bb3cd5dbedde \ + --hash=sha256:611057cc5d5c0afc743ba8be6bd828c17e0aaa8643f9d0a9b9bb7dea80eb8012 \ + --hash=sha256:6366a16e1a25018694d6a5d784d09b046edc9eac40ea2b54065c3052672516a1 \ + --hash=sha256:65a7ff3f705e57d392f7261b6d0550fe137c3019477431f1c355e0db0a7d3e15 \ + --hash=sha256:673611bbd43f0810bec0b0f028ddeaaa501190339cac411f347ac76917c3ae7b \ + --hash=sha256:67830fc78e67501f47bb950471b2dcb9b35b140084429318e862895a8e89c993 \ + --hash=sha256:68ce9f4d6b26d5ccbf7fd4459bf75f74a0a146677ebba80597df60cbdb20e6f4 \ + --hash=sha256:68e5f26a1ad57ded6d1cfb85331d1c1a195314756471d97758c48498bb4dcdf5 \ + --hash=sha256:69b157c5d3292bcd443faca052f3096f637f1e074b98212a933c074ae23dc3b8 \ + --hash=sha256:75286256590a6320cf106a0d28970d3560aad9ee09aa7b34fb40524792436d35 \ + --hash=sha256:78841cccf1af7b40f6f716338d50c0902dbe88d9f800b3c973b7a9a0a693a642 \ + --hash=sha256:78fa18e436a1a0e58dbd7e02fc4473f3f32cceb12df9dfca542d075961c307d2 \ + --hash=sha256:79580094b00d1789d1f93ea55bc43cb2f611910c72235b7657f3482ddcc1b22d \ + --hash=sha256:7b86a2b16095d250c6f58b3d9b2eee6f4147754344f3dab0922f7c9bf7d226c9 \ + --hash=sha256:83aed2c10721ddd90f68140685391b50811a880af20654c59af6b6c66c40513c \ + --hash=sha256:84fd18bcc17526fc2b3c1af7d2b9217d32c9c04448c16ec693b9b4f1985c3d33 \ + --hash=sha256:871ff67ea1aad4dfd91736464934d56b32dac49f9fbe16cddba36198a7b3a0db \ + --hash=sha256:898f0e9068ca27d37f8e83a5b962821df851532e6c4a7d615c1c033f9da6eedf \ + --hash=sha256:8a79d9f4d8001473a30c163556b3c3bfebec837495a412dde78b51672f6134f9 \ + --hash=sha256:8c041122946b7ba21bb32c45b1aa57b1be35527690aeb3c5c234521085632eee \ + --hash=sha256:90c44bc373b7687f6948b693cceaea1348ae0975d7474746559494468e3c1d84 \ + --hash=sha256:9104ed0bd76a429d46f9ec0dbc9b08ad1d2dcdf2b00a5a0daa1c145329b35b44 \ + --hash=sha256:920079c3f7456fa213e0829ed2073aaa727fd39d889ead5b4f35d0de5460d04f \ + --hash=sha256:93d59d504b230e83c7a843251681959a0b6a9cd76f6e146ce1b8a80eb8739af9 \ + --hash=sha256:9b2aff1c7b3884512b9512c3eaadd9bab39fb45042ffaaa1dd08ff2b9f8109d9 \ + --hash=sha256:9b8e0f3107e2200b76f6054de99016eac3ee6762713587b36baaa7e4bd2ae177 \ + --hash=sha256:9bb41182d93ea91f60b4bc8fbf4c820c69ef8a12ab2d917f3f1834f1acad07e8 \ + --hash=sha256:9cdef90ae47919cae358d8ab15797a800ed41da7aba5d72419fb510729e2ed4b \ + --hash=sha256:a1786910334ed46ab1dd73222f2cd1e05c2c3bb39f6dddb4f8b36fc382058a39 \ + --hash=sha256:a4cfde78a9f2880208d16a93b795726a3017d5977e08d1e162a7a31322479c41 \ + --hash=sha256:a4fbdde9dd4a9ce5fd52c2b3a347bb50cc89483ef783f1cb00d408c13f7a96c0 \ + --hash=sha256:aa99adc8f081b475a12843953db36831eaf83ec33eb46a90629ca6a5de45a616 \ + --hash=sha256:ac351b3b8014eead140e77e9717e2992c6bbe30b63bc3422422eb84865412e3d \ + --hash=sha256:ad41ba96094304aa090f5a30cb6e4fb3b3f1c264c523394b4c39bbacc4dc92ba \ + --hash=sha256:b5314963fce9b0b12743891de876e724997864ee22aa496f903f426c7e2fa5b2 \ + --hash=sha256:bcf74c1df76758a395bf0af608c04c82257523f55c9868b334f06270d0f2112b \ + --hash=sha256:bd47ba7fc3ca94896759ea0109775132d3e7ab921fbf54038e1bab2e46c313c9 \ + --hash=sha256:c0323c9daef75ef2e5083624b4585018a0c9d5e3b40f607eed81a311270b934b \ + --hash=sha256:c1225416b463483160e4af85d5fc3a9690ccb53fd4b1865a6437825f5ede3209 \ + --hash=sha256:c1c948747b03be832dceed96ca815cef7360de9aa19d37c730f8e3f6101aca48 \ + --hash=sha256:c25fe15c70c59eb7c5ce8c06a1f3fa1da0ecc5ea1e7a5922c40fd2fa9b0d5046 \ + --hash=sha256:cc1b0fff8ead343dae06305f954eb8468ba0ec1a97881f42489d198e4ce3c632 \ + --hash=sha256:cd6280cf040f233bd7d3407b743b4b4c74f70e8e1c4199cb112a62c941c0772a \ + --hash=sha256:cd6c3d4b783c556fa00bf540854e42f135e2f256abd29669fcd0da0f2dec79c2 \ + --hash=sha256:d4d6fcde76f94f5cb9e43e9e9a61f16dacefd228cbbf6f1a09bd9b219a92f1a1 \ + --hash=sha256:ddf4af30b417d9fe16481e9b81c27ab2a7cde1ff7ba3e85653b02db7d145dc7b \ + --hash=sha256:df115d4d83168fdf2cae48ef1ff6d1cb4c466364e30861b37121de0f3bf1b990 \ + --hash=sha256:df7276909358e5635ae203673ab7e509ddd224225a8d6b0790bf13eb2bde1cc5 \ + --hash=sha256:e4fd89cc178bced6ad29cb3e6dd4aa63fa5017c3524dbd0b25998fb64a87cc8b \ + --hash=sha256:e9701d0049d92c16703a42771b98d560b95248949f23f8cf7b4eddd201814fb9 \ + --hash=sha256:ee2f2a527e3c1a6e6411eb4209642e138b544a2d72fe5d0d76daf77b24063534 \ + --hash=sha256:f7fb7d750cfa0a070d2c24e831fd3481019a60dd317ea2b39acbcebc08b6ed81 \ + --hash=sha256:f840ed6d8ecba8255df8c42b87fadeda98ddfc6eeec05e2dc66e26d46dd6f58a \ + --hash=sha256:f86c6358749bd4fda175388691e3ba8c46e24c5347d0afd20f9b7edfc9faf07d \ + --hash=sha256:fa36ec09ef71d158186bc79e359ff5fdd6e7996fe8ab638f00d6b93139ba4fcf \ + --hash=sha256:fe2c7201c642b7c308f1675355ad7ff7b66acfe3541625efe5a3ad38f29d6115 + # via requests +click==8.4.2 \ + --hash=sha256:9a6cea6e60b17ebe0a44c5cc636d94f09bd66142c1cd7d8b4cd731c4917a15f6 \ + --hash=sha256:e6f9f66136c816745b9d65817da91d61d957fb16e02e4dcd0552553c5a197b76 + # via yake +colorama==0.4.6 ; sys_platform == 'win32' \ + --hash=sha256:08695f5cb7ed6e0531a20572697297273c47b8cae5a63ffc6d6ed5c201be6e44 \ + --hash=sha256:4f1d9991f5acc0ca119f9d443620b77f9d6b33703e51011c16baf57afb285fc6 + # via + # click + # tqdm + # typer +datasketch==2.0.0 \ + --hash=sha256:aea5ffafcce776e03d085740e78b874e778d779b07ee11ca636ca51b3fef09ed \ + --hash=sha256:e0570e170f7e64b8d6fb1cc2e4ce36a9f7036c5100167e50a0770addc50558c2 + # via tht (harness/pyproject.toml) +greenlet==3.5.3 ; platform_machine == 'AMD64' or platform_machine == 'WIN32' or platform_machine == 'aarch64' or platform_machine == 'amd64' or platform_machine == 'ppc64le' or platform_machine == 'win32' or platform_machine == 'x86_64' \ + --hash=sha256:0909f9355a9f24845d3299f3112e266a06afb68302041989fd26bd68894933db \ + --hash=sha256:0f41e4a05a3c0cb31b17023eff28dd111e1d16bf7d7d00406cd7df23f31398a7 \ + --hash=sha256:0f6ff50ff8dbd51fae9b37f4101648b04ea0df19b3f50ab2beb5061e7716a5c8 \ + --hash=sha256:0f71be4920368fe1fabeeaa53d1e3548337e2b223d9565f8ad5e392a75ba23fc \ + --hash=sha256:12a248ba75f6a9a236375f52296c498c89ff1d8badf32deb9eca7abd5853f7da \ + --hash=sha256:1540dd8e5fc2a5aec40fbb98ef8e149fa47c89a4b4a1cf2575a14d3d1869d7a8 \ + --hash=sha256:16d192579ed281051396dddd7f7754dac6259e6b1fb26378c87b66622f8e3f91 \ + --hash=sha256:176bc16a721fa5fc294d70b87b4dfa5fbdd251b3da5d5372735ecef9bd7d6d0c \ + --hash=sha256:19131729ae0ddc3c2e1ef85e650169b5e37ee32e400f215f78b94d7b0d567310 \ + --hash=sha256:1c514a468149bf8fbbab874188a3535cd8a48a3e353eb53a3d424296f8dbacd3 \ + --hash=sha256:1dae6e0091eae084317e411f047f0b7cb241c6db570f7c45fd6b900a274914ce \ + --hash=sha256:215275b1b49320987352e6c1b054acca0064f965a2c66992bed9a6f7d913f149 \ + --hash=sha256:232fec92e823addaf02d9472cf7381e24a1d046a6ced1103c5caa4c21b9dfc1d \ + --hash=sha256:2421c3564da9429d5586d46ca31ebb26516b5498a802cf65c041a8e8a8980d34 \ + --hash=sha256:271a8ea7c1024e8a0d7dd2be66dd66dda8a07193f41a17b9e924f7600f5b62be \ + --hash=sha256:2b2e857ae16f5f72142edf75f9f176fe7526ba19a2841df1420516f83831c9f2 \ + --hash=sha256:2ecda9ec22edf38fa389369eaed8c3d37c05f3c54e69f69438dbb2cc1de1458b \ + --hash=sha256:3236754d423955ea08e9bb5f6c04a7895f9e22c290b66aa7653fcb922d839eb0 \ + --hash=sha256:37bf9c538f5ae6e63d643f88dec37c0c83bdf0e2ebc62961dedcf458822f7b71 \ + --hash=sha256:4399eb8d041f20b68d943918bc55502a93d6fdc0a37c14da7881c04139acee9d \ + --hash=sha256:483d08c11181c83a6ce1a7a61df0f624a208ec40817a3bb2302714592eee4f04 \ + --hash=sha256:499fef2acede88c1864a57bb586b4bf533c81e1b82df7ab93451cdb47dfec227 \ + --hash=sha256:4b9d501b40e80b70e32323c799dd9b420a5577a9601469d362ae1ffb690f3a7c \ + --hash=sha256:4d77e67f65f98449e3fb83f795b5d0a8437aead2f874ca89c96576caf4be3af6 \ + --hash=sha256:5121af01cf911e70056c00d4b46d5e9b5d1415550038573d744138bacb59e6b8 \ + --hash=sha256:55cf4d777485d43110e47133cbba6d74a8885a87ec1227ef0267f9ee80c5aa21 \ + --hash=sha256:5795cd1101371140551c645f2d408b8d3c01a5a29cf8a9bce6e759c983682d23 \ + --hash=sha256:5b4807c4082c9d1b6d9eed56fcd041863e37f2228106eef24c30ca096e238605 \ + --hash=sha256:6219b6d04dbf6ba6084d77dc609e8473060dc55f759cbf626d512122781fa128 \ + --hash=sha256:629b614d2b786e89c50440e246f33eea78f58a962d0bdbbcc809e6d13605903f \ + --hash=sha256:6b1b0eed82364b0e32c4ea0f221452d33e6bb17ae094d9f72aed9851812747ea \ + --hash=sha256:6f73857adb8fee13fa56c172bd11262f888c0c648f9fea113e777bb2c7904a81 \ + --hash=sha256:719757059f5a53fd0dde23f78cffeafcdd97b21c850ddb7ca684a3c1a1f122e2 \ + --hash=sha256:73f152c895e09907e0dbe24f6c2db37beb085cd63db91c3825a0fcd0064124a8 \ + --hash=sha256:7669aa24cf2a1041d6f7899575b494a3ab4cf68bfcc8609b1dc0be7272db835e \ + --hash=sha256:766cfd421c13e450feb340cd472a3ed9957d438727b7b4593ad7c76c5d2b0deb \ + --hash=sha256:78dbef602fda6d97d957eb7937f70c9ce9e9527330347f8f6b6f9e554a9e7a47 \ + --hash=sha256:7ef56fe650f50575bf843acde967b9c567687f3c22340941a899b7bc56e956a8 \ + --hash=sha256:7faba15ac005376e02a0384504e0243be3370ce010296a44a820feb342b505ab \ + --hash=sha256:8540f1e6205bd13ca0ce685581037219ca54a1b41a0a15d228c6c9b8ad5903d7 \ + --hash=sha256:87142215824be6ac05e2e8e2786eec307ccbc27c36723c3881959df654af6861 \ + --hash=sha256:8bdb43e1a1d1873721acab2be99c5befd4d2044ddfd52e4d610801019880a702 \ + --hash=sha256:8d19fe6c39ebff9259f07bcc685d3290f8fa4ea2278e51dd0008e4d6b0f2d814 \ + --hash=sha256:8ff8bed3e3baa20a3ea261ce00526f1898ad4801d4886fd2220580ee0ad8fadf \ + --hash=sha256:915f887cf2682b66419b879423a2e072634aa7b7dce6f3ada4957cfced3f1e9a \ + --hash=sha256:962c5df2db8cb446da51edf1ca5296c389d93b99c9d8aa2ee4c7d0d8f1218260 \ + --hash=sha256:9ad04dd75458c6300b047c61b8639092433d205a25a14e310d6582a480efcca1 \ + --hash=sha256:9bcd2d72ccd70a1ec68ba6ef93e7fbb4420ef9997dabc7010d893bd4015e0bec \ + --hash=sha256:a1fad1d11e7d6aab184107baa8e4ece11ccba3ec9599cd7efa5ff4d70d43256a \ + --hash=sha256:a2d185dd1621757e70c3861cceffd5317ab4e7ed7eb09c82994828468527ade5 \ + --hash=sha256:a61efc018fd3eb317eeca31aba90ee9e7f26f22884a79b6c6ec715bf71bb62f1 \ + --hash=sha256:aca9b4ce85b152b5524ef7d88170efdff80dc0032aa8b75f9aaf7f3479ea95b4 \ + --hash=sha256:af4923b3096e26a36d7e9cf24ab88083a20f97d191e3b97f253731ce9b41b28c \ + --hash=sha256:afaabdd554cd7ae9bbb3ca070b0d7fdfd207dbf1d16865f7233837709d354bda \ + --hash=sha256:b363d46ed1ea431825fdb01471bb024fc08399bad1572a616e853c7684415adb \ + --hash=sha256:b7068bd09f761f3f5b4d214c2bed063186b2a86148c740b3873e3f56d79bac31 \ + --hash=sha256:b897d97759425953f69a9c0fac67f8fe333ec0ce7377ef186fb2b0c3ad5e354d \ + --hash=sha256:c180d22d325fb613956b443c3c6f4406eb70e6defc70d3974da2a7b59e06f48c \ + --hash=sha256:c4e7b79d83805475f0102008843f6eb45fd3bb0b2e88c774adab5fbaab27117d \ + --hash=sha256:c82304750f057167ff60d188df1d0cc1764ce9567eadf03e6a7443bcedd0b30b \ + --hash=sha256:c8d87c2134d871df96ecdea9cec7cbaab286dadab0f56476e57aaf9e8ac11550 \ + --hash=sha256:cde8adafa2365676f74a979744629589999093bc86e2484214f58e61df08902c \ + --hash=sha256:cefa9cef4b371f9844c6053db71f1138bc6807bab1578b0dae5149c1f1141357 \ + --hash=sha256:d27c0c653a60d9535f690226474a5cc1036a8b0d7b57504d1c4f89c44a07a80c \ + --hash=sha256:dc133a1569ee667b2a6ef56ce551084aeefd87a5acbc4736d336d1e2edc6cfc4 \ + --hash=sha256:dd99329bbc15ca78dcc583dba05d0b1b0bae01ab6c2174989f5aaee3e41ac930 \ + --hash=sha256:df0a0628d1597eb0897b62f55d1343f772405fd25f3b2a796c76874b0c2e22e8 \ + --hash=sha256:e0f0d160f0b2e558e6c75f7930967183255dc9735e5f5b8cae58ee09c9576d8b \ + --hash=sha256:e18619ba655ac05d78d80fc83cac4ba892bd6927b99e3b8237aee861aaacc8bb \ + --hash=sha256:e44da2f5bbdaabaf7d80b73dbb430c7035771e9f244e3c8b769715c9d8fa0a16 \ + --hash=sha256:e515757e2e36bcbf1fad09a46e1557e8b1ae1797d4b44d09da7deed88ad28608 \ + --hash=sha256:e81fa194a1d20967877bdf9c7794db2bc99063e5be36aee710c08f04c5bb087f \ + --hash=sha256:ea03f2f04367845d6b58eeed276e1e56e51f0b97d8ad5a88a7d20a91dc9056cc \ + --hash=sha256:ebd933a6adabc298bab47731a130fe6bfb888bd934eee37810f151159544540d \ + --hash=sha256:ec6f1af59f6b5f3fc9678e2ea062d8377d22ac644f7844cb7a292910cf12ff44 \ + --hash=sha256:efa9f765dd09f9d0cdac651ffdf631ee59ec5dc6ee7a73e0c012ba9c52fbdf5b \ + --hash=sha256:efc6bd60ea02e085862c74a3ef64b147ffc6f1a5ea7d9f26e7a939943f68c1e3 \ + --hash=sha256:fad5aec764399f1b5cc347ad250a59660f20c8f8888ea6bae1f93b769cce1154 \ + --hash=sha256:fd2e02fa07485778536a036222d616ab957b1d533f36b3ed98ce725d9c9d3117 + # via sqlalchemy +idna==3.18 \ + --hash=sha256:7f952cbe720b688055e3f87de14f5c3e5fdaa8bc3928985c4077ca689de849a2 \ + --hash=sha256:ffb385a7e039654cef1ab9ef32c6fafe283c0c0467bba1d9029738ce4a14a848 + # via requests +jellyfish==1.2.1 \ + --hash=sha256:0028857c5381c9d55e21cc6cb0d7f9545c3a9a7bb7dbca3960fe0a898c691ac2 \ + --hash=sha256:01647c12261bc1f7b102e918e7665497176d87f6fc96271439c8855872bc2606 \ + --hash=sha256:0368596e176bf548b3be2979ff33e274fb6d5e13b2cebe85137b8b698b002a85 \ + --hash=sha256:05be396aebe3dce7a8cb2f97727ecdf99e86457c48e97190775dce33f8b7e39d \ + --hash=sha256:07b022412ebece96759006cb015d46b8218d7f896d8b327c6bbee784ddf38ed9 \ + --hash=sha256:0b21c1596ce283fd7ee954eb0eeb007d59e480364324bcd91ad55146e91f3936 \ + --hash=sha256:1098ce1f84ae3f147f0a18a6803ffb09b9c8cd5fedce42465643ca0b5c9d0224 \ + --hash=sha256:10da696747e2de0336180fd5ba77ef769a7c80f9743123545f7fc0251efbbcec \ + --hash=sha256:1354b558a0a16597b6032dd0af64bebd24994f7e7484cf14993320eb764b06cb \ + --hash=sha256:137cfcc26396d0f2e1265ac61f800bb921921ea722a43dd897e58190f767c474 \ + --hash=sha256:13f1ac9caba22af10bfe42f674822643c0266009f882e0fe652079706dc5d13a \ + --hash=sha256:14bbb30d988dec1d12183cf5d4621c908f98add2009c72a185e8c3e8d00b804f \ + --hash=sha256:15318c13070fe6d9caeb7e10f9cdf89ff47c9d20f05a9a2c0d3b5cb8062a7033 \ + --hash=sha256:1a3ccff843822e7f3ad6f91662488a3630724c8587976bce114f3c7238e8ffa1 \ + --hash=sha256:1ffeeb6c78c45fbb6d2a22b0173fb8a6af849001d6c26fab49c525136dbd9734 \ + --hash=sha256:212aaf177236192a735bbbf5938717aa8518d14a25b08b015e47e783e70be060 \ + --hash=sha256:21baa92d4a5112167721156f6d061c2ae105f2995b3a5e19cec6662928f0c439 \ + --hash=sha256:2348f698f9c1d72023afc8d39939045421a01da9b7e3078e3029227e35f28419 \ + --hash=sha256:29cfa8bfb72aacf2d611a3313b358ed4d4140fa3d3efcffea750c8e7f8acb1aa \ + --hash=sha256:2c28a4ae3e201e1c1b7bacacd40e2e76c4068b90c9ae3a0d525e0ac98206f1cc \ + --hash=sha256:32581c50b34a09889b2d96796170e53da313a1e7fde32be63c82e50e7e791e3c \ + --hash=sha256:32a85b752cb51463face13e2b1797cfa617cd7fb7073f15feaa4020a86a346ce \ + --hash=sha256:393f609fd6139ce782e747e22c399483ffc58341009e6a97e39ffe5f5b2c674c \ + --hash=sha256:4072e21ad4036af41bd57b447b1dda64fe60aa679cfa8854ba0a0338152439f1 \ + --hash=sha256:451ddf4094e108e33d3b86d7817a7e20a2c5e6812d08c34ee22f6a595f38dcca \ + --hash=sha256:4a21d7eda5e6996772055f798e3fe1de1b33b3edad7f6cf0567097a21585a812 \ + --hash=sha256:4b013876109d91fa6fc871ffa4e0dbfda11820c33dc4ad0e2967b3fc1187f804 \ + --hash=sha256:4b28fcefc0c3534277ff0306e6c10672fb050f4784b5f3be7037e80801569fb5 \ + --hash=sha256:4b3e3223aaad74e18aacc74775e01815e68af810258ceea6fa6a81b19f384312 \ + --hash=sha256:4c5acb213aa75a61bcfc176566e20f2503069667e760d83d403b59e115fef0dd \ + --hash=sha256:4e36d9000d4f7e1a35689a74ec7749d27a216dfa6c47cac2e5ad3de8a523bd69 \ + --hash=sha256:509355ebedec69a8bf0cc113a6bf9c01820d12fe2eea44f47dfa809faf2d5463 \ + --hash=sha256:5335f622458aa105289a8e358bc32ecd1b9634b6ffec3e77ea3577e49c297171 \ + --hash=sha256:536c80d8d4ec7f39cbb10b85d926ff96cef3cde4a83ca0991c07cd9835d5dc13 \ + --hash=sha256:56da7632e029912af25e25422fae3b6df318400297d552791f4b21da6d815ed6 \ + --hash=sha256:5bda2275f31a64adf3483e39f7a4e2107f7dfe3a3f85f0d2c0cb6ae5fbe4a443 \ + --hash=sha256:5fa0ba0946f3c274f6a87aaa3c631dc70a363bd46cceea828ce777e8db653b6f \ + --hash=sha256:63770120cc3386dcc13bcc4df508ab281a6b14c3b2c0e33586439a6c40ee122f \ + --hash=sha256:675ab43840488944899ca87f02d4813c1e32107e56afaba7489705a70214e8aa \ + --hash=sha256:68080af234256ef943f0add6fc79816b0c643d8df291c17a85c1b6e45bdfbb96 \ + --hash=sha256:68ea3ddd4dae1152a7f7155ef02a7bfad919611158d71b301f9aa167685819af \ + --hash=sha256:6a49ce2a580edd3b16b69421137deef464e2f8907f9ef906d49950b1a52908c1 \ + --hash=sha256:6c51e565f85ce38cf9388c4f916d53888b0fa34788fcebe3aff3db24948e0960 \ + --hash=sha256:6d2bac5982d7a08759ea487bfa00149e6aa8a3be7cd43c4ed1be1e3505425c69 \ + --hash=sha256:6e76b23431a667cd485fb562428d1ad29bae9fdd0fcdfb5a51cc8087bae0e88c \ + --hash=sha256:72d2fda61b23babe862018729be73c8b0dc12e3e6601f36f6e65d905e249f4db \ + --hash=sha256:748dc45a0394fbe9120b8b3b9a39fab0967c7e2d6ecdd5304af018e774f80f96 \ + --hash=sha256:7853d2ed7d6929c029312ec849410f1ea7ae76ce72ad1140fb73f6e8a1e6aa4f \ + --hash=sha256:80a49eb817eaa6591f43a31e5c93d79904de62537f029907ef88c050d781a638 \ + --hash=sha256:91cad49a4fb731b726afc5ae385a3217a7016ed88a04da40c131cff8136a5db5 \ + --hash=sha256:98a133b40dc00cfda6609e1b0cb0ab0b77796fc2719aae886a12009514f73499 \ + --hash=sha256:9913789a98ccf49213fbb1dabc597847a0ec33d3b0e151689498f4b38ba9be0f \ + --hash=sha256:9930e20f0e9f65ad1d57d98290c2be3abd75812d058815605f44a56056fb9a66 \ + --hash=sha256:9a73b5c6425a70ebd440579a677eb4f03b327b2f59090db34e6c937aeea5aabd \ + --hash=sha256:9c747ae5c0fb4bd519f6abbfe4bd704b2f1c63fd4dd3dbb8d8864478974e1571 \ + --hash=sha256:9d4448c874959ae012cda0f6d570ac0bd7f0fcf12007714eaebf86b86919b66f \ + --hash=sha256:a058f4c6a591d5e5a47569f5648a26303ba19c76a960fef7e0beba2aa959e52e \ + --hash=sha256:a0ef6f0ecc085c1f8fddb048f538c8bb89989e5d470eab45d4e9bd48ee73a40d \ + --hash=sha256:a3cab91020e3ff7565e55a611ec3e3257c093ac950d55778a48bfc8c57562b6e \ + --hash=sha256:ab1bfea271ce4bda09d975080d5465cf5a8b127e7c0ea61ea3f972417a7a2193 \ + --hash=sha256:b35d4b5b688f759ffd075190a9850b04671bad14c5b37124eb43e99306ec16ea \ + --hash=sha256:b37b76ea338c4a473c34a9b9e1e033a78aafb9040a8c0eea579fc5805d8e4b46 \ + --hash=sha256:b8986d9768daddd5e87abf513ae168ea0afe690a444d4c82d5b1b14b0d045820 \ + --hash=sha256:baa30c7b59bd1c5e105693108a6d7a98f3e7a1a59e23e15bc5897b91fd5849f5 \ + --hash=sha256:bcdcd603a7737cd3f5a2ab10ce9b49844329deb81c2daafcd8131e54fc730205 \ + --hash=sha256:bd186c041d9be86c4fa5e2490943ce5d7f05b472f45d7f49426f259f3dd20bc4 \ + --hash=sha256:bebccd0652ac1c7e438ae1f451edefde63d14b3af6f6daa30c599919dcb92886 \ + --hash=sha256:c3c18f13175a9c90f3abd8805720b0eb3e10eca1d5d4e0cf57722b2a62d62016 \ + --hash=sha256:c499ea3a134130797c50e367687a6a46a12653c59af381bee92c41a5ab0bd55d \ + --hash=sha256:c85aa2bc76a36d92a3197f406f86636664d5b323727dfec4fa2842a8a24a06ae \ + --hash=sha256:c888f624d03e55e501bc438906505c79fb307d8da37a6dda18dd1ac2e6d5ea9c \ + --hash=sha256:cf6cd68921f2bacc547ba1cf64ad0e76bc1727f3bab13bba2e5f5869aba038b1 \ + --hash=sha256:d2b56a1fd2c5126c4a3362ec4470291cdd3c7daa22f583da67e75e30dc425ce6 \ + --hash=sha256:d7be8021658b46b22500a77f1707901bd98fc210f185c229b81c74efd3c1baf2 \ + --hash=sha256:db97d873f23b0c15b4ed911ece10e5cc0bb96cdc53666d5c3788bd0af81807f1 \ + --hash=sha256:dd895cf63fac0a9f11b524fff810d9a6081dcf3c518b34172ac8684eb504dd43 \ + --hash=sha256:ddf05ea471da2808d77ecfa425d8884124b4754f4d483afa7703b6655530cf5c \ + --hash=sha256:e1b990fb15985571616f7f40a12d6fa062897b19fb5359b6dec3cd811d802c24 \ + --hash=sha256:e4a210a960f3917da757b0581750b6e0a8db9acef68dafbc1b6e2ae39e847ba8 \ + --hash=sha256:e5977810972c6f0b2e61252c4758fd5aee21abf663ff309881195a99d37daa94 \ + --hash=sha256:e967e67058b78189d2b20a9586c7720a05ec4a580d6a98c796cd5cd2b7b11303 \ + --hash=sha256:ecf62d4aad0baa8832ab60f96e7baedbe6558bd292597503d927e9c5bce745d8 \ + --hash=sha256:f121218dc33fb318c34ddd889dc7362606ce1316af2bb63b73cc1df81523ca34 \ + --hash=sha256:f69aeb08659a6c81d559bbe319075e3417434ae5b3a5e4a758d1c4055a03497a \ + --hash=sha256:fb3c6e537cb4605c22895a8d4a10cdb26611ba2bbfc7f0b4c1d06bb9d8aad648 + # via yake +jmespath==1.1.0 \ + --hash=sha256:472c87d80f36026ae83c6ddd0f1d05d4e510134ed462851fd5f754c8c3cbb88d \ + --hash=sha256:a5663118de4908c91729bea0acadca56526eb2698e83de10cd116ae0f4e97c64 + # via + # boto3 + # botocore +markdown-it-py==4.2.0 \ + --hash=sha256:04a21681d6fbb623de53f6f364d352309d4094dd4194040a10fd51833e418d49 \ + --hash=sha256:9f7ebbcd14fe59494226453aed97c1070d83f8d24b6fc3a3bcf9a38092641c4a + # via rich +mdurl==0.1.2 \ + --hash=sha256:84008a41e51615a49fc9966191ff91509e3c40b939176e643fd50a5c2196b8f8 \ + --hash=sha256:bb413d29f5eea38f31dd4754dd7377d4465116fb207585f97bf925588687c1ba + # via markdown-it-py +networkx==3.6.1 \ + --hash=sha256:26b7c357accc0c8cde558ad486283728b65b6a95d85ee1cd66bafab4c8168509 \ + --hash=sha256:d47fbf302e7d9cbbb9e2555a0d267983d2aa476bac30e90dfbe5669bd57f3762 + # via yake +numpy==2.5.1 \ + --hash=sha256:08d60c810432eb83360958dea0999ac4cfb94531ea8efcbf0b7f277c2068aeb2 \ + --hash=sha256:09e9bfd8d2cf479c7d174804fb3811c53a8e9f20a37444008606b57d6b7a826d \ + --hash=sha256:0bfebd8695f9863592fe744be833a258120b14a9f39da255e8aa8fade2c0ddd1 \ + --hash=sha256:17a25e09640602e10bc8de0e6fa2b3fd68eedd84ba6d7842dc8f32f9ab87bd0b \ + --hash=sha256:1c6759f538fb912fc46de0a6b1758ccf7b57bc7c7ebebc23974fdac3de8db0cd \ + --hash=sha256:224ca51130ef7da85bea2191625181cb4f337f9cb64b471f10c1a12aa8b60077 \ + --hash=sha256:24d0eb82c0541d3415a33425db64ae439dffccd7b4dbcb30e7c35120205c506a \ + --hash=sha256:2ae0ca40bcb22d6ba59c1dfd5446f49940b0f2d821fde133f10dda11f816b84e \ + --hash=sha256:2c889b56fe48b1018f764b0eec8df59ab654e9148aa91faa12596043500de277 \ + --hash=sha256:30b44a6b53a7ae63c54c089a8726e5563ed302716c5b7ccc85afade40b0e7ff6 \ + --hash=sha256:32985c896d897419ef8da6917872d80b78ad0ea26d85b23245c7366ffde76d75 \ + --hash=sha256:3935f3b419b244a02732676fa5317a9193cc596a4c0646db07e5b421229ac9f7 \ + --hash=sha256:4939237038ada79308dda3204ac6462df056b5672b2e25db1149cf873668b3e1 \ + --hash=sha256:4b4ff1608417eb7a59da7b967bbb798cacfe071d2caf526a24281cd562072ed9 \ + --hash=sha256:54ad769f17bc2d833b620851989f62054fb9ab93c969d9e1dc3c8e3d56beea21 \ + --hash=sha256:59fda5e192b570217ec2580c96f00e9a7e12ef6866a900eb089b62c1a32545ca \ + --hash=sha256:5a4c988b38d261deeeaad9954e3deb091ad905c94e8bb6708654ef1d97f286b0 \ + --hash=sha256:5a6db61f9aaa57e369905c67d852045d3c4f7126405b29d09b19dec118e9c9cb \ + --hash=sha256:6165343f81b56ef8f514f396989e529b61d9dc709b99421b07e9f3e698e2287d \ + --hash=sha256:61ac47e772e6b8ea489e1d2f441a34c5c3ac17327e7ce294cbdf535795ad4e75 \ + --hash=sha256:6c3fe51bc6a16453d452997053454f309e8e0ed7b42d6b361ce4ac8c32913d74 \ + --hash=sha256:6eab239876581b2b3c5a242281b6007bbdbcd1c7085d7709bb57c5929b11e6bf \ + --hash=sha256:78798bd5b9ad744056af8efa90e3b9ddaa53272a0848a483084a1cc0a13b2dc0 \ + --hash=sha256:7c786fe9a5bbe360022e584c5a34cf6b54265c71bd7ec8ac3d8fec38968071f8 \ + --hash=sha256:83ce9c80d5b521b0d77ddcbe5447c218d247929b6cc056ca5351342accfff0af \ + --hash=sha256:9726558e8db4a5bf7929a70ae50f63abda4daf0efe810e3bfbab95976f75fc1a \ + --hash=sha256:99d5095fa265a0c4152e7bb12759e14381ef5496152f1ce58f44bdf55c44beb4 \ + --hash=sha256:a33276be12fa045805f477f22482088b66bb758ffbe89a9d21457de863a32e22 \ + --hash=sha256:a48a113e6afea91f5608793bafa7ef2ad481fefbda87ec5069f483de61cb9fa3 \ + --hash=sha256:ab451b59c5643c570974c43aef780703ef1d3b4965d2be07afd530615a9358d1 \ + --hash=sha256:ab84dc6b074fa881cae55bea94cc4f68e285181ba7f32497bf7dee6b1496165b \ + --hash=sha256:ab87a91b3cc3382b8956095bd8f95e00cf679bb81554339be1a2ba404a1473c1 \ + --hash=sha256:c12afb53450fa976d4c681c50a7423729a4c51c0465ed9f32b8a9cabbc472373 \ + --hash=sha256:caf3e317d33d60c37986b452613f4ab51246d0691350c03d0cb4a898627f4a95 \ + --hash=sha256:dc932a65ded7ce9013d120845a2514dcccb1a67bfc8deb8d37633762951904a6 \ + --hash=sha256:e68d8dd1e7eba712948f2053a29ec86917bc70ba1358df869d9f06649ef9cf09 \ + --hash=sha256:e824c2acf8862052246be5a44c15da1777940c60d010dd2aab897824d9c430f9 \ + --hash=sha256:e8c11c405efc5ff6816d5983c96cdfa215bab3428961243af3ff59b228490438 \ + --hash=sha256:efd736408cc97c79b9e6917338dfc8f06013b2274f992e96b1d9a81a71e2a2c2 \ + --hash=sha256:f089d7b00756190aacf1f5d34bdf38c3c430ac82b4f868f8cede73380460fce7 \ + --hash=sha256:f2479a47f8d5932d1718168a681ad6e536a9df484c83cfcf9de365e164537ace \ + --hash=sha256:f7119ebff1a9829e9f431a4f9d28e703023bb6b9fe7c8f724467dbfc27c94ab3 \ + --hash=sha256:f7d60026c0bdb1380e83bfa7a0419c4577ee4b9a08880afcb6dadeb74c649fa2 \ + --hash=sha256:f7feb014281029e628ba2d5a007407443b06e418b6fe451d1e2adcbc8eba0107 + # via + # datasketch + # scipy + # yake +psycopg2-binary==2.9.12 \ + --hash=sha256:00814e40fa23c2b37ef0a1e3c749d89982c73a9cb5046137f0752a22d432e82f \ + --hash=sha256:049366c6d884bdcd65d66e6ca1fdbebe670b56c6c9ba46f164e6667e90881964 \ + --hash=sha256:0dc9228d47c46bda253d2ecd6bb93b56a9f2d7ad33b684a1fa3622bf74ffe30c \ + --hash=sha256:1006fb62f0f0bc5ce256a832356c6262e91be43f5e4eb15b5eaf38079464caf2 \ + --hash=sha256:127467c6e476dd876634f17c3d870530e73ff454ff99bff73d36e80af28e1115 \ + --hash=sha256:1c8ad4c08e00f7679559eaed7aff1edfffc60c086b976f93972f686384a95e2c \ + --hash=sha256:29d4d134bd0ab46ffb04e94aa3c5fa3ef582e9026609165e2f758ff76fc3a3be \ + --hash=sha256:3471336e1acfd9c7fe507b8bad5af9317b6a89294f9eb37bd9a030bb7bebcdc6 \ + --hash=sha256:36512911ebb2b60a0c3e44d0bb5048c1980aced91235d133b7874f3d1d93487c \ + --hash=sha256:398fcd4db988c7d7d3713e2b8e18939776fd3fb447052daae4f24fa39daede4c \ + --hash=sha256:3d999bd982a723113c1a45b55a7a6a90d64d0ed2278020ed625c490ff7bef96c \ + --hash=sha256:40e7b28b63aaf737cb3a1edc3a9bbc9a9f4ad3dcb7152e8c1130e4050eddcb7d \ + --hash=sha256:411e85815652d13560fbe731878daa5d92378c4995a22302071890ec3397d019 \ + --hash=sha256:4413d0caef93c5cf50b96863df4c2efe8c269bf2267df353225595e7e15e8df7 \ + --hash=sha256:4766ab678563054d3f1d064a4db19cc4b5f9e3a8d9018592a8285cf200c248f3 \ + --hash=sha256:4dfcf8e45ebb0c663be34a3442f65e17311f3367089cd4e5e3a3e8e62c978777 \ + --hash=sha256:527e6342b3e44c2f0544f6b8e927d60de7f163f5723b8f1dfa7d2a84298738cd \ + --hash=sha256:54a0dfecab1b48731f934e06139dfe11e24219fb6d0ceb32177cf0375f14c7b5 \ + --hash=sha256:5a0253224780c978746cb9be55a946bcdaf40fe3519c0f622924cdabdafe2c39 \ + --hash=sha256:5ac9444edc768c02a6b6a591f070b8aae28ff3a99be57560ac996001580f294c \ + --hash=sha256:5c7cb4cbf894a1d36c720d713de507952c7c58f66d30834708f03dbe5c822ccf \ + --hash=sha256:5c8ce6c61bd1b1f6b9c24ee32211599f6166af2c55abb19456090a21fd16554b \ + --hash=sha256:5cdc05117180c5fa9c40eea8ea559ce64d73824c39d928b7da9fb5f6a9392433 \ + --hash=sha256:612b965daee295ae2da8f8218ce1d274645dc76ef3f1abf6a0a94fd57eff876d \ + --hash=sha256:63a3ebbd543d3d1eda088ac99164e8c5bac15293ee91f20281fd17d050aee1c4 \ + --hash=sha256:66a7685d7e548f10fb4ce32fb01a7b7f4aa702134de92a292c7bd9e0d3dbd290 \ + --hash=sha256:6f3b3de8a74ef8db215f22edffb19e32dc6fa41340456de7ec99efdc8a7b3ec2 \ + --hash=sha256:6f9cae1f848779b5b01f417e762c40d026ea93eb0648249a604728cda991dde3 \ + --hash=sha256:718e1fc18edf573b02cb8aea868de8d8d33f99ce9620206aa9144b67b0985e94 \ + --hash=sha256:77b348775efd4cdab410ec6609d81ccecd1139c90265fa583a7255c8064bc03d \ + --hash=sha256:7af18183109e23502c8b2ae7f6926c0882766f35b5175a4cd737ad825e4d7a1b \ + --hash=sha256:7c729a73c7b1b84de3582f73cdd27d905121dc2c531f3d9a3c32a3011033b965 \ + --hash=sha256:83946ba43979ebfdc99a3cd0ee775c89f221df026984ba19d46133d8d75d3cd9 \ + --hash=sha256:840066105706cd2eb29b9a1c2329620056582a4bf3e8169dec5c447042d0869f \ + --hash=sha256:863f5d12241ebe1c76a72a04c2113b6dc905f90b9cef0e9be0efd994affd9354 \ + --hash=sha256:864c261b3690e1207d14bbfe0a61e27567981b80c47a778561e49f676f7ce433 \ + --hash=sha256:89d19a9f7899e8eb0656a2b3a08e0da04c720a06db6e0033eab5928aabe60fa9 \ + --hash=sha256:8ffdb59fe88f99589e34354a130217aa1fd2d615612402d6edc8b3dbc7a44463 \ + --hash=sha256:96937c9c5d891f772430f418a7a8b4691a90c3e6b93cf72b5bd7cad8cbca32a5 \ + --hash=sha256:98062447aebc20ed20add1f547a364fd0ef8933640d5372ff1873f8deb9b61be \ + --hash=sha256:995ce929eede89db6254b50827e2b7fd61e50d11f0b116b29fffe4a2e53c4580 \ + --hash=sha256:9b818ceff717f98851a64bffd4c5eb5b3059ae280276dcecc52ac658dcf006a4 \ + --hash=sha256:9fe06d93e72f1c048e731a2e3e7854a5bfaa58fc736068df90b352cefe66f03f \ + --hash=sha256:a46fe069b65255df410f856d842bc235f90e22ffdf532dda625fd4213d3fd9b1 \ + --hash=sha256:a7e39a65b7d2a20e4ba2e0aaad1960b61cc2888d6ab047769f8347bd3c9ad915 \ + --hash=sha256:a99eaab34a9010f1a086b126de467466620a750634d114d20455f3a824aae033 \ + --hash=sha256:ab29414b25dcb698bf26bf213e3348abdcd07bbd5de032a5bec15bd75b298b03 \ + --hash=sha256:ace94261f43850e9e79f6c56636c5e0147978ab79eda5e5e5ebf13ae146fc8fe \ + --hash=sha256:b4a9eaa6e7f4ff91bec10aa3fb296878e75187bced5cc4bafe17dc40915e1326 \ + --hash=sha256:b6937f5fe4e180aeee87de907a2fa982ded6f7f15d7218f78a083e4e1d68f2a0 \ + --hash=sha256:b9a339b79d37c1b45f3235265f07cdeb0cb5ad7acd2ac7720a5920989c17c24e \ + --hash=sha256:ba3df2fc42a1cfa45b72cf096d4acb2b885937eedc61461081d53538d4a82a86 \ + --hash=sha256:c41321a14dd74aceb6a9a643b9253a334521babfa763fa873e33d89cfa122fb5 \ + --hash=sha256:c5ee5213445dd45312459029b8c4c0a695461eb517b753d2582315bd07995f5e \ + --hash=sha256:c6528cefc8e50fcc6f4a107e27a672058b36cc5736d665476aeb413ba88dbb06 \ + --hash=sha256:cb4a1dacdd48077150dc762a9e5ddbf32c256d66cb46f80839391aa458774936 \ + --hash=sha256:cfa2517c94ea3af6deb46f81e1bbd884faa63e28481eb2f889989dd8d95e5f03 \ + --hash=sha256:d2fa0d7caca8635c56e373055094eeda3208d901d55dd0ff5abc1d4e47f82b56 \ + --hash=sha256:d3227a3bc228c10d21011a99245edca923e4e8bf461857e869a507d9a41fe9f6 \ + --hash=sha256:d6fcbba8c9fed08a73b8ac61ea79e4821e45b1e92bb466230c5e746bbf3d5256 \ + --hash=sha256:e4e184b1fb6072bf05388aa41c697e1b2d01b3473f107e7ec44f186a32cfd0b8 \ + --hash=sha256:ee2d84ef5eb6c04702d2e9c372ad557fb027f26a5d82804f749dfb14c7fdd2ab \ + --hash=sha256:f12ae41fcafadb39b2785e64a40f9db05d6de2ac114077457e0e7c597f3af980 \ + --hash=sha256:f625abb7020e4af3432d95342daa1aa0db3fa369eed19807aa596367ba791b10 \ + --hash=sha256:f921f3cd87035ef7df233383011d7a53ea1d346224752c1385f1edfd790ceb6a \ + --hash=sha256:fb1828cf3da68f99e45ebce1355d65d2d12b6a78fb5dfb16247aad6bdef5f5d2 \ + --hash=sha256:ffdd7dc5463ccd61845ac37b7012d0f35a1548df9febe14f8dd549be4a0bc81e + # via tht (harness/pyproject.toml) +pydantic==2.13.4 \ + --hash=sha256:45a282cde31d808236fd7ea9d919b128653c8b38b393d1c4ab335c62924d9aba \ + --hash=sha256:c40756b57adaa8b1efeeced5c196f3f3b7c435f90e84ea7f443901bec8099ef6 + # via tht (harness/pyproject.toml) +pydantic-core==2.46.4 \ + --hash=sha256:00c603d540afdd6b80eb39f078f33ebd46211f02f33e34a32d9f053bba711de0 \ + --hash=sha256:0186750b482eefa11d7f435892b09c5c606193ef3375bcf94aa00ae6bfb66262 \ + --hash=sha256:041bde0a48fd37cf71cab1c9d56d3e8625a3793fef1f7dd232b3ff37e978ecda \ + --hash=sha256:0c563b08bca408dc7f65f700633d8442fffb2421fc47b8101377e9fd65051ff0 \ + --hash=sha256:0cbe8b01f948de4286c74cdd6c667aceb38f5c1e26f0693b3983d9d74887c65e \ + --hash=sha256:0ce40cd7b21210e99342afafbd4d0f76d784eb5b1d60f3bdc566be4983c6c73b \ + --hash=sha256:0e96592440881c74a213e5ad528e2b24d3d4f940de2766bed9010ab1d9e51594 \ + --hash=sha256:10e17cbb10a330363733efc4d7c4d0dd827ac0909b8f6a6542298fed1ea62f29 \ + --hash=sha256:133878133d271ade3d41d1bfb2a45ec38dbdbda40bc065921c6b04e4630127e2 \ + --hash=sha256:14d4edf427bdcf950a8a02d7cb44a08614388dd6e1bdcbf4f67504fa7887da9c \ + --hash=sha256:14f4c5d6db102bd796a627bbb3a17b4cf4574b9ae861d8b7c9a9661c6dd3362d \ + --hash=sha256:17299feefe090f2caa5b8e37222bb5f663e4935a8bfa6931d4102e5df1a9f398 \ + --hash=sha256:184c081504d17f1c1066e430e117142b2c77d9448a97f7b65c6ac9fd9aee238d \ + --hash=sha256:18e5ceec2ab67e6d5f1a9085e5a24c9c4e2ac4545730bfe668680bca05e555f3 \ + --hash=sha256:19e51f073cd3df251856a8a4189fbdf1de4012c3ebacfb1884f94f1eb406079f \ + --hash=sha256:1a7dd0b3ee80d90150e3495a3a13ac34dbcbfd4f012996a6a1d8900e91b5c0fb \ + --hash=sha256:1d8ba486450b14f3b1d63bc521d410ec7565e52f887b9fb671791886436a42f7 \ + --hash=sha256:2108ba5c1c1eca18030634489dc544844144ee36357f2f9f780b93e7ddbb44b5 \ + --hash=sha256:228ee9bae8bef5b1e97ec58302f80357c37199e0d0a99174e138d28e6957b9d9 \ + --hash=sha256:23ace664830ee0bfe014a0c7bc248b1f7f25ed7ad103852c317624a1083af462 \ + --hash=sha256:2412e734dcb48da14d4e4006b82b46b74f2518b8a26ee7e58c6844a6cd6d03c4 \ + --hash=sha256:29c61fc04a3d840155ff08e475a04809278972fe6aef51e2720554e96367e34b \ + --hash=sha256:2f84c03c8607173d16b5a854ec68a2f9079ae03237a54fb506d13af47e1d018d \ + --hash=sha256:3009f12e4e90b7f88b4f9adb1b0c4a3d58fe7820f3238c190047209d148026df \ + --hash=sha256:3245406455a5d98187ec35530fd772b1d799b26667980872c8d4614991e2c4a2 \ + --hash=sha256:3447661d99f75a3683a4cf5c87da72f2161964611864dbbeac7fbb118bb4bfc0 \ + --hash=sha256:372429a130e469c9cd698925ce5fc50940b7a1336b0d82038e63d5bbc4edc519 \ + --hash=sha256:395aebd9183f9d112f569aeb5b2214d1a10a33bec8456447f7fbdfa51d38d4cd \ + --hash=sha256:3a233125ac121aa3ffba9a2b59edfc4a985a76092dc8279586ab4b71390875e7 \ + --hash=sha256:3be77f45df024d789a672ae34f8b06fb346c4f9f46ea714956660ea4862e89ac \ + --hash=sha256:3bf92c5d0e00fefaab325a4d27828fe6b6e2a21848686b5b60d2d9eeb09d76c6 \ + --hash=sha256:3ecbc122d18468d06ca279dc26a8c2e2d5acb10943bb35e36ae92096dc3b5565 \ + --hash=sha256:3fb702cd90b0446a3a1c5e470bfa0dd23c0233b676a9099ddcc964fa6ca13898 \ + --hash=sha256:428e04521a40150c85216fc8b85e8d39fece235a9cf5e383761238c7fa9b96fb \ + --hash=sha256:432c179df7874eeb73307aad2df0755e1ae0efa61ff0ea89b93e194411ae3928 \ + --hash=sha256:4a05d69cba51d852c5c3e92758653245a50c0b646ced0cf05bd793ed592839d6 \ + --hash=sha256:4c63ebc82684aa89d9a3bcbd13d515b3be44250dc68dd3bd81526c1cb31286c3 \ + --hash=sha256:4fc73cb559bdb54b1134a706a2802a4cddd27a0633f5abb7e53056268751ac6a \ + --hash=sha256:4fcbe087dbc2068af7eda3aa87634eba216dbda64d1ae73c8684b621d33f6596 \ + --hash=sha256:56cb4851bcaf3d117eddcef4fe66afd750a50274b0da8e22be256d10e5611987 \ + --hash=sha256:5855698a4856556d86e8e6cd8434bc3ac0314ee8e12089ae0e143f64c6256e4e \ + --hash=sha256:5a4330cdbc57162e4b3aa303f588ba752257694c9c9be3e7ebb11b4aca659b5d \ + --hash=sha256:5b712b53160b79a5850310b912a5ef8e57e56947c8ad690c227f5c9d7e561712 \ + --hash=sha256:5d5902252db0d3cedf8d4a1bc68f70eeb430f7e4c7104c8c476753519b423008 \ + --hash=sha256:617d7e2ca7dcb8c5cf6bcb8c59b8832c94b36196bbf1cbd1bfb56ed341905edd \ + --hash=sha256:62f875393d7f270851f20523dd2e29f082bcc82292d66db2b64ea71f64b6e1c1 \ + --hash=sha256:633147d34cf4550417f12e2b1a0383973bdf5cdfde212cb09e9a581cf10820be \ + --hash=sha256:66ce7632c22d837c95301830e111ad0128a32b8207533b60896a96c4915192ea \ + --hash=sha256:6b3ace8194b0e5204818c92802dcdca7fc6d88aabbb799d7c795540d9cd6d292 \ + --hash=sha256:6f2eeda33a839975441c86a4119e1383c50b47faf0cbb5176985565c6bb02c33 \ + --hash=sha256:7027560ee92211647d0d34e3f7cd6f50da56399d26a9c8ad0da286d3869a53f3 \ + --hash=sha256:7283d57845ecf5a163403eb0702dfc220cc4fbdd18919cb5ccea4f95ee1cdab4 \ + --hash=sha256:7a5f930472650a82629163023e630d160863fce524c616f4e5186e5de9d9a49b \ + --hash=sha256:7bfb192b3f4b9e8a89b6277b6ce787564f62cfd272055f6e685726b111dc7826 \ + --hash=sha256:811ff8e9c313ab425368bcbb36e5c4ebd7108c2bbf4e4089cfbb0b01eff63fac \ + --hash=sha256:8233f2947cf85404441fd7e0085f53b10c93e0ee78611099b5c7237e36aacbf7 \ + --hash=sha256:82cf5301172168103724d49a1444d3378cb20cdee30b116a1bd6031236298a5d \ + --hash=sha256:8358a950c8909158e3df31538a7e4edc2d7265a7c54b47f0864d9e5bae9dcebf \ + --hash=sha256:85bb3611ff1802f3ee7fdd7dbff26b56f343fb432d57a4728fdd49b6ef35e2f4 \ + --hash=sha256:86e1a4418c6cd97d60c95c71164158eaf7324fae7b0923264016baa993eba6fc \ + --hash=sha256:8b9bab013d1c7a79d3501ff86d0bc9c31bf587db4551677b96bec07df78c6b15 \ + --hash=sha256:8c5dac79fa1614d1e06ca695109c6105923bd9c7d1d6c918d4e637b7e6b32fd3 \ + --hash=sha256:8d0820e8192167f80d88d64038e609c31452eeca865b4e1d9950a27a4609b00b \ + --hash=sha256:8daafc69c93ee8a0204506a3b6b30f586ef54028f52aeeeb5c4cfc5184fd5914 \ + --hash=sha256:9037063db01f09b09e237c282b6792bd4da634b5402c4e7f0c61effed7701a04 \ + --hash=sha256:905a0ed8ea6f2d61c1738835f99b699348d7857379083e5fc497fa0c967a407c \ + --hash=sha256:90884113d8b48f760e9587002789ddd741e76ab9f89518cd1e43b1f1a52ec44b \ + --hash=sha256:91a06d2e259ecfbd8c901d70c3c507900458498142b3026a296b7de4d1322cc9 \ + --hash=sha256:926c9541b14b12b1681dca8a0b75feb510b06c6341b70a8e500c2fdcff837cce \ + --hash=sha256:9401557acd873c3a7f3eb9383edef8ac4968f9510e340f4808d427e75667e7b4 \ + --hash=sha256:9551187363ffc0de2a00b2e47c25aeaeb1020b69b668762966df15fc5659dd5a \ + --hash=sha256:962ccbab7b642487b1d8b7df90ef677e03134cf1fd8880bf698649b22a69371f \ + --hash=sha256:97e7cf2be5c77b7d1a9713a05605d49460d02c6078d38d8bef3cbe323c548424 \ + --hash=sha256:9aa768456404a8bf48a4406685ac2bec8e72b62c69313734fa3b73cf33b3a894 \ + --hash=sha256:9bc519fbf2b7578398853d815009ae5e4d4603d12f4e3f91da8c06852d3da3e9 \ + --hash=sha256:9d56801be94b86a9da183e5f3766e6310752b99ff647e38b09a9500d88e46e76 \ + --hash=sha256:9f444c499b3eefd3a92e348059471ea0c3a6e303d9c1cec09fa748fd9f895201 \ + --hash=sha256:9fa8ae11da9e2b3126c6426f147e0fba88d96d65921799bb30c6abd1cb2c97fb \ + --hash=sha256:a0f62d0a58f4e7da165457e995725421e0064f2255d8eccebc49f41bbc23b109 \ + --hash=sha256:a396dcc17e5a0b164dbe026896245a4fa9ff402edca1dff0be3d53a517f74de4 \ + --hash=sha256:aaa2a54443eff1950ba5ddc6b6ccda0d9c84a364276a62f969bdf2a390650848 \ + --hash=sha256:ad785e92e6dc634c21555edc8bd6b64957ab844541bcb96a1366c202951ae526 \ + --hash=sha256:af8244b2bef6aaad6d92cda81372de7f8c8d36c9f0c3ea36e827c60e7d9467a0 \ + --hash=sha256:b078afbc25f3a1436c7a1d2cd3e322497ee99615ba97c563566fdf46aff1ee01 \ + --hash=sha256:b2f69dec1725e79a012d920df1707de5caf7ed5e08f3be4435e25803efc47458 \ + --hash=sha256:b8458003118a712e66286df6a707db01c52c0f52f7db8e4a38f0da1d3b94fc4e \ + --hash=sha256:bb63e0198ca18aad131c089b9204c23079c3afa95487e561f4c522d519e55aba \ + --hash=sha256:bfec22eab3c8cc2ceec0248aec886624116dc079afa027ecc8ad4a7e62010f8a \ + --hash=sha256:c1747f85cee84c26985853c6f3d9bd3e75da5212912443fa111c113b9c246f39 \ + --hash=sha256:c1b3f518abeca3aa13c712fd202306e145abf59a18b094a6bafb2d2bbf59192c \ + --hash=sha256:c50f2528cf200c5eed56faf3f4e22fcd5f38c157a8b78576e6ba3168ec35f000 \ + --hash=sha256:c68fcd102d71ea85c5b2dfac3f4f8476eff42a9e078fd5faefff6d145063536b \ + --hash=sha256:c7a7bd4e39e8e4c12c39cd480356842b6a8a06e41b23a55a5e3e191718838ddf \ + --hash=sha256:c94f0688e7b8d0a67abf40e57a7eaaecd17cc9586706a31b76c031f63df052b4 \ + --hash=sha256:cbaf13819775b7f769bf4a1f066cb6df7a28d4480081a589828ef190226881cd \ + --hash=sha256:cd2213145bcc2ba85884d0ac63d222fece9209678f77b9b4d76f054c561adb28 \ + --hash=sha256:ce5c1d2a8b27468f433ca974829c44060b8097eedc39933e3c206a90ee49c4a9 \ + --hash=sha256:d396ec2b979760aaf3218e76c24e65bd0aca24983298653b3a9d7a45f9e47b30 \ + --hash=sha256:d51026d73fcfd93610abc7b27789c26b313920fcfb20e27462d74a7f8b06e983 \ + --hash=sha256:d80ee3d731373b24cebbc10d689ca4ee1875caf0d5703a245db18efd4dd37fc1 \ + --hash=sha256:d995260fdf4e1db774581b4900e0f832abe3c7c84996726bbc161b19c8f29e76 \ + --hash=sha256:da4b951fe36dc7c3a1ccb4e3cd1747c3542b8c9ceede8fc86cae054e764485f5 \ + --hash=sha256:daa27d92c36f24388fe3ad306b174781c747627f134452e4f128ea00ce1fe8c4 \ + --hash=sha256:db06ffe51636ffe9ca531fe9023dd64bdd794be8754cb5df57c5498ae5b518a7 \ + --hash=sha256:e0d65b8c354be7fb5f720c3caa8bc940bc2d20ce749c8e06135f07f8ed95dd7c \ + --hash=sha256:e68b7a074f65a2fd746c52a7ce6142ab7006074ac269ace0c25cd8ba171f8066 \ + --hash=sha256:e739fee756ba1010f8bcccb534252e85a35fe45ae92c295a06059ce58b74ccd3 \ + --hash=sha256:e846ae7835bf0703ae43f534ab79a867146dadd59dc9ca5c8b53d5c8f7c9ef02 \ + --hash=sha256:e9c26f834c65f5752f3f06cb08cb86a913ceb7274d0db6e267808a708b46bc89 \ + --hash=sha256:ea793e075b70290d89d8142074262885d3f7da19634845135751bd6344f73b50 \ + --hash=sha256:f027324c56cd5406ca49c124b0db10e56c69064fec039acc571c29020cc87c76 \ + --hash=sha256:f13a646d65d09fbf1bc6b3a9635d30095c8e7e5cc419ff35ecc563c5fd04cd49 \ + --hash=sha256:f47286a97f0bc9b8859519809077b91b2cefe4ae47fcbf5e466a009c1c5d742b \ + --hash=sha256:f747929cf940cddb5b3668a390056ddd5ba2e5010615ea2dcf4f9c4f3ab8791d \ + --hash=sha256:f99626688942fb746e545232e7726926f3be91b5975f8b55327665fafda991c7 \ + --hash=sha256:f9fa868638bf362d3d138ea55829cefb3d5f4b0d7f142234382a15e2485dbec4 \ + --hash=sha256:fbdb89b3e1c94a30cc5edfce477c6e6a5dc4d8f84665b455c27582f211a1c72c \ + --hash=sha256:fc010ab034c8c7452522748bf937df58020d256ccae0874463d1f4d01758af8e \ + --hash=sha256:fc3e9034a63de20e15e8ade85358bc6efc614008cab72898b4b4952bea0509ff \ + --hash=sha256:fd8b3d9fd264be37976686c7f65cd52a83f5e84f4bfd2adf9c1d469676bbb6ae + # via pydantic +pygments==2.20.0 \ + --hash=sha256:6757cd03768053ff99f3039c1a36d6c0aa0b263438fcab17520b30a303a82b5f \ + --hash=sha256:81a9e26dd42fd28a23a2d169d86d7ac03b46e2f8b59ed4698fb4785f946d0176 + # via rich +python-dateutil==2.9.0.post0 \ + --hash=sha256:37dd54208da7e1cd875388217d5e00ebd4179249f90fb72437e91a35459a0ad3 \ + --hash=sha256:a8b2bc7bffae282281c8140a97d3aa9c14da0b136dfe83f850eea9a5f7470427 + # via botocore +python-dotenv==1.2.2 \ + --hash=sha256:1d8214789a24de455a8b8bd8ae6fe3c6b69a5e3d64aa8a8e5d68e694bbcb285a \ + --hash=sha256:2c371a91fbd7ba082c2c1dc1f8bf89ca22564a087c2c287cd9b662adde799cf3 + # via tht (harness/pyproject.toml) +pyyaml==6.0.3 \ + --hash=sha256:00c4bdeba853cc34e7dd471f16b4114f4162dc03e6b7afcc2128711f0eca823c \ + --hash=sha256:0150219816b6a1fa26fb4699fb7daa9caf09eb1999f3b70fb6e786805e80375a \ + --hash=sha256:02893d100e99e03eda1c8fd5c441d8c60103fd175728e23e431db1b589cf5ab3 \ + --hash=sha256:02ea2dfa234451bbb8772601d7b8e426c2bfa197136796224e50e35a78777956 \ + --hash=sha256:0f29edc409a6392443abf94b9cf89ce99889a1dd5376d94316ae5145dfedd5d6 \ + --hash=sha256:10892704fc220243f5305762e276552a0395f7beb4dbf9b14ec8fd43b57f126c \ + --hash=sha256:16249ee61e95f858e83976573de0f5b2893b3677ba71c9dd36b9cf8be9ac6d65 \ + --hash=sha256:1d37d57ad971609cf3c53ba6a7e365e40660e3be0e5175fa9f2365a379d6095a \ + --hash=sha256:1ebe39cb5fc479422b83de611d14e2c0d3bb2a18bbcb01f229ab3cfbd8fee7a0 \ + --hash=sha256:214ed4befebe12df36bcc8bc2b64b396ca31be9304b8f59e25c11cf94a4c033b \ + --hash=sha256:2283a07e2c21a2aa78d9c4442724ec1eb15f5e42a723b99cb3d822d48f5f7ad1 \ + --hash=sha256:22ba7cfcad58ef3ecddc7ed1db3409af68d023b7f940da23c6c2a1890976eda6 \ + --hash=sha256:27c0abcb4a5dac13684a37f76e701e054692a9b2d3064b70f5e4eb54810553d7 \ + --hash=sha256:28c8d926f98f432f88adc23edf2e6d4921ac26fb084b028c733d01868d19007e \ + --hash=sha256:2e71d11abed7344e42a8849600193d15b6def118602c4c176f748e4583246007 \ + --hash=sha256:34d5fcd24b8445fadc33f9cf348c1047101756fd760b4dacb5c3e99755703310 \ + --hash=sha256:37503bfbfc9d2c40b344d06b2199cf0e96e97957ab1c1b546fd4f87e53e5d3e4 \ + --hash=sha256:3c5677e12444c15717b902a5798264fa7909e41153cdf9ef7ad571b704a63dd9 \ + --hash=sha256:3ff07ec89bae51176c0549bc4c63aa6202991da2d9a6129d7aef7f1407d3f295 \ + --hash=sha256:41715c910c881bc081f1e8872880d3c650acf13dfa8214bad49ed4cede7c34ea \ + --hash=sha256:418cf3f2111bc80e0933b2cd8cd04f286338bb88bdc7bc8e6dd775ebde60b5e0 \ + --hash=sha256:44edc647873928551a01e7a563d7452ccdebee747728c1080d881d68af7b997e \ + --hash=sha256:4a2e8cebe2ff6ab7d1050ecd59c25d4c8bd7e6f400f5f82b96557ac0abafd0ac \ + --hash=sha256:4ad1906908f2f5ae4e5a8ddfce73c320c2a1429ec52eafd27138b7f1cbe341c9 \ + --hash=sha256:501a031947e3a9025ed4405a168e6ef5ae3126c59f90ce0cd6f2bfc477be31b7 \ + --hash=sha256:5190d403f121660ce8d1d2c1bb2ef1bd05b5f68533fc5c2ea899bd15f4399b35 \ + --hash=sha256:5498cd1645aa724a7c71c8f378eb29ebe23da2fc0d7a08071d89469bf1d2defb \ + --hash=sha256:5cf4e27da7e3fbed4d6c3d8e797387aaad68102272f8f9752883bc32d61cb87b \ + --hash=sha256:5e0b74767e5f8c593e8c9b5912019159ed0533c70051e9cce3e8b6aa699fcd69 \ + --hash=sha256:5ed875a24292240029e4483f9d4a4b8a1ae08843b9c54f43fcc11e404532a8a5 \ + --hash=sha256:5fcd34e47f6e0b794d17de1b4ff496c00986e1c83f7ab2fb8fcfe9616ff7477b \ + --hash=sha256:5fdec68f91a0c6739b380c83b951e2c72ac0197ace422360e6d5a959d8d97b2c \ + --hash=sha256:6344df0d5755a2c9a276d4473ae6b90647e216ab4757f8426893b5dd2ac3f369 \ + --hash=sha256:64386e5e707d03a7e172c0701abfb7e10f0fb753ee1d773128192742712a98fd \ + --hash=sha256:652cb6edd41e718550aad172851962662ff2681490a8a711af6a4d288dd96824 \ + --hash=sha256:66291b10affd76d76f54fad28e22e51719ef9ba22b29e1d7d03d6777a9174198 \ + --hash=sha256:66e1674c3ef6f541c35191caae2d429b967b99e02040f5ba928632d9a7f0f065 \ + --hash=sha256:6adc77889b628398debc7b65c073bcb99c4a0237b248cacaf3fe8a557563ef6c \ + --hash=sha256:79005a0d97d5ddabfeeea4cf676af11e647e41d81c9a7722a193022accdb6b7c \ + --hash=sha256:7c6610def4f163542a622a73fb39f534f8c101d690126992300bf3207eab9764 \ + --hash=sha256:7f047e29dcae44602496db43be01ad42fc6f1cc0d8cd6c83d342306c32270196 \ + --hash=sha256:8098f252adfa6c80ab48096053f512f2321f0b998f98150cea9bd23d83e1467b \ + --hash=sha256:850774a7879607d3a6f50d36d04f00ee69e7fc816450e5f7e58d7f17f1ae5c00 \ + --hash=sha256:8d1fab6bb153a416f9aeb4b8763bc0f22a5586065f86f7664fc23339fc1c1fac \ + --hash=sha256:8da9669d359f02c0b91ccc01cac4a67f16afec0dac22c2ad09f46bee0697eba8 \ + --hash=sha256:8dc52c23056b9ddd46818a57b78404882310fb473d63f17b07d5c40421e47f8e \ + --hash=sha256:9149cad251584d5fb4981be1ecde53a1ca46c891a79788c0df828d2f166bda28 \ + --hash=sha256:93dda82c9c22deb0a405ea4dc5f2d0cda384168e466364dec6255b293923b2f3 \ + --hash=sha256:96b533f0e99f6579b3d4d4995707cf36df9100d67e0c8303a0c55b27b5f99bc5 \ + --hash=sha256:9c57bb8c96f6d1808c030b1687b9b5fb476abaa47f0db9c0101f5e9f394e97f4 \ + --hash=sha256:9c7708761fccb9397fe64bbc0395abcae8c4bf7b0eac081e12b809bf47700d0b \ + --hash=sha256:9f3bfb4965eb874431221a3ff3fdcddc7e74e3b07799e0e84ca4a0f867d449bf \ + --hash=sha256:a33284e20b78bd4a18c8c2282d549d10bc8408a2a7ff57653c0cf0b9be0afce5 \ + --hash=sha256:a80cb027f6b349846a3bf6d73b5e95e782175e52f22108cfa17876aaeff93702 \ + --hash=sha256:b30236e45cf30d2b8e7b3e85881719e98507abed1011bf463a8fa23e9c3e98a8 \ + --hash=sha256:b3bc83488de33889877a0f2543ade9f70c67d66d9ebb4ac959502e12de895788 \ + --hash=sha256:b865addae83924361678b652338317d1bd7e79b1f4596f96b96c77a5a34b34da \ + --hash=sha256:b8bb0864c5a28024fac8a632c443c87c5aa6f215c0b126c449ae1a150412f31d \ + --hash=sha256:ba1cc08a7ccde2d2ec775841541641e4548226580ab850948cbfda66a1befcdc \ + --hash=sha256:bdb2c67c6c1390b63c6ff89f210c8fd09d9a1217a465701eac7316313c915e4c \ + --hash=sha256:c1ff362665ae507275af2853520967820d9124984e0f7466736aea23d8611fba \ + --hash=sha256:c2514fceb77bc5e7a2f7adfaa1feb2fb311607c9cb518dbc378688ec73d8292f \ + --hash=sha256:c3355370a2c156cffb25e876646f149d5d68f5e0a3ce86a5084dd0b64a994917 \ + --hash=sha256:c458b6d084f9b935061bc36216e8a69a7e293a2f1e68bf956dcd9e6cbcd143f5 \ + --hash=sha256:d0eae10f8159e8fdad514efdc92d74fd8d682c933a6dd088030f3834bc8e6b26 \ + --hash=sha256:d76623373421df22fb4cf8817020cbb7ef15c725b9d5e45f17e189bfc384190f \ + --hash=sha256:ebc55a14a21cb14062aa4162f906cd962b28e2e9ea38f9b4391244cd8de4ae0b \ + --hash=sha256:eda16858a3cab07b80edaf74336ece1f986ba330fdb8ee0d6c0d68fe82bc96be \ + --hash=sha256:ee2922902c45ae8ccada2c5b501ab86c36525b883eff4255313a253a3160861c \ + --hash=sha256:efd7b85f94a6f21e4932043973a7ba2613b059c4a000551892ac9f1d11f5baf3 \ + --hash=sha256:f7057c9a337546edc7973c0d3ba84ddcdf0daa14533c2065749c9075001090e6 \ + --hash=sha256:fa160448684b4e94d80416c0fa4aac48967a969efe22931448d853ada8baf926 \ + --hash=sha256:fc09d0aa354569bc501d4e787133afc08552722d3ab34836a80547331bb5d4a0 + # via tht (harness/pyproject.toml) +regex==2026.7.10 \ + --hash=sha256:0639b2488b775a0109f55a5a2172deebdedb4b6c5ab0d48c90b43cbf5de58d17 \ + --hash=sha256:081acf191b4d614d573a56cab69f948b6864daa5e3cc69f209ee92e26e454c2f \ + --hash=sha256:0911e34151a5429d0325dae538ba9851ec0b62426bdfd613060cda8f1c36ec7f \ + --hash=sha256:103e8f3acc3dcede88c0331c8612766bdcfc47c9250c5477f0e10e0550b9da49 \ + --hash=sha256:1050fedf0a8a92e843971120c2f57c3a99bea86c0dfa1d63a9fac053fe54b135 \ + --hash=sha256:13fba679fe035037e9d5286620f88bbfd105df4d5fcd975942edd282ab986775 \ + --hash=sha256:14d27f6bd04beb01f6a25a1153d73e58c290fd45d92ba56af1bb44199fd1010d \ + --hash=sha256:177f930af3ad72e1045f8877540e0c43a38f7d328cf05f31963d0bd5f7ecf067 \ + --hash=sha256:1f0d4ccf70b1d13711242de0ba78967db5c35d12ac408378c70e06295c3f6644 \ + --hash=sha256:21150500b970b12202879dfd82e7fd809d8e853140fff84d08e57a90cf1e154e \ + --hash=sha256:2129e4a5e86f26926982d883dff815056f2e98220fdf630e59f961b578a26c43 \ + --hash=sha256:221f2771cb780186b94bbf125a151bbeb242fa1a971da6ad59d7b0370f19de9a \ + --hash=sha256:234f8e0d65cf1df9becadae98648f74030ee85a8f12edcb5eb0f60a22a602197 \ + --hash=sha256:28a0973eeffff4292f5a7ee498ab65d5e94ee8cc9cea364239251eb4a260a0f1 \ + --hash=sha256:2b93eafd92c4128bab2f93500e8912cc9ecb3d3765f6685b902c6820d0909b6b \ + --hash=sha256:2bc350e1c5fa250f30ab0c3e38e5cfdffcd82cb8af224df69955cab4e3003812 \ + --hash=sha256:2c66a8a1969cfd506d1e203c0005fd0fc3fe6efc83c945606566b6f9611d4851 \ + --hash=sha256:2f98ef73a13791a387d5c841416ad7f52040ae5caf10bcf46fa12bd2b3d63745 \ + --hash=sha256:31fa17378b29519bfd0a1b8ba4e9c10cf0baf1cf4099b39b0689429e7dc2c795 \ + --hash=sha256:3750c42d47712e362158a04d0fd80131f73a55e8c715b2885442a0ff6f9fc3fc \ + --hash=sha256:38a5926601aaccf379512746b86eb0ac1d29121f6c776dac6ac5b31077432f2c \ + --hash=sha256:396ea70e4ea1f19571940add3bad9fd3eb6a19dc610d0d01f692bc1ba0c10cb4 \ + --hash=sha256:39f81d1fdf594446495f2f4edd8e62d8eda0f7a802c77ac596dc8448ad4cc5ca \ + --hash=sha256:3d8ef9df02c8083c7b4b855e3cb87c8e0ebbcfea088d98c7a886aaefdf88d837 \ + --hash=sha256:3e23458d8903e33e7d27196d7a311523dc4e2f4137a5f34e4dbd30c8d37ff33e \ + --hash=sha256:3f03b92fb6ec739df042e45b06423fc717ecf0063e07ffe2897f7b2d5735e1e8 \ + --hash=sha256:3f361215e000d68a4aff375106637b83c80be36091d83ee5107ad3b32bd73f48 \ + --hash=sha256:41a47c2b28d9421e2509a4583a22510dc31d83212fcf38e1508a7013140f71a8 \ + --hash=sha256:441edc66a54063f8269d1494fc8474d06605e71e8a918f4bcfd079ebda4ce042 \ + --hash=sha256:4533af6099543db32ef26abc2b2f824781d4eebb309ab9296150fd1a0c7eb07d \ + --hash=sha256:4574feca202f8c470bf678aed8b5d89df04aaf8dc677f3b83d92825051301c0f \ + --hash=sha256:460176b2db044a292baaee6891106566739657877af89a251cded228689015a6 \ + --hash=sha256:494b19a5805438aeb582de99f9d97603d8fd48e6f4cc74d0088bb292b4da3b70 \ + --hash=sha256:4db009b4fc533d79af3e841d6c8538730423f82ea8508e353a3713725de7901c \ + --hash=sha256:538ddb143f5ca085e372def17ef3ed9d74b50ad7fc431bd85dc50a9af1a7076f \ + --hash=sha256:53bbbd6c610489700f7110db1d85f3623924c3f7c760f987eca033867360788a \ + --hash=sha256:53f54993b462f3f91fea0f2076b46deb6619a5f45d70dbd1f543f789d8b900ef \ + --hash=sha256:58a4571b2a093f6f6ee4fd281faa8ebf645abcf575f758173ea2605c7a1e1ecb \ + --hash=sha256:5c363de7c0339d39341b6181839ed32509820b85ef506deafcf2e7e43baadab4 \ + --hash=sha256:5e792367e5f9b4ffb8cad93f1beaa91837056b94da98aa5c65a0db0c1b474927 \ + --hash=sha256:5eab9d3f981c423afd1a61db055cfe83553c3f6455949e334db04722469dd0a2 \ + --hash=sha256:617e8f10472e34a8477931f978ff3a88d46ae2ba0e41927e580b933361f60948 \ + --hash=sha256:64722a5031aeace7f6c8d5ea9a9b22d9368af0d6e8fa532585da8158549ea963 \ + --hash=sha256:65ee5d1ac3cd541325f5ac92625b1c1505f4d171520dd931bda7952895c5321a \ + --hash=sha256:668ab85105361d0200e3545bec198a1acfc6b0aeb5fff8897647a826e5a171be \ + --hash=sha256:66d2c35587cd601c95965d5c0415058ba5cfd6ffbab7624ce198bd967102b341 \ + --hash=sha256:6cbedeb5112f59dbd169385459b9943310bdd241c6966c19c5f6e2295055c93a \ + --hash=sha256:6e3448e86b05ce87d4eb50f9c680860830f3b32493660b39f43957d6263e2eba \ + --hash=sha256:724ee9379568658ec06362cf24325c5315cc5a67f61dfe585bfeff58300a355b \ + --hash=sha256:7252b48b0c60100095088fbeb281fca9a4fcf678a4e04b1c520c3f8613c952c4 \ + --hash=sha256:732c19e5828eb287d01edb83b2eb87f283ba8e5fc3441c732709d3e8cbd14aaa \ + --hash=sha256:749b92640e1970e881fdf22a411d74bf9d049b154f4ef7232eeb9a90dd8be7f3 \ + --hash=sha256:74ae61d8573ecd51b5eeee7be2218e4c56e99c14fa8fcf97cf7519611d4be92e \ + --hash=sha256:78712d4954234df5ca24fdadb65a2ab034213f0cdfde376c272f9fc5e09866bb \ + --hash=sha256:799a369bdab91dcf0eb424ebd7aa9650897025ce22f729248d8f2c72002c4daa \ + --hash=sha256:80151ca5bfc6c4524186b3e08b499e97319b2001fc265ed2d4fc12c0d5692cdf \ + --hash=sha256:82ab8330e7e2e416c2d42fcec67f02c242393b8681014750d4b70b3f158e1f08 \ + --hash=sha256:8331484450b3894298bef8abecce532171ff6ac60b71f999eed10f2c01941a8a \ + --hash=sha256:834271b1ff2cfa1f67fcd65a48bf11d11e9ab837e21bf79ce554efb648599ae8 \ + --hash=sha256:8679f0652a183d93da646fcec8da8228db0be40d1595da37e6d74c2dc8c4713c \ + --hash=sha256:87794549a3f5c1c2bdfba2380c1bf87b931e375f4133d929da44f95e396bf5fe \ + --hash=sha256:87b776cf2890e356e4ab104b9df846e169da3eb5b0f110975547091f4e51854e \ + --hash=sha256:8e26a075fa9945b9e44a3d02cc83d776c3b76bb1ff4b133bbfa620d5650131da \ + --hash=sha256:91b916d495db3e1b473c7c8e68733beec4dce8e487442db61764fff94f59740e \ + --hash=sha256:948dfc62683a6947b9b486c4598d8f6e3ecc542478b6767b87d52be68aeb55c6 \ + --hash=sha256:982d07727c809b42a3968785354f11c3728414e4e90af0754345b431b2c32561 \ + --hash=sha256:9a094ed44a22f9da497453137c3118b531fd783866ab524b0b0fc146e7395e1d \ + --hash=sha256:9cd5b6805396157b4cf993a6940cbb8663161f29b4df2458c1c9991f099299c5 \ + --hash=sha256:9d028d189d8f38d7ff292f22187c0df37f2317f554d2ed9a2908ada330af57c0 \ + --hash=sha256:9dc55698737aca028848bde418d6c51d74f2a5fd44872d3c8b56b626729adb89 \ + --hash=sha256:9e9aaef25a40d1f1e1bbb1d0eb0190c4a64a7a1750f7eb67b8399bed6f4fd2a6 \ + --hash=sha256:a2d6d30be35ddd70ce0f8ee259a4c25f24d6d689a45a5ac440f03e6bcc5a21d1 \ + --hash=sha256:a68b637451d64ba30ed8ae125c973fa834cc2d37dfa7f154c2b479015d477ba8 \ + --hash=sha256:a72ecf5bfd3fc8d57927f7e3ded2487e144472f39010c3acaec3f6f3ff53f361 \ + --hash=sha256:aa34473fbcc108fea403074f3f45091461b18b2047d136f16ffaa4c65ad46a68 \ + --hash=sha256:ab2fb1f7a2deb4ca3ddebbae6b93905d21480a3b4e11de28d79d9fb0d316fcf8 \ + --hash=sha256:ab39d2c967aae3b48a412bff9cdbe7cd7559cd1e277599aceaeada7bc82b7200 \ + --hash=sha256:b04583e8867136ae66353fa274f45121ab3ec3166dc45aaff3655a5db90d9f0e \ + --hash=sha256:b1963ec5ba4d52788fb0eac6aca6eb8040e8e318c7e47ebbdfc09440c802919c \ + --hash=sha256:b56416091bfd7a429f958f69aaf6823c517be9a49cb5bf1daa3767ce8bf8095e \ + --hash=sha256:b862572b7a5f5ed47d2ba5921e63bf8d9e3b682f859d8f11e0e5ca46f7e82173 \ + --hash=sha256:b96341cb29a3faa5db05aff29c77d141d827414f145330e5d8846892119351c1 \ + --hash=sha256:bb52e10e453b5493afe1f7702a2973bc10f4dd8901c0f2ed869ffaa3f8319296 \ + --hash=sha256:bb5aab464a0c5e03a97abad5bdf54517061ebbf72340d576e99ff661a42575cc \ + --hash=sha256:be4223af640d0aa04c05db81d5d96ada3ead9c09187d892fd37f4f97829480be \ + --hash=sha256:c2cbd385d82f63bb35edb60b09b08abad3619bd0a4a492ae59e55afaf98e1b9d \ + --hash=sha256:c57b6ad3f7a1bdd101b2966f29dc161adf49727b1e8d3e1e89db2eda8a75c344 \ + --hash=sha256:c622f4c638a725c39abcb2e680b1bd592663c83b672a4ed350a17f806d75618e \ + --hash=sha256:cae27622c094558e519abf3242cf4272db961d12c5c9a9ffb7a1b44b2627d5c6 \ + --hash=sha256:cfcec18f7da682c4e2d82112829ce906569cb8d69fa6c26f3a50dfbed5ceb682 \ + --hash=sha256:cfeb11990f59e59a0df26c648f0adfcbf27be77241250636f5769eb08db662be \ + --hash=sha256:d0834c84ae8750ae1c4cede59b0afd4d2f775be958e11b18a3eea24ed9d0d9f1 \ + --hash=sha256:d3c75d57a00109255e60bc9c623b6ececaf7905eaab845c79f036670ed4750a2 \ + --hash=sha256:d3e10779f60c000213a5b53f518824bd07b3dc119333b26d70c6be1c27b5c794 \ + --hash=sha256:d50714405845c1010c871098558cfe5718fe39d2a2fab5f95c8863caeb7a82b3 \ + --hash=sha256:da6ef4cb8d457aab0482b50120136ae94238aaa421863eaa7d599759742c72d6 \ + --hash=sha256:dd3b6d97beb39afb412f2c79522b9e099463c31f4c49ab8347c5a2ca3531c478 \ + --hash=sha256:dd7715817a187edd7e2a2390908757f7ba42148e59cad755fb8ee1160c628eca \ + --hash=sha256:e21e888a6b471b2bb1cdd4247e8d86632672232f29be583e7eafaa5f4634d34c \ + --hash=sha256:e37aba1994d73b4944053ab65a15f313bd5c28c885dd7f0d494a11749d89db6e \ + --hash=sha256:e54e088dc64dd2766014e7cfe5f8bc45399400fd486816e494f93e3f0f55da06 \ + --hash=sha256:e6b6a11bf898cca3ce7bfaa17b646901107f3975677fbd5097f36e5eb5641983 \ + --hash=sha256:eac1207936555aa691ce32df1432b478f2729d54e6d93a1f4db9215bcd8eb47d \ + --hash=sha256:ebbf0d83ed5271991d666e54bb6c90ac2c55fb2ef3a88740c6af85dc85de2402 \ + --hash=sha256:ec1c44cf9bd22079aac37a07cb49a29ced9050ab5bddf24e50aba298f1e34d90 \ + --hash=sha256:ecae626449d00db8c08f8f1fc00047a32d6d7eb5402b3976f5c3fda2b80a7a4f \ + --hash=sha256:ed7c886a2fcbf14493ceaf9579394b33521730c161ebb8dad7db9c3e9fcab1a8 \ + --hash=sha256:ee877b6d78f9dff1da94fef51ae8cf9cce0967e043fdcc864c40b85cf293c192 \ + --hash=sha256:f0192e5f1cfc70e3cb35347135dd02e7497b3e7d83e378aa226d8b3e53a93f19 \ + --hash=sha256:f3463a5f26be513a49e4d497debcf1b252a2db7b92c77d89621aa90b83d2dd38 \ + --hash=sha256:f6222cafe00e072bb2b8f14142cd969637411fbc4dd3b1d73a90a3b817fa046f \ + --hash=sha256:f988a1cec68058f71a38471813fba9e87dffe855582682e8a10e40ece12567a2 \ + --hash=sha256:fadb07dbe36a541283ff454b1a268afd54b077d917043f2e1e5615372cb5f200 \ + --hash=sha256:fe7ff456c22725c9d9017f7a2a7df2b51af6df77314176760b22e2d05278e181 + # via segtok +requests==2.34.2 \ + --hash=sha256:2a0d60c172f83ac6ab31e4554906c0f3b3588d37b5cb939b1c061f4907e278e0 \ + --hash=sha256:f288924cae4e29463698d6d60bc6a4da69c89185ad1e0bcc4104f584e960b9ed + # via tht (harness/pyproject.toml) +rich==15.0.0 \ + --hash=sha256:33bd4ef74232fb73fe9279a257718407f169c09b78a87ad3d296f548e27de0bb \ + --hash=sha256:edd07a4824c6b40189fb7ac9bc4c52536e9780fbbfbddf6f1e2502c31b068c36 + # via + # tht (harness/pyproject.toml) + # typer +s3transfer==0.19.1 \ + --hash=sha256:d3d6371dc3f1e5c5427b2b457bcf13bcf87bec334c95aed18642eae61f6926f3 \ + --hash=sha256:d5fd7005ee39307455ad5f310b5ea67f4b1960d7fed5b3671ee50c249de675de + # via boto3 +scipy==1.18.0 \ + --hash=sha256:09143f676d157d9f546d663504ef9c1becb819824f1afc018814176411942446 \ + --hash=sha256:0d13bca67c096d89fb95ced0d8921807300fce0275643aef9533cc63a0773468 \ + --hash=sha256:18e9575f1569b2c54174e6159d32942e03731177f63dce7975f0a0c88d102f5b \ + --hash=sha256:1a4441f15d620578772a49e5ab48c0ee1f7a0220e387110283062729136b2553 \ + --hash=sha256:1ad44305cfa24b1ba5803cbbebf033590ccbac1aa5d612d727b785325ab408b0 \ + --hash=sha256:1afac4a847207c7ff8efd321734a50b06d0280b3b2a2c0fc2f413101747ad7c7 \ + --hash=sha256:1f55797419e16e7f30cf88ffb3113ce0467f00cfe3f70d5c281730b21769bfc2 \ + --hash=sha256:265915e79107de9f946b855e50d7470d5893ec3f54b342e1aa6201cbdcd8bb6b \ + --hash=sha256:2d8bbdc6c817f5b4006a54d799d4f5bab6f910193cbb9a1ff310833d4d270f61 \ + --hash=sha256:2ef3abc54a4ffc53765374b0d5728532dfdd2585ed23f6b11c206a1f0b1b9af8 \ + --hash=sha256:368e0a705903c466aa5f08eefb39e6b1b6b2d659e7352a31fd9e2438365be0f8 \ + --hash=sha256:3f1ac564d3bf6c03d861d2cd87a1bea0da2887136f7fb1bf519c05a8971452d6 \ + --hash=sha256:40395a5fcd1abee49a5c7aaa98c29db393eedc835138560a588c47ec16156690 \ + --hash=sha256:4a55985d54c769c872e64b7f4c8a81cc30ef700cc04296abbbf3705439c126de \ + --hash=sha256:4c256ee70c0d1a8a2ace807e199ccd4e3f57037433842abb3fb36bc17eaa9578 \ + --hash=sha256:52a96e21517c7292375c0e27dd796a811f03fcea5fd4d108fdfea8145dcf17ab \ + --hash=sha256:56abf29a7c067dde59be8b9a22d606a4ea1b2f2a4b756d9d903c62818f5dacce \ + --hash=sha256:5aba46108853ddfc77906b6557aac839d2b52e900c1d72a1180adaaab58d265f \ + --hash=sha256:5efe260f69417b97ddae455bfb5a95e8359f7f66ad7fa9522a60feb66f169520 \ + --hash=sha256:67b2ad2ad54c72ca6d04975a9b2df8c3638c34ddd5b28738e94fc2b57929d378 \ + --hash=sha256:68363b7eaacd8b5dd426df56d782cc156468ac79a127a1b87ca597d6e2e82197 \ + --hash=sha256:6aa94e78ec192a30063a5e72e561c28af769dc311190b24fe91774eff1969709 \ + --hash=sha256:71ccc8faa2dd16ac310233203474a8b5cb67f10dedd54a3116d34943f4b19132 \ + --hash=sha256:7a7f3b01647384dbc3a711e8c6778e0aabbe93959249fef5c7393396bcac0867 \ + --hash=sha256:7bd21faaf5a1a3b2eff922d02db5f191b99a6518db9078a8fb23169f6d22259a \ + --hash=sha256:7c7a51b33ce387193c97f228320cf8e87361daa1bba750638677729598b3e677 \ + --hash=sha256:84031d7b052a54fae2f8632e0ec802073d385476eb9a63079bce6e23ef9283d4 \ + --hash=sha256:8ca01e8ae69f1b18e9a58d91afead31be3cef0dd905a10249dac559ee15460a0 \ + --hash=sha256:945c1761b93f38d7f99ae81ae80c63e621471608c7eeead563f6df025585cd58 \ + --hash=sha256:97b6cddaaee0a779ef6b5ca83c9604b27cc16b2b8fc22c142652df8793319fb8 \ + --hash=sha256:9aac6192fac56bf2ca534389d24623f07b39ff83317d58287285e7fbd622ff76 \ + --hash=sha256:9ab7b758be6940954a713ee466e2043e9f6e2ed965c1fce5c91039f4be3d90a9 \ + --hash=sha256:a46f9273dbd0eb1cefba61c9b8648b4dfe3cbc14a080176f9a73e44b8336dc7f \ + --hash=sha256:ad033410e2e0672ffdc1042110cef20e1c46f8fd0616cee1d44d8d58fad8fc11 \ + --hash=sha256:b6f758e35f12757b5d95c00bc6de2438e229c2664b7a92e96f205959d9f2dfa4 \ + --hash=sha256:c5557d8be5da8e41353fcd4d21491fdbab83b062fc579e94dc09a7c8ab4f669b \ + --hash=sha256:c5dbddf60e58c2312316d097271a8e73d40eaf2eabfa4d95ed7d3695bbf2ce7b \ + --hash=sha256:d88363fd9d8fbd3511bd273f1a49efb2a540773ddf92a91d57498ce7dd7f3e76 \ + --hash=sha256:e40baea28ae7f5475c779741e2d90b1247c78531207b49c7030e698ff81cee3f \ + --hash=sha256:f2a6af57bd9e4a75d70e4117e78a1bbee84f79ae3fbb6d0111005d6ebcc4cb8d \ + --hash=sha256:f351e0dd702687d12a402b867a1b4146a256923e1c38317cbc472f6372b94707 + # via datasketch +segtok==1.5.11 \ + --hash=sha256:8ab2dd44245bcbfec25b575dc4618473bbdf2af8c2649698cd5a370f42f3db23 \ + --hash=sha256:910616b76198c3141b2772df530270d3b706e42ae69a5b30ef115c7bd5d1501a + # via yake +setuptools==80.9.0 \ + --hash=sha256:062d34222ad13e0cc312a4c02d73f059e86a4acbfbdea8f8f76b28c99f306922 \ + --hash=sha256:f36b47402ecde768dbfafc46e8e4207b4360c654f1f3bb84475f0a28628fb19c + # via -r docker/python-runtime/build-requirements.in +shellingham==1.5.4 \ + --hash=sha256:7ecfff8f2fd72616f7481040475a65b2bf8af90a56c89140852d1120324e8686 \ + --hash=sha256:8dbca0739d487e5bd35ab3ca4b36e11c4078f3a234bfce294b0a0291363404de + # via typer +six==1.17.0 \ + --hash=sha256:4721f391ed90541fddacab5acf947aa0d3dc7d27b2e1e8eda2be8970586c3274 \ + --hash=sha256:ff70335d468e7eb6ec65b95b99d3a2836546063f63acc5171de367e834932a81 + # via python-dateutil +sqlalchemy==2.0.51 \ + --hash=sha256:0378d055e9e8cd6ce4d8dff683bdd3d7d413533c4ee51d67a2b1e0f9eacc0f23 \ + --hash=sha256:0592bdadf86ddcabfd72d9ab66ea8a5d8d2cc6be1cc51fa7e66c03868ac5eac1 \ + --hash=sha256:08a204d8b5638717c26a24df18fcf40af45a6b22e35b70b1d62f0113c2e278e8 \ + --hash=sha256:0c2c62877097e1a0db401fba5cb4debee33265e5b2a55c4ccb489c02c53b4f72 \ + --hash=sha256:0e8203d2fbd5c6254692ef0a72c740d75b2f3c7ca345404f4c1a4604813c77c0 \ + --hash=sha256:0f053118c30e53161857a953e4de667d90e274980dccbe5dd3829bbbeece72a5 \ + --hash=sha256:0f6bcad487aee1c638d707235682fc96f741de00663619881ab235400d03289e \ + --hash=sha256:111604e637da87031255ddc26c7d7bc22bc6af6f5d459ccff3af1b4660233a85 \ + --hash=sha256:1181256e0f16479691b5616d36375dc2620ad8332b25978763c3d206ad3f3f1d \ + --hash=sha256:159bb6ba32059f57ad7375a8f50d844dd2f19d14954ecf820cd33e20debd46b2 \ + --hash=sha256:1aa10c0daee6705294d181daadaa793221e1a59ed55000a3fab1d42b088ce4ba \ + --hash=sha256:1af05726b3d0cdba1c55284bf408fd3b792e690fe2399bfb8304565551cda652 \ + --hash=sha256:1bed1ee8b01da6088210aa9412023326fb98a599ba502e6118308601dcbef77f \ + --hash=sha256:1d21ce524ab86c23046e992a5b81cb54c21079c6df6e78b8fc77d77cac70a6b9 \ + --hash=sha256:1e47b1199c2e832e325eacabc8d32d2487f58c9358f97e9a00f5eb93c5680d84 \ + --hash=sha256:247acaa29ccef6250dfd6a3eedf8f94ddf23564180a39fe362e32ae9dbdbde46 \ + --hash=sha256:2a97eaad21c84b4ef8010b11eeba9fe6153eb0b3df3ff8b6abc309df1b978ef7 \ + --hash=sha256:2cf39aabdf48e87c1c2c2ed6d20d33ffa0733b3071ce9c5f66357947dd009080 \ + --hash=sha256:2e54ff2dd657f2e3e0fbf2b097db1182f7bfea263eca4353f00065bae2a67c3d \ + --hash=sha256:39a76529db6305693d8d4affa58ad5b5e2e18edd62daea628b29b97930b3513d \ + --hash=sha256:4004ada0aafe8ae1991b2cd1d99c6d9146126e123bd6f883c260d974aa012e54 \ + --hash=sha256:436728ce18a80f6951a1e11cc6112c2ede9faf20766f1a26195a7c441ca12dbd \ + --hash=sha256:483b11bd46bf35fc14c52faf338b04300c9e6ce554bce9b11be85bfec3bc3195 \ + --hash=sha256:4a011ea4510683319ce4ed274b56ee05194b39b6da9d09ca7a39388f0fa84dcc \ + --hash=sha256:581921d849d6e6f994d560389192955e80e2950e18fcdfe2ccea863e01158e6e \ + --hash=sha256:59cab3686b1bc039dd9cded2f8d0c08a246e84e76bd4ab5b4f18c7cdae293825 \ + --hash=sha256:6b588fd681ddf0c196b8df1ea49a8913514894b2b8f945a9511b4b48871f99c8 \ + --hash=sha256:6e46fc36029eff666391e0531e5387b62ce6c4f1d8e50b3fb3099eaca1b42522 \ + --hash=sha256:6ea306caaae6bd5afd0a46050003c88f6bf33227377a49298c498c3cb88ff491 \ + --hash=sha256:72ca54c952107ba5cd58854b67a5a6268631289d21651a1235396f3b98b47400 \ + --hash=sha256:740cf6f35351b1ac3d82369152acf1d51d37e3dcf85d4dc0a22ca01410eabe2a \ + --hash=sha256:7c2056838b6685b72fdb36c99996cf862753461a62f2e84f4196371d3b2d6a07 \ + --hash=sha256:7c6b36ed71f41942bdcd2ad2522be46bfce09d5705be5640ecf19bbc7660e4b7 \ + --hash=sha256:7d78702b26ba1c18b2d0fb2ea940ba7f17a9581b42e8361ff93920ebbee1235a \ + --hash=sha256:804dccd8a4a6242c4e30ad961e540e18a588f6527202f2d6791b01845d59fdc9 \ + --hash=sha256:9161cfc9efce70d1715f47d6ff40f79c6778c00d53be4fbc09d70301e4b83ba7 \ + --hash=sha256:96747bfbadb055466e5b46d572618170046b45ce5a4879167f50d70a5319a499 \ + --hash=sha256:9f380393be5abeb6815f68fd39271b95127173511b6706b0a630a9995d53f8f5 \ + --hash=sha256:a42ad6afcbaaa777241e347aa2e29155993045a0d6b7db74da61053ffe875fe0 \ + --hash=sha256:a5b2ed6d828f1f09bd812861f4f59ca3bc3803f9df871f4555187f0faf018604 \ + --hash=sha256:a6d26094615306d116dd5e4a51b0304c99dd2356fc569eed6922a80a6bd3b265 \ + --hash=sha256:aa18ae738b5170e253ad0bb6c4b0f07585081e8a6e50893e4d911d47b39a0904 \ + --hash=sha256:ad30ae663711786303fbcd46a47516302d201ee49a877cb3fac61f672895110a \ + --hash=sha256:b21f0e7efc7a5c509e953784e9d1575ebb8b4318960e7e7d7a93bb803626cf64 \ + --hash=sha256:b3e693d15533a45cd5906f0589f9c35090bef6ef45bf1e8195c424aa0ae06a8d \ + --hash=sha256:b7f08588854bbb724041d9ae9d980d40040c922382e1d9a2ecb390edc4fd5032 \ + --hash=sha256:b93ab07b5292dbe7e6b8da89475275e7042744283921344b56105f3eeb0f828b \ + --hash=sha256:bb024d8b621d0be75f4f44ecc7c950450026e76d66dc8f791bb5331d7fed59d5 \ + --hash=sha256:bb1f5062f98b0b3290e72b707747fdd7e0f22d6956b236ba7ca7f5c9971d2da2 \ + --hash=sha256:c45a496d6bc05dec41dcd4c3a2b183723f47473255c159cd80b503c8f246424d \ + --hash=sha256:c5d98a2709840027f5a347c3af0a7c3d5f6c1ff93af2ca1c54494e23cba8f389 \ + --hash=sha256:c68568f3facf8f66fa76c60e0ced69b67666ffa9941d1d0a3756fda196049080 \ + --hash=sha256:c95ef01f53233a305a874a44a63fbfb1d81cd79b49de0f8529b3548cde437e37 \ + --hash=sha256:ca216e8af5c05e326efc7e28716ac2381a7cf9791749f5ee1849dccdc99c9b00 \ + --hash=sha256:ca8435d13829b92f4a97362d91975154a4015db3a2634154e1754e9a915e6b86 \ + --hash=sha256:dc261707bf5739aea8a541593f3cc1d463c2701fb05fbcbba0ce031b69a21260 \ + --hash=sha256:e5ea1a213be1fcd5e49d9904c3b9939211ded90bc2a64e93f4c01963474285de \ + --hash=sha256:fa268106c8987639a17a18514cfe0cd9bf17420ab887e1e1bf486da8836135b1 + # via tht (harness/pyproject.toml) +sqlglot==30.12.0 \ + --hash=sha256:6b8369704662d4f654bc934cea4dd31c916c2a571b389210cb9e951a275e5fd9 \ + --hash=sha256:86cccc610073c645c03e72b55b60ae0518aa3253a7fc3bd56551370d003c6554 + # via tht (harness/pyproject.toml) +tabulate==0.10.0 \ + --hash=sha256:e2cfde8f79420f6deeffdeda9aaec3b6bc5abce947655d17ac662b126e48a60d \ + --hash=sha256:f0b0622e567335c8fabaaa659f1b33bcb6ddfe2e496071b743aa113f8774f2d3 + # via yake +tqdm==4.68.4 \ + --hash=sha256:19829c9673638f2a0b8617da4cdcb927e831cd88bcfcb6e78d42a4d1af131520 \ + --hash=sha256:5168118b2368f48c561afda8020fd79195b1bdb0bdf8086b88442c267a315dc2 + # via tht (harness/pyproject.toml) +typer==0.26.8 \ + --hash=sha256:3512ca79ac5c11113414b36e80281b872884477722440691c89d1112e321a49c \ + --hash=sha256:c244a6bd558886fe3f8780efb6bdd28bb9aff005a94eedebaa5cb32926fe2f7e + # via tht (harness/pyproject.toml) +typing-extensions==4.16.0 \ + --hash=sha256:481caa481374e813c1b176ada14e97f1f67a4539ce9cfeb3f350d78d6370c2e8 \ + --hash=sha256:dc983d19a509c94dba722ee6abd33940f7c05a89e243c47e907eb4db6f1a43e5 + # via + # pydantic + # pydantic-core + # sqlalchemy + # typing-inspection +typing-inspection==0.4.2 \ + --hash=sha256:4ed1cacbdc298c220f1bd249ed5287caa16f34d44ef4e9c3d0cbad5b521545e7 \ + --hash=sha256:ba561c48a67c5958007083d386c3295464928b01faa735ab8547c5692e87f464 + # via pydantic +urllib3==2.7.0 \ + --hash=sha256:231e0ec3b63ceb14667c67be60f2f2c40a518cb38b03af60abc813da26505f4c \ + --hash=sha256:9fb4c81ebbb1ce9531cce37674bbc6f1360472bc18ca9a553ede278ef7276897 + # via + # botocore + # requests +yake==0.7.3 \ + --hash=sha256:38f7f135ff8ed4bcdc05e16b533a9dc93299f1e694b0c308c3c086bab316c5fe \ + --hash=sha256:8778fb2832e58d26d838d6d7ac967b4947521f1fe8cdf23dd872636161fc53ed + # via tht (harness/pyproject.toml) diff --git a/docker/smoke/core-smoke.sh b/docker/smoke/core-smoke.sh new file mode 100755 index 00000000..77c968d7 --- /dev/null +++ b/docker/smoke/core-smoke.sh @@ -0,0 +1,36 @@ +#!/bin/sh +set -eu + +test "$(id -u)" != "0" + +node_version="$(node --version)" +python_version="$(python --version 2>&1)" +case "$node_version" in + v22.19.*|v22.2[0-9].*|v2[3-9].*|v[3-9][0-9].*) ;; + *) echo "Node 22.19+ required, found $node_version" >&2; exit 1 ;; +esac +case "$python_version" in + "Python 3.1"[1-9].*|"Python 3."[2-9][0-9].*) ;; + *) echo "Python 3.11+ required, found $python_version" >&2; exit 1 ;; +esac + +tht --help >/dev/null +pi --version >/dev/null + +/app/docker/core-entrypoint.sh server & +server_pid=$! +trap 'kill "$server_pid" 2>/dev/null || true; wait "$server_pid" 2>/dev/null || true' EXIT INT TERM + +attempt=0 +until curl --fail --silent --show-error http://127.0.0.1:8787/health >/dev/null; do + attempt=$((attempt + 1)) + if [ "$attempt" -ge 30 ]; then + echo "backend health check did not become ready" >&2 + exit 1 + fi + sleep 1 +done + +echo "$node_version" +echo "$python_version" +echo "core smoke: ok" diff --git a/docker/smoke/frontend-policy-smoke.sh b/docker/smoke/frontend-policy-smoke.sh new file mode 100755 index 00000000..309009d4 --- /dev/null +++ b/docker/smoke/frontend-policy-smoke.sh @@ -0,0 +1,21 @@ +#!/bin/sh +set -eu + +corpus=/etc/thothii/backend-url-cases.json + +jq -c '.[]' "$corpus" | while IFS= read -r case_json; do + value=$(printf '%s' "$case_json" | jq -r '.value') + valid=$(printf '%s' "$case_json" | jq -r '.valid') + if BACKEND_BASE_URL="$value" /usr/local/bin/frontend-entrypoint true \ + >/dev/null 2>&1; then + actual=true + else + actual=false + fi + if [ "$actual" != "$valid" ]; then + echo "entrypoint policy mismatch for BACKEND_BASE_URL=$value: expected $valid" >&2 + exit 1 + fi +done + +echo "frontend entrypoint canonical URL corpus: ok" diff --git a/docker/smoke/frontend-smoke.sh b/docker/smoke/frontend-smoke.sh new file mode 100644 index 00000000..0a159f9b --- /dev/null +++ b/docker/smoke/frontend-smoke.sh @@ -0,0 +1,14 @@ +#!/bin/sh +set -eu + +assignment=$(sed \ + -e 's/^window\.__THOTHII_CONFIG__ = //' \ + -e 's/;$//' \ + /usr/share/nginx/html/config.js) + +printf '%s\n' "$assignment" \ + | jq -e --arg expected "${BACKEND_BASE_URL-/api}" \ + 'type == "object" and keys == ["backendBaseUrl"] and .backendBaseUrl == $expected' \ + >/dev/null + +printf '%s\n' "frontend runtime config smoke: ok" diff --git a/docker/validate-backend-url.sh b/docker/validate-backend-url.sh new file mode 100755 index 00000000..374e581b --- /dev/null +++ b/docker/validate-backend-url.sh @@ -0,0 +1,30 @@ +#!/bin/sh +set -eu + +value=${1-} +policy_file=${BACKEND_URL_POLICY_FILE:-/etc/thothii/backend-url-policy.json} + +if jq -e --arg value "$value" '.relativeBases | index($value) != null' \ + "$policy_file" >/dev/null; then + exit 0 +fi + +if ! jq -e --arg value "$value" \ + '.absolutePattern as $pattern | $value | test($pattern)' \ + "$policy_file" >/dev/null; then + exit 2 +fi + +authority=${value#*://} +authority=${authority%%/*} +port="" +case "$authority" in + *]:*) port=${authority##*:} ;; + *]) ;; + *:*) port=${authority##*:} ;; +esac + +if [ -n "$port" ]; then + max_port=$(jq -r '.maxPort' "$policy_file") + if [ "${#port}" -gt 5 ] || [ "$port" -gt "$max_port" ]; then exit 2; fi +fi diff --git a/docs/general/pi-configuration.md b/docs/general/pi-configuration.md index 0f7b2770..ecc2606b 100644 --- a/docs/general/pi-configuration.md +++ b/docs/general/pi-configuration.md @@ -2,6 +2,29 @@ Pi (il coding agent che orchestra il workflow NL→SQL) può risolvere un `provider/model` in tre modi diversi. Non sono alternativi: coesistono, e la scelta di quale usare dipende da **quanto è standard l'endpoint** e da **quanto deve essere ampia la visibilità** del modello (tutti i progetti vs. un progetto solo). +## Credenziali nel backend container + +In produzione configurare una sola sorgente generica, `THT_MODEL_API_KEY_FILE`, come secret file +assoluto e non il valore della chiave. `PiProcessManager` rilegge e valida il file per ogni processo, +normalizza il provider selezionato e passa al solo child Pi la variabile nativa appropriata +(`ANTHROPIC_API_KEY`, `OPENAI_API_KEY`, `GEMINI_API_KEY`, `ZAI_API_KEY`, ecc.). Il percorso generico, +le chiavi di provider non selezionati e il vecchio `PI_PROVIDER_API_KEY` vengono rimossi dall'ambiente +del child. Provider locali come `ollama`, `lmstudio` e `aritmolab` continuano senza chiave; un provider +hosted non mappato o un secret mancante/non sicuro fallisce prima dello spawn con errore sanitizzato. + +La sorgente generica supporta soltanto provider con una singola chiave: `ant-ling`, `anthropic`, +`cerebras`, `deepseek`, `fireworks`, `github-copilot`, `google` (anche tramite alias `gemini`), +`google-vertex` in modalità API key, `groq`, `huggingface`, `kimi-coding`, `minimax`, `minimax-cn`, +`mistral`, `moonshotai`, `moonshotai-cn`, `nvidia`, `openai`, `opencode`, `opencode-go`, +`openrouter`, `together`, `vercel-ai-gateway`, `xai`, i quattro provider `xiaomi*`, `zai` e +`zai-coding-cn`. + +I provider composti `amazon-bedrock`, `azure-openai-responses`, `cloudflare-workers-ai` e +`cloudflare-ai-gateway` non sono rappresentabili da un solo file. La selezione fallisce prima +dello spawn (anche durante l'elenco modelli); tutte le credenziali ambientali AWS, Azure e +Cloudflare restano comunque rimosse. Servirà una futura configurazione dedicata per provider per +supportare questi bundle senza ambiguità. + ## I tre livelli di provenienza di un modello ### 1. Built-in (compilato dentro Pi) diff --git a/docs/index.md b/docs/index.md index 99cfa341..fa2a74c7 100644 --- a/docs/index.md +++ b/docs/index.md @@ -8,6 +8,10 @@ La documentazione è divisa in due aree: Come funziona il sistema: architettura, specifiche di design delle singole funzionalità, piani di implementazione, report di test. Parte da qui: [Panoramica dell'architettura](architecture/overview.md). +Per installare l'applicazione in Docker nei quattro contesti operativi, partendo dal comando +predefinito `docker compose up --build -d` e dal bundle unico dei secret: +[Installazione Docker nei quattro contesti](installazione-docker-4-contesti.md). + ## Considerazioni Generali Note operative e di configurazione che non sono specifiche del dominio ThothII ma riguardano l'ambiente di sviluppo condiviso con altri progetti — ad esempio come Pi (il coding agent) risolve i modelli a livello built-in, utente e progetto. Parte da qui: [Configurazione dei modelli in Pi](general/pi-configuration.md). diff --git a/docs/installazione-docker-4-contesti.md b/docs/installazione-docker-4-contesti.md new file mode 100644 index 00000000..b268bb4f --- /dev/null +++ b/docs/installazione-docker-4-contesti.md @@ -0,0 +1,234 @@ +# Installazione Docker nei quattro contesti operativi + +ThothII viene distribuito con due immagini applicative: + +- `thothii-core`: backend Fastify, harness `tht` e Pi; +- `thothii-frontend`: frontend React servito da nginx. + +PostgreSQL/pgvector, DWH ed Evidence restano esterni nel profilo predefinito. Il profilo opzionale `local-vector` avvia PostgreSQL/pgvector nel progetto Compose. + +## Installazione comune (il comando standard) + +Servono Docker Engine/Compose v2 su Linux oppure Docker Desktop su macOS/Windows. Dalla directory in cui si vuole conservare il clone: + +```sh +git clone ThothII +cd ThothII +cp .env.example .env +mkdir -p deploy/secrets deploy/workspaces +cp deploy/secrets/thothii.secrets.example deploy/secrets/thothii.secrets +chmod 600 deploy/secrets/thothii.secrets +``` + +Modificare **solo** questi file interni al clone: + +| File | Cosa contiene | +|---|---| +| `.env` | endpoint, database, provider, `COMPOSE_FILE` e `COMPOSE_PROFILES`; mai password/token | +| `deploy/secrets/thothii.secrets` | un bundle `NOME=VALORE`, mode host `0600` o `0400` | +| `deploy/workspaces/.yaml` | adapter, endpoint non riservati, `roots` ed Evidence | + +Il file `.env` viene caricato automaticamente da Docker Compose perché è nella radice del progetto. Il valore predefinito è `COMPOSE_FILE=compose.yaml`, con profili vuoti e `THT_SECRETS_FILE=deploy/secrets/thothii.secrets`. Perciò, dopo aver compilato `.env`, il bundle e almeno il workspace, l'avvio normale è sempre: + +```sh +docker compose up --build -d +``` + +Non occorre usare `--env-file`, `-f` o `--profile` per questa installazione. Verificare lo stato con `docker compose ps` e aprire . `docker compose down` conserva il volume `thoth_data`; usare `down --volumes` solo per un ambiente effimero. + +### Formato del bundle unico + +`deploy/secrets/thothii.secrets` è un file di testo locale, non uno script shell. Sono ammessi commenti e righe vuote; ogni altra riga deve essere una sola assegnazione senza spazi: + +```dotenv +THT_MODEL_API_KEY=... +THT_DWH_API_KEY=... +THT_VEC_API_KEY=... +THT_VEC_WRITE_API_KEY=... +THT_VECTOR_BOOTSTRAP_PASSWORD=... +THT_VECTOR_MIGRATOR_PASSWORD=... +THT_VECTOR_READER_PASSWORD=... +THT_VECTOR_WRITER_PASSWORD=... +``` + +Inserire solo le chiavi necessarie al profilo scelto. Il bundle viene montato in sola lettura nel container come `/run/secrets/thothii.secrets`; il parser rifiuta duplicati, chiavi sconosciute, valori vuoti, symlink e permessi host troppo aperti. Non inserire secret in `.env`, nei workspace, negli URL o nell'output di `docker compose config`. + +Una catena CA PEM **non può essere inserita nel bundle**: contiene whitespace e viene rifiutata dal parser. Se un endpoint usa una CA privata, conservarla nel secret manager/host e aggiungere un override Compose revisionato che monti il file in `/run/secrets/ca-chain.pem` e imposti `THT_SSL_CA` (o il parametro dell'adapter). Il clone base non crea quel mount: questa è una limitazione intenzionale da considerare in fase di deployment. + +### Overlay opzionali tramite `.env` + +Gli overlay non cambiano il comando operativo. Impostare in `.env`: + +```dotenv +# DWH/vector/embedding remoti (server applicativo o server con i DB): +COMPOSE_FILE=compose.yaml:deploy/compose.production.yaml +COMPOSE_PROFILES= + +# pgvector locale (Mac, Windows o server autonomo): +COMPOSE_FILE=compose.yaml:deploy/compose.local-vector.yaml +COMPOSE_PROFILES=local-vector +``` + +Su Windows usare `;` come separatore di `COMPOSE_FILE`. Per il preprocessing locale aggiungere `deploy/compose.preprocess.yaml:deploy/compose.preprocess-local-vector.yaml` e impostare `COMPOSE_PROFILES=local-vector,preprocess`; poi usare `docker compose run --rm preprocess-evidence` oppure `docker compose run --rm preprocess-dwh`. + +## Workspace, adapter e Evidence + +Il workspace YAML seleziona il trasporto disponibile. Esempio DWH REST e vector DB HTTP: + +```yaml +language: en +dwh: + type: thoth_rest + database: {database: warehouse, schema: datawarehouse} + endpoint: {base_url: https://dwh.example.test} +vectors: + type: thoth_vector_http + reader: {base_url: https://vectors.example.test} + writer: {base_url: https://vectors.example.test} +roots: {artifacts: artifacts, indexes: indexes, sessions: sessions} +evidence: {source_root: /data/source, evidence_dir: evidence} +embeddings: {base_url: https://embeddings.example.test, model: nomodel, dim: 768} +``` + +Esempio con accesso diretto a PostgreSQL e pgvector: + +```yaml +language: en +dwh: + type: postgres_direct + connection: {host: dwh.internal, database: warehouse, schema: public, + user: thoth_reader, password_file: /run/secrets/dwh_password} +vectors: + type: pgvector_direct + reader: {host: vector.internal, database: thoth, schema: vectors, + user: thoth_vector_reader, password_file: /run/secrets/vector_reader_password} + writer: {host: vector.internal, database: thoth, schema: vectors, + user: thoth_vector_writer, password_file: /run/secrets/vector_writer_password} +roots: {artifacts: artifacts, indexes: indexes, sessions: sessions} +``` + +Questo esempio mostra il contratto dell'adapter: i file indicati da `password_file` devono +essere montati da un override Compose approvato. Il profilo base monta soltanto il bundle unico; +per un DWH diretto occorre quindi materializzare il file password dal secret manager e aggiungere +il bind mount/runtime adapter corrispondente. Non inserire la password nel workspace o nell'URL. + +`roots` sono relativi e vengono risolti sotto `/data/workspaces/` nel volume Docker; non inserire path host come `/Users/...` o `C:\\...`. Per Evidence usare una radice filesystem montata in sola lettura oppure l'adapter HTTP/S3 previsto dal workspace. Per HTTP/S3 definire allowlist, limiti di dimensione/paginazione e una politica egress; non mettere token nelle URI. + +## 1. Server remoto insieme ai database e al vector DB + +Usare quando il server Docker è nella stessa rete del DWH e del vector DB (containerizzati o meno). Il file `.env` può restare sul default, senza profili, impostando gli endpoint raggiungibili localmente: + +```dotenv +COMPOSE_FILE=compose.yaml +COMPOSE_PROFILES= +THT_DB_NAME=warehouse +THT_DWH_REST_URL=https://dwh.internal.example +THT_VEC_REST_URL=https://vectors.internal.example +THT_OLLAMA_URL=https://embeddings.internal.example +AUTH_MODE=none +THOTH_PUBLIC_EXPOSURE=false +``` + +Riempire nel bundle le chiavi DWH/vector/model necessarie e avviare: + +```sh +docker compose up --build -d +docker compose exec core /opt/venv/bin/tht doctor --json +``` + +Se si abilita l'overlay production, il proxy autenticato TLS deve essere l'unico listener pubblico +e deve iniettare `X-Authenticated-User`; non esporre direttamente la porta pubblicata da nginx. + +Se il server deve essere raggiungibile da altri host, sostituire `COMPOSE_FILE` con +`compose.yaml:deploy/compose.production.yaml`, configurare il proxy autenticato e impostare +`AUTH_MODE=upstream`/`THOTH_PUBLIC_EXPOSURE=true` come descritto nella sezione di trust boundary. + +## 2. Mac locale + +Installare Docker Desktop e, se usato, Ollama sul Mac. Nel `.env` selezionare il profilo locale: + +```dotenv +COMPOSE_FILE=compose.yaml:deploy/compose.local-vector.yaml +COMPOSE_PROFILES=local-vector +THT_DB_NAME=warehouse +THT_DWH_REST_URL=https://dwh.example.test +THT_OLLAMA_URL=http://host.docker.internal:11434 +THT_DOCS_ROOT=/data/source/evidence +``` + +Nel bundle aggiungere quattro password generate localmente: + +```dotenv +THT_VECTOR_BOOTSTRAP_PASSWORD= +THT_VECTOR_MIGRATOR_PASSWORD= +THT_VECTOR_READER_PASSWORD= +THT_VECTOR_WRITER_PASSWORD= +``` + +Poi eseguire il comando standard `docker compose up --build -d`. Il primo avvio esegue reconciliation dei ruoli e migrazione pgvector. Per preprocessing, impostare il preset indicato sopra e usare `docker compose run --rm preprocess-evidence`/`preprocess-dwh`. + +## 3. PC Windows locale + +Usare Docker Desktop con backend WSL2 e abilitare la condivisione della directory del clone. Modificare `.env` con il separatore Windows: + +```dotenv +COMPOSE_FILE=compose.yaml;deploy/compose.local-vector.yaml +COMPOSE_PROFILES=local-vector +THT_DB_NAME=warehouse +THT_DWH_REST_URL=https://dwh.example.test +THT_OLLAMA_URL=http://host.docker.internal:11434 +THT_DOCS_ROOT=/data/source/evidence +``` + +Creare `deploy/secrets/thothii.secrets` con un editor locale protetto (ACL leggibile solo dall'utente Docker) e le stesse quattro chiavi pgvector del profilo Mac. Non usare `ConvertFrom-SecureString`: il bundle deve contenere il valore in chiaro per il servizio, con accesso limitato al file. Da PowerShell, dalla radice del clone, eseguire: + +```powershell +docker compose up --build -d +docker compose ps +``` + +Se un bind mount viene rifiutato, aggiungere la cartella del repository a Docker Desktop → Settings → Resources → File Sharing. Per Ollama eseguito in WSL2 usare l'indirizzo raggiungibile dalla rete Docker invece di assumere `localhost`. + +## 4. Server applicativo distinto da DB ed Evidence + +Usare il profilo production e consentire dal firewall solo le destinazioni necessarie: + +```dotenv +COMPOSE_FILE=compose.yaml:deploy/compose.production.yaml +COMPOSE_PROFILES= +THT_DB_NAME=warehouse +THT_DWH_REST_URL=https://dwh.example.test +THT_VEC_REST_URL=https://vectors.example.test +THT_OLLAMA_URL=https://embeddings.example.test +``` + +Il DWH e il vector DB possono essere REST/HTTP oppure adapter diretti (`postgres_direct`, `pgvector_direct`) se il server ha connettività TCP. Le Evidence possono essere: + +- filesystem NFS/SMB montato sul server e presentato come root read-only; +- endpoint HTTPS, con allowlist e limiti SSRF; +- bucket S3 con secret references e endpoint custom esplicitamente autorizzati. + +Il preprocessing può girare sul server applicativo usando il volume `/data`; mantenere separati workspace, lock e artefatti dei job. Avviare con il comando standard e verificare `tht doctor`. + +## Migrazione da installazioni con secret separati + +Le variabili `THT_*_SECRET_FILE` e i file `dwh-api-key`, `vector-reader-api-key`, `vector-writer-api-key`, `model-api-key` e `vector_*_password` appartengono al layout precedente. Non vengono importati automaticamente dal bundle. Per migrare: + +1. creare `deploy/secrets/thothii.secrets` mode `0600`; +2. copiare ogni valore nel nome chiave corrispondente (`THT_DWH_API_KEY`, `THT_VEC_API_KEY`, `THT_VEC_WRITE_API_KEY`, `THT_MODEL_API_KEY` o `THT_VECTOR_*_PASSWORD`), senza virgolette né newline; +3. rimuovere dal `.env` le variabili `_SECRET_FILE` e impostare `THT_SECRETS_FILE` al percorso del bundle (il default relativo è già corretto); +4. eseguire `docker compose config --quiet` e poi `docker compose up --build -d`; +5. solo dopo la verifica, cancellare i vecchi file separati. + +Una CA PEM resta un'eccezione esterna come descritto sopra. Provider Pi con credenziali composte (Bedrock, Azure OpenAI Responses, Cloudflare Workers AI/Gateway) restano rifiutati finché non viene implementato un adapter dedicato. + +## Controlli post-installazione + +```sh +docker compose config --quiet +docker compose ps +docker compose exec core /opt/venv/bin/tht doctor --json +./scripts/docker-smoke.sh +``` + +Per il profilo locale usare anche `./scripts/local-vector-smoke.sh`; per il preprocessing `./scripts/preprocess-smoke.sh`. Non pubblicare `.env` o `deploy/secrets/thothii.secrets` nei log, nei backup Git o nei ticket. diff --git a/docs/superpowers/plans/2026-07-11-adapter-foundations.md b/docs/superpowers/plans/2026-07-11-adapter-foundations.md index f5974c6f..3de4f6ae 100644 --- a/docs/superpowers/plans/2026-07-11-adapter-foundations.md +++ b/docs/superpowers/plans/2026-07-11-adapter-foundations.md @@ -26,7 +26,7 @@ - Test: `harness/tests/test_dwh_port_contract.py` **Interfaces:** -- Produces: `DwhCapabilities`, `DwhAdapter`, `DwhHealth`, and `UnsupportedCapability`. +- Produces: `DwhCapabilities`, `DwhAdapter`, `DwhHealth`, `DistinctValues`, and `UnsupportedCapability`. - Consumes: existing catalog models from `tht.db.introspect` and execution result types from `tht.db.execute`. - [ ] **Step 1: Write the failing protocol-shape test** @@ -54,16 +54,21 @@ class DwhCapabilities: sampling: bool = True distinct_values: bool = True +@dataclass(frozen=True) +class DistinctValues: + values: list[object] + truncated: bool + @runtime_checkable class DwhAdapter(Protocol): @property def capabilities(self) -> DwhCapabilities: ... def health(self) -> DwhHealth: ... - def introspect(self) -> DatabaseCatalog: ... - def run_query(self, sql: str, *, limit: int | None = None) -> QueryResult: ... + def introspect(self) -> PhysicalSchema: ... + def run_query(self, sql: str, *, limit: int) -> ExecResult: ... def explain(self, sql: str) -> PlanSummary: ... def sample_column(self, table: str, column: str, *, limit: int) -> list[object]: ... - def distinct_values(self, table: str, column: str) -> list[object]: ... + def distinct_values(self, table: str, column: str) -> DistinctValues: ... ``` - [ ] **Step 4: Run contract test and type-oriented import smoke test** @@ -85,8 +90,15 @@ git commit -m "refactor(dwh): define adapter contract" - Create: `harness/tht/adapters/dwh/postgres.py` - Create: `harness/tht/adapters/dwh/thoth_rest.py` - Test: `harness/tests/test_dwh_adapters.py` +- Test: `harness/tests/test_dwh_port_contract.py` +- Test: `harness/tests/l0/test_db_sampling.py` +- Modify: `harness/tht/ports/__init__.py` +- Modify: `harness/tht/ports/dwh.py` +- Modify: `harness/tht/execute/__init__.py` - Modify: `harness/tht/db/execute.py` +- Modify: `harness/tht/db/sampling.py` - Modify: `harness/tht/rest/execute.py` +- Modify: `docs/superpowers/plans/2026-07-11-adapter-foundations.md` **Interfaces:** - Consumes: `DwhAdapter` from Task 1; existing `DatabaseConfig`, `RestConfig`, catalog, sampling, execute, and explain functions. @@ -97,8 +109,8 @@ git commit -m "refactor(dwh): define adapter contract" ```python @pytest.mark.parametrize("factory", [postgres_factory, rest_factory]) def test_adapter_rejects_write_sql(factory): - with pytest.raises(ReadOnlyViolation): - factory().run_query("delete from fact_sales") + with pytest.raises(ExecutionError): + factory().run_query("delete from fact_sales", limit=10) ``` - [ ] **Step 2: Verify failure** @@ -111,12 +123,18 @@ Expected: FAIL because the adapter classes are absent. ```python class PostgresDwhAdapter: capabilities = DwhCapabilities() - def __init__(self, config: DatabaseConfig): self._config = config - def run_query(self, sql: str, *, limit: int | None = None) -> QueryResult: - return run_query(self._config, sql, limit=limit) + def __init__(self, config: DatabaseConfig): + self._config = config + self._engine = make_engine(config) + def run_query(self, sql: str, *, limit: int) -> ExecResult: + return run_query(self._engine, sql, limit=limit) ``` -Implement the analogous REST wrapper by delegating to `tht.rest.*`; translate transport-specific errors only at the adapter boundary. +Implement the analogous REST wrapper by delegating to `tht.rest.*`; translate transport-specific +errors only at the adapter boundary. Both wrappers delegate frequency-ranked, distinct sampling to +the paired implementations in `tht.db.sampling`. Query and sampling limits must be runtime-positive +integers (booleans and floats are rejected), and `distinct_values` reports any cap through +`DistinctValues.truncated`. - [ ] **Step 4: Run adapter, read-only, sampling, and REST tests** @@ -126,7 +144,11 @@ Expected: PASS; L0 may deselect when Docker is unavailable. - [ ] **Step 5: Commit** ```bash -git add harness/tht/adapters harness/tht/db/execute.py harness/tht/rest/execute.py harness/tests/test_dwh_adapters.py +git add docs/superpowers/plans/2026-07-11-adapter-foundations.md \ + harness/tht/ports harness/tht/adapters/dwh harness/tht/execute/__init__.py \ + harness/tht/db/execute.py harness/tht/db/sampling.py harness/tht/rest/execute.py \ + harness/tests/test_dwh_port_contract.py harness/tests/test_dwh_adapters.py \ + harness/tests/l0/test_db_sampling.py git commit -m "refactor(dwh): adapt direct and REST transports" ``` @@ -141,7 +163,8 @@ git commit -m "refactor(dwh): adapt direct and REST transports" - Modify: `harness/tht/vectorstore/reader.py` **Interfaces:** -- Produces: `VectorStore`, `VectorCapabilities`, `VectorHealth`, `VectorRecord`, `VectorHit`, `ThothHttpVectorStore`. +- Produces: `VectorStore`, `VectorCapabilities`, `VectorHealth`, `VectorRecord`, + `VectorWriteRecord`, `VectorHit`, `ThothHttpVectorStore`. - Preserves: current `VectorRestClient`, `DirectSearcher`, and `RestSearcher` behavior behind wrappers. - [ ] **Step 1: Write read/write capability and dual-credential tests** @@ -171,9 +194,13 @@ class VectorStore(Protocol): def search(self, collections: list[str], embedding: list[float], *, limit: int, kinds: list[str] | None = None) -> list[VectorHit]: ... def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]: ... - def upsert(self, collection: str, records: list[VectorRecord]) -> int: ... + def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int: ... ``` +`VectorWriteRecord` is the transport-neutral write envelope: it contains the canonical +`VectorRecord`, a precomputed embedding, and a content hash. Adapters must preserve +`VectorRecord.metadata` unchanged, including semantic keys named `embedding` or `content_hash`. + - [ ] **Step 4: Run vector regression tests** Run: `cd harness && .venv/bin/pytest tests/test_vector_port_contract.py tests/test_vector_dual_key.py tests/test_search_similar_kinds.py tests/test_memory_save_one.py tests/test_solved_question.py -q` @@ -262,6 +289,18 @@ git commit -m "feat(config): add typed resource schema" - Produces: `build_dwh(cfg: Config) -> DwhAdapter` and `build_vector_store(cfg: Config, *, require_write: bool = False) -> VectorStore`. - Consumes: resource configs from Task 4 and wrappers from Tasks 2-3. +Correction: `DwhAdapter.distinct_values(table, column, *, limit)` requires an explicit +positive limit, and direct DWH construction injects `cfg.execution.statement_timeout_ms`. +Targeted vector writes consume the factory-returned `VectorStore` and pass +`VectorWriteRecord` objects to `upsert`. + +Transitional exception: `build_vector_loader` remains solely for bulk collection sync +(`vector init`/rebuild/index flows). It may still construct the legacy table-scoped writer +directly until `docs/superpowers/plans/2026-07-11-local-pgvector-profile.md` migrates the +local pgvector/vector schema and bulk-sync path. Interactive and targeted writes +(`memory save-one` and solved-question indexing) are not covered by this exception and must +continue through `build_vector_store(..., require_write=True)` and the public vector port. + - [ ] **Step 1: Write exact factory selection and missing-writer tests** ```python diff --git a/docs/superpowers/plans/2026-07-11-container-packaging-portable-storage.md b/docs/superpowers/plans/2026-07-11-container-packaging-portable-storage.md index 45bf4937..9349edf6 100644 --- a/docs/superpowers/plans/2026-07-11-container-packaging-portable-storage.md +++ b/docs/superpowers/plans/2026-07-11-container-packaging-portable-storage.md @@ -236,6 +236,13 @@ git commit -m "build(docker): add runtime-configured frontend image" ### Task 5: Compose external profile and end-to-end smoke gate +> **Final-review security amendment (2026-07-12):** the frontend port binds to `127.0.0.1` by +> default. Public deployment uses an authenticated upstream proxy with `AUTH_MODE=upstream`; +> `THOTH_PUBLIC_EXPOSURE=true` plus `AUTH_MODE=none` is invalid. Local env files are development +> only; production uses read-only Compose secrets. Image gates pin exact tags and multi-platform +> digests and verify both linux/amd64 and linux/arm64 using the shared container verification +> script. + **Files:** - Create: `compose.yaml` - Create: `deploy/env.example` diff --git a/docs/superpowers/plans/2026-07-12-local-and-server-docker-deployment.md b/docs/superpowers/plans/2026-07-12-local-and-server-docker-deployment.md new file mode 100644 index 00000000..f64a5075 --- /dev/null +++ b/docs/superpowers/plans/2026-07-12-local-and-server-docker-deployment.md @@ -0,0 +1,70 @@ +# Local and Server Docker Deployment Implementation Plan + +> **For Codex:** execute this plan in the current isolated worktree; keep runtime credentials out of Git. + +**Goal:** Configure and verify a Docker Desktop deployment using GLM 5.2 and the existing PSD workspace, while retaining a portable server deployment contract. + +**Architecture:** The base Compose file builds two applications and consumes only generic environment values and a Docker secret bundle. A tracked GLM Pi registry is mounted read-only in the core container. A Git-ignored local override supplies Mac-specific PSD workspace and CA mounts; server operators supply equivalent server runtime values separately. + +**Tech Stack:** Docker Compose v2, Node 22, Python 3.12, Pi RPC, Fastify, nginx. + +--- + +### Task 1: Add the non-secret GLM Pi registry + +**Files:** +- Create: `deploy/pi/models.json` +- Modify: `docker/core.Dockerfile` +- Modify: `compose.yaml` +- Test: Compose configuration and Pi model discovery + +1. Define the `zai/glm-5.2` OpenAI-compatible model registry without a credential. +2. Create the Pi user configuration directory in the core image and mount the registry read-only. +3. Verify that `get_available_models` returns `zai/glm-5.2` when the bundle supplies the model key. + +### Task 2: Add generic PSD-compatible runtime templates + +**Files:** +- Create: `deploy/workspaces/psd.yaml.example` +- Create: `deploy/compose.psd-local.yaml.example` +- Modify: `deploy/env.example` +- Modify: `README.md` + +1. Define a relative `/data/workspaces/psd` workspace configuration with external REST DWH/vector adapters. +2. Document required non-secret environment values and the local/server boundary. +3. Keep host paths and credential values out of all tracked files. + +### Task 3: Materialize local runtime configuration securely + +**Files (ignored):** +- Create: `.env` +- Create: `deploy/secrets/thothii.secrets` +- Create: `deploy/compose.psd-local.yaml` +- Create: `deploy/workspaces/psd.yaml` + +1. Transfer only required values from the existing local configuration without writing them to logs. +2. Set `PI_PROVIDER=zai`, `PI_MODEL=glm-5.2`, and the Docker Desktop host gateway for Ollama. +3. Bind-mount the PSD workspace and private CA read-only where appropriate; sessions remain writable. +4. Enforce restricted modes on the secret bundle. + +### Task 4: Build and verify the Docker deployment + +**Commands:** +- `docker compose config --quiet` +- `docker compose build` +- `docker compose up -d` +- health/API/model/session smoke checks + +1. Validate rendered Compose configuration without exposing secrets. +2. Build the core and frontend images. +3. Verify secret mount, core and frontend health, and model listing. +4. Start a PSD session using GLM 5.2 and verify Pi emits a workflow event or gate. +5. Capture sanitized diagnostics and stop only disposable test resources; leave the validated local stack running unless it fails. + +### Task 5: Record the deployment result + +**Files:** +- Modify: `README.md` or deployment documentation + +1. Record the exact local startup command and server-equivalent configuration steps. +2. State verified endpoints, model, and session-start result without secret values. diff --git a/docs/superpowers/plans/2026-07-12-simple-docker-config.md b/docs/superpowers/plans/2026-07-12-simple-docker-config.md new file mode 100644 index 00000000..e205bc57 --- /dev/null +++ b/docs/superpowers/plans/2026-07-12-simple-docker-config.md @@ -0,0 +1,156 @@ +# Simple Docker Configuration Implementation Plan + +> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development to implement this plan task-by-task with review checkpoints. + +**Goal:** Make a fresh ThothII clone runnable with `docker compose up --build -d`, using one +`deploy/secrets/thothii.secrets` bundle while preserving a tested legacy fallback. + +**Architecture:** A strict Python secret-bundle loader becomes the single in-process source of +secret values. Compose mounts the one bundle only where needed; the core converts values to +provider/database runtime interfaces without logging or placing them in argv. The root `.env` +is the default Compose interpolation file and selects the appropriate overlay through +`COMPOSE_FILE`/`COMPOSE_PROFILES`; legacy `THT_*_SECRET_FILE` installations remain supported. + +**Tech Stack:** Docker Compose v2, YAML, Python 3.12/Pydantic, Fastify/TypeScript, shell smoke +tests, pytest, Vitest. + +## Global Constraints + +- The normal command must be exactly `docker compose up --build -d` from `ThothII/`. +- The canonical secret bundle is `deploy/secrets/thothii.secrets`, key/value syntax, mode `0600`, + ignored by Git and excluded from image build contexts. +- Secret values must never appear in Compose config output, logs, argv, settings, health, or + committed workspace files. +- Existing `THT_*_SECRET_FILE` variables remain a documented compatibility path until removed by + a later migration. +- External, local-vector, and preprocess overlays must remain independently renderable. +- Provider compound credentials remain fail-closed; only supported single-key providers are + restored from the bundle. +- Every task starts with a failing regression test and ends with focused tests, diff checks, and + a small commit. + +--- + +### Task 1: Add the strict secret-bundle loader and compatibility adapter + +**Files:** +- Create: `backend/src/config/secret-bundle.ts` +- Modify: `backend/src/config.ts` +- Modify: `backend/src/pi/provider-credentials.ts` +- Modify: `backend/src/pi/pi-process-manager.ts` +- Modify: `backend/src/pi/list-models.ts` +- Create: `backend/test/secret-bundle.test.ts` +- Modify: `backend/test/provider-credentials.test.ts` +- Modify: `backend/test/pi-process-manager.test.ts` +- Modify: `backend/test/list-models.test.ts` + +**Interfaces:** +- `loadSecretBundle(file: string): ReadonlyMap` validates `NAME=VALUE` lines, + duplicate/unknown/empty keys, `lstat`/`open(O_NOFOLLOW)`/`fstat` identity, owner and mode. +- `secretValue(config, key)` first reads `THT_SECRETS_FILE`, then falls back to the existing + `THT_*_SECRET_FILE` variable for compatibility. +- The existing provider environment builder consumes a value map, so session and model-listing + children share identical scrubbing and canonical-provider mapping. + +- [ ] **Step 1: Write failing tests** for valid bundle parsing, comments/blank lines, duplicate + keys, unknown keys, missing file, mode/owner failure, inode replacement, and secret redaction. +- [ ] **Step 2: Run** `cd backend && npx vitest run test/secret-bundle.test.ts`; expected failure + because the loader does not exist. +- [ ] **Step 3: Implement** the loader with bounded line lengths, strict key allowlist, no shell + evaluation, sanitized errors, and legacy adapter lookup. +- [ ] **Step 4: Add tests** proving session spawn and model listing use the same bundle values and + do not inherit bundle path or unselected provider credentials. +- [ ] **Step 5: Run** `cd backend && npm run build && npx tsc --noEmit -p . && npx vitest run`; + expected all backend tests pass. +- [ ] **Step 6: Commit** `git commit -m "feat(config): load one validated secret bundle"`. + +### Task 2: Make the root Compose command the default + +**Files:** +- Create: `.env.example` +- Modify: `.gitignore` +- Modify: `compose.yaml` +- Modify: `deploy/compose.production.yaml` +- Modify: `deploy/compose.local.yaml` +- Modify: `deploy/env.example` +- Create: `deploy/secrets/thothii.secrets.example` +- Create: `scripts/test-default-compose.sh` +- Modify: `scripts/test-container-deployment.sh` + +**Interfaces:** +- Root `.env` is Compose's automatic interpolation file; `.env.example` contains relative + `THT_SECRETS_FILE=deploy/secrets/thothii.secrets`, default `COMPOSE_FILE=compose.yaml`, and + the selected overlay/profile values. +- `compose.yaml` starts `core` and `frontend` without requiring a profile; overlays extend it. +- Core receives one `/run/secrets/thothii.secrets` mount and `THT_SECRETS_FILE` path. + +- [ ] **Step 1: Write failing static tests** that run `docker compose config --quiet` from a + temporary clone with `.env` and assert the default services are `core` and `frontend`, one + bundle is declared, and no legacy secret file is required. +- [ ] **Step 2: Run** `./scripts/test-default-compose.sh`; expected failure because root defaults + still require profiles/separate secret files. +- [ ] **Step 3: Implement** `.env.example`, `.gitignore`, Compose defaults and one secret mount. + Preserve `deploy/compose.production.yaml` as an optional authenticated production override. +- [ ] **Step 4: Run** `docker compose --env-file .env.example config --quiet` and the existing + deployment/security scripts; expected no secret values in rendered YAML. +- [ ] **Step 5: Commit** `git commit -m "build(compose): make root startup the default"`. + +### Task 3: Convert local-vector and preprocess services to the bundle + +**Files:** +- Modify: `deploy/compose.local-vector.yaml` +- Modify: `deploy/compose.preprocess-local-vector.yaml` +- Modify: `deploy/compose.preprocess.yaml` +- Modify: `deploy/workspaces/local-vector.yaml` +- Modify: `deploy/workspaces/preprocess-evidence.yaml` +- Modify: `deploy/workspaces/preprocess-dwh.yaml` +- Modify: `scripts/local-vector-smoke.sh` +- Modify: `scripts/preprocess-smoke.sh` +- Modify: `scripts/test-preprocess-compose-config.sh` +- Modify: `scripts/test-vector-backup-restore-safety.sh` + +**Interfaces:** +- Every local-vector/preprocess service reads the same mounted bundle path and selects only the + named value through the shared loader/helper. +- No service declares four file-backed Compose secrets after this task. + +- [ ] **Step 1: Add failing tests** asserting one bundle mount, no `vector_*_password` secret + declarations, and valid local-vector workspace resolution. +- [ ] **Step 2: Run** focused Compose config and smoke tests; expected failure with current + separate-secret declarations. +- [ ] **Step 3: Implement** bundle mounts and helper invocations for bootstrap/migrator/reader/ + writer operations, keeping passwords out of URLs and shell logs. +- [ ] **Step 4: Run** `./scripts/test-preprocess-compose-config.sh`, local-vector smoke and + preprocess smoke with a clean generated project; expected all pass. +- [ ] **Step 5: Commit** `git commit -m "feat(compose): use one secret bundle for local services"`. + +### Task 4: Finish documentation and end-to-end default verification + +**Files:** +- Modify: `README.md` +- Modify: `docs/installazione-docker-4-contesti.md` +- Modify: `docs/index.md` +- Modify: `deploy/secrets/README.md` +- Modify: `scripts/docker-smoke.sh` +- Modify: `scripts/test-default-compose.sh` + +**Interfaces:** +- Installation docs show only `cp .env.example .env`, create/fill one bundle, then + `docker compose up --build -d`. +- Advanced overlays are shown as optional `.env` presets, not mandatory command-line flags. + +- [ ] **Step 1: Add failing documentation/smoke assertions** for the exact command and default + files. +- [ ] **Step 2: Implement** concise context-specific instructions and migration notes for old + separate secret files. +- [ ] **Step 3: Run** all shell syntax/config gates, backend/frontend builds/tests, full harness, + default Docker smoke, local-vector smoke, preprocess smoke and `git diff --check`. +- [ ] **Step 4: Commit** `git commit -m "docs: document one-command Docker installation"`. + +### Task 5: Whole-plan review and handoff + +- [ ] Review `a6b195b..HEAD` against this plan and confirm no secret leakage, profile regression, + or legacy fallback bypass. +- [ ] Run the complete verification matrix and report exact counts, skipped L2 tests, and any + unavailable Docker/registry prerequisites. +- [ ] Keep the branch/worktree intact for the user's integration choice. diff --git a/docs/superpowers/specs/2026-07-11-portable-deployment-architecture-design.md b/docs/superpowers/specs/2026-07-11-portable-deployment-architecture-design.md index 4bd21ce5..5119b1ad 100644 --- a/docs/superpowers/specs/2026-07-11-portable-deployment-architecture-design.md +++ b/docs/superpowers/specs/2026-07-11-portable-deployment-architecture-design.md @@ -36,6 +36,7 @@ Non sono presenti Dockerfile o file Compose. Il processo di sviluppo assume Pi e 6. **Configurazione dichiarativa e validata.** I workspace contengono riferimenti logici e configurazioni non segrete; i segreti sono in environment variables o secret store. 7. **Capability esplicite.** Un adapter dichiara ciò che supporta. Le funzioni mancanti producono degradazione o blocco comprensibile, non emulazioni implicite. 8. **Read-only by construction sul DWH.** Credenziali, API e guard client-side mantengono la separazione dall'autorità di scrittura. +9. **Esposizione sicura per default.** La porta applicativa pubblicata è vincolata a loopback. Un'esposizione pubblica richiede un reverse proxy autenticante esterno e `AUTH_MODE=upstream`; la combinazione pubblico + `none` viene rifiutata all'avvio. OIDC interno non fa parte di questa fase. ## 4. Packaging e runtime @@ -143,6 +144,11 @@ La configurazione si divide in: - **workspace:** lingua, adapter, namespace, collezioni e policy di preprocessing; - **segreti:** password, token, certificati e chiavi reader/writer. +Nel profilo locale i segreti possono provenire da un env-file non versionato. In produzione sono +file read-only sotto `/run/secrets`, leggibili dall'UID 10001. Il reverse proxy autenticante è un +confine fidato: rimuove header identità forniti dal client e inserisce +`X-Authenticated-User` soltanto dopo autenticazione. + Deve esistere un comando di diagnostica che produca sia output umano sia JSON pristino, rispettando il contratto CLI corrente. ## 7. Pipeline di preprocessing diff --git a/docs/superpowers/specs/2026-07-12-local-and-server-docker-deployment-design.md b/docs/superpowers/specs/2026-07-12-local-and-server-docker-deployment-design.md new file mode 100644 index 00000000..4df20348 --- /dev/null +++ b/docs/superpowers/specs/2026-07-12-local-and-server-docker-deployment-design.md @@ -0,0 +1,34 @@ +# Local and Server Docker Deployment Design + +## Goal + +Run ThothII locally in Docker against the existing PSD services, while keeping the +same tracked Docker package deployable on a server hosting the DWH and pgvector. + +## Decisions + +- `compose.yaml` remains portable and contains no customer paths, credentials, or + private certificate material. +- The GLM registry is a tracked, non-secret Pi configuration. The provider key is + supplied only through the existing Docker secret bundle. +- A local-only Compose override binds the existing PSD workspace and CA material + from the developer machine. It is ignored by Git and exists solely to validate + Docker Desktop against the real workflow. +- A server profile consumes its own workspace mount, secret bundle, and service + endpoints. It never relies on macOS paths or a developer's `~/.pi` directory. +- The verification scope is a live session start through Pi using `zai/glm-5.2`; + it stops at the first human-review gate and does not finalize a datamart. + +## Configuration Boundaries + +Tracked files define images, Compose service contracts, templates, validation, and +documentation. Ignored runtime files hold the selected endpoint values, secret +bundle, certificate mount source, and local workspace path. The server receives +only the tracked package; its operator materializes equivalent runtime files with +server-specific values. + +## Validation + +The local run must validate Compose syntax, image builds, secret mount readability, +core/frontend health endpoints, model discovery, session creation, and receipt of a +Pi workflow event. Failure diagnostics must omit secret values. diff --git a/docs/superpowers/specs/2026-07-12-simple-docker-config-design.md b/docs/superpowers/specs/2026-07-12-simple-docker-config-design.md new file mode 100644 index 00000000..33fe2a46 --- /dev/null +++ b/docs/superpowers/specs/2026-07-12-simple-docker-config-design.md @@ -0,0 +1,98 @@ +# Design: configurazione Docker semplificata + +## Obiettivo + +Ridurre l'installazione a un file di configurazione `.env` interno al clone e a un solo file +contenente tutti i secret, mantenendo il comando operativo standard: + +```sh +docker compose up --build -d +``` + +Il comportamento di default deve essere determinato dai file presenti nella directory radice +`ThothII/`, senza obbligare l'operatore a ricordare `-f`, `--env-file` o profili Compose. + +## Struttura installativa + +```text +ThothII/ +├── .env # configurazione non segreta e default Compose +├── .env.example # template versionato +├── compose.yaml # file Compose principale, usabile senza -f +├── deploy/ +│ ├── secrets/thothii.secrets # unico file secret, escluso da Git +│ └── workspaces/ # workspace YAML versionati +└── data/ # dati persistenti solo se bind mount esplicito +``` + +`.env` contiene host, endpoint, profilo scelto, `COMPOSE_FILE`, `COMPOSE_PROFILES` e il percorso +del bundle secret. Non contiene valori secret. Il file viene creato copiando `.env.example` e +rimane nella directory `ThothII/`. + +Il bundle `deploy/secrets/thothii.secrets` usa righe `NOME=VALORE`, con nomi documentati e +validazione rigorosa. Non sono ammesse espansioni shell, comandi, URL con credenziali o righe +duplicate. Il file deve essere `0600` sull'host e viene montato read-only nei soli servizi che +ne hanno bisogno. + +## Default Compose + +`compose.yaml` diventa il file principale per il profilo applicativo esterno: `core` e +`frontend` non sono nascosti dietro un profilo obbligatorio. Il `.env` seleziona eventuali +overlay tramite la variabile Compose standard `COMPOSE_FILE` e il profilo tramite +`COMPOSE_PROFILES`. + +Esempi: + +- server con DWH/vector esterni: `COMPOSE_FILE=compose.yaml:deploy/compose.production.yaml`; +- Mac/Windows con pgvector locale: `COMPOSE_FILE=compose.yaml:deploy/compose.local-vector.yaml` e + `COMPOSE_PROFILES=local-vector`; +- preprocessing locale: aggiunta dell'overlay preprocess nel valore `COMPOSE_FILE`. + +Quando `.env` è configurato, il comando non cambia tra i contesti: + +```sh +docker compose up --build -d +``` + +I comandi con `-f` e `--env-file` restano documentati solo come override diagnostico, non come +percorso normale di installazione. + +## Bundle secret e runtime + +Il core riceve `THT_SECRETS_FILE=/run/secrets/thothii.secrets`. Un loader comune: + +1. apre il bundle con `O_NOFOLLOW`, verifica owner, permessi e inode; +2. rifiuta chiavi sconosciute, duplicate, vuote o provider composti non supportati; +3. espone i singoli valori solo in memoria al componente che ne ha bisogno; +4. non stampa il bundle, non lo inserisce in `settings.json`, argv, health o log. + +Per pgvector locale, il servizio di inizializzazione e le migrazioni usano lo stesso loader; non +si creano più file `bootstrap`, `reader`, `writer` e `migrator`. Le password non vengono passate +come argomenti URL. I workspace ricevono riferimenti logici al secret bundle, mai valori. + +La compatibilità temporanea con le variabili `THT_*_SECRET_FILE` viene mantenuta come fallback +esplicito per installazioni già esistenti, ma il template e la documentazione nuovi usano solo +`THT_SECRETS_FILE`. + +## Compatibilità e sicurezza + +- `docker compose config --quiet` deve funzionare dalla radice senza opzioni aggiuntive; +- il default non deve avviare pgvector locale se il `.env` seleziona servizi esterni; +- i profili local-vector e preprocess devono aggiungere solo i servizi necessari; +- il bundle secret deve essere escluso da `.gitignore` e dai build context Docker; +- errori di secret mancanti o non validi devono terminare prima dell'avvio applicativo, con messaggi + sanitizzati; +- i test devono coprire sia il percorso standard `docker compose up --build -d` sia gli override + legacy con file secret separati. + +## Verifica prevista + +La verifica finale comprende: + +1. rendering Compose del default e dei quattro preset `.env.example`; +2. test unitari del parser bundle e della compatibilità legacy; +3. build delle immagini core/frontend; +4. smoke health/SSE/persistenza; +5. smoke local-vector con un solo bundle e migrazioni; +6. smoke preprocess con il default selezionato dal `.env`; +7. controllo che nessun secret compaia in `docker compose config`, log, argv o immagini. diff --git a/frontend/index.html b/frontend/index.html index 4efb5f19..07fe9ee2 100644 --- a/frontend/index.html +++ b/frontend/index.html @@ -13,6 +13,7 @@
+ diff --git a/frontend/public/config.js b/frontend/public/config.js new file mode 100644 index 00000000..00a5e294 --- /dev/null +++ b/frontend/public/config.js @@ -0,0 +1 @@ +window.__THOTHII_CONFIG__ = {}; diff --git a/frontend/src/api/backend-url-cases.json b/frontend/src/api/backend-url-cases.json new file mode 100644 index 00000000..5391a1ed --- /dev/null +++ b/frontend/src/api/backend-url-cases.json @@ -0,0 +1,26 @@ +[ + { "value": "", "valid": true }, + { "value": "/", "valid": true }, + { "value": "/api", "valid": true }, + { "value": "/api/", "valid": true }, + { "value": "http://localhost:8787", "valid": true }, + { "value": "https://api.example.test/v1", "valid": true }, + { "value": "https://api.example.test/base/path/", "valid": true }, + { "value": "http://127.0.0.1:1/api", "valid": true }, + { "value": "http://[::1]:8787/api", "valid": true }, + { "value": "/backend", "valid": false }, + { "value": "api", "valid": false }, + { "value": "//evil.test", "valid": false }, + { "value": "http:///missing-authority", "valid": false }, + { "value": "https:///triple-slash", "valid": false }, + { "value": "http://", "valid": false }, + { "value": "http://example.test:abc", "valid": false }, + { "value": "http://example.test:65536", "valid": false }, + { "value": "http://example.test:999999999999999999999", "valid": false }, + { "value": "http://example.test:", "valid": false }, + { "value": "https://user:pass@example.test", "valid": false }, + { "value": "https://example.test/path with space", "valid": false }, + { "value": "ftp://example.test", "valid": false }, + { "value": "https://example.test/api?tenant=x", "valid": false }, + { "value": "https://example.test/api#fragment", "valid": false } +] diff --git a/frontend/src/api/backend-url-policy.json b/frontend/src/api/backend-url-policy.json new file mode 100644 index 00000000..b97c5da2 --- /dev/null +++ b/frontend/src/api/backend-url-policy.json @@ -0,0 +1,8 @@ +{ + "relativeBases": ["", "/", "/api", "/api/"], + "absolutePattern": "^https?://(?:\\[[0-9A-Fa-f:.]+\\]|[A-Za-z0-9](?:[A-Za-z0-9.-]*[A-Za-z0-9])?)(?::[0-9]+)?(?:/[^\\s?#]*)?/?$", + "maxPort": 65535, + "queryAllowed": false, + "fragmentAllowed": false, + "credentialsAllowed": false +} diff --git a/frontend/src/api/client.ts b/frontend/src/api/client.ts index 58d8c175..d602ba13 100644 --- a/frontend/src/api/client.ts +++ b/frontend/src/api/client.ts @@ -1,4 +1,4 @@ -const BASE = import.meta.env.VITE_BACKEND_URL ?? "http://localhost:8787"; +import { backendBaseUrl as BASE, joinBackendPath } from "./runtime-config"; export async function apiFetch(path: string, init?: RequestInit): Promise { // Only declare a JSON content-type when we actually send a body. Body-less @@ -10,7 +10,7 @@ export async function apiFetch(path: string, init?: RequestInit): Promise if (init?.body != null && !("content-type" in headers) && !("Content-Type" in headers)) { headers["content-type"] = "application/json"; } - const res = await fetch(`${BASE}${path}`, { ...init, headers }); + const res = await fetch(joinBackendPath(BASE, path), { ...init, headers }); if (!res.ok) throw new Error(`${res.status} ${await res.text().catch(() => "")}`); return res.status === 204 ? (undefined as T) : ((await res.json()) as T); } diff --git a/frontend/src/api/runtime-config.test.ts b/frontend/src/api/runtime-config.test.ts new file mode 100644 index 00000000..0753694d --- /dev/null +++ b/frontend/src/api/runtime-config.test.ts @@ -0,0 +1,51 @@ +import { describe, expect, it } from "vitest"; + +import { backendBaseUrl, joinBackendPath, resolveBackendUrl } from "./runtime-config"; +import cases from "./backend-url-cases.json"; + +describe("resolveBackendUrl", () => { + it("uses the runtime-injected backend URL", () => { + expect(resolveBackendUrl({ backendBaseUrl: "/api" })).toBe("/api"); + }); + + it("falls back to the Vite backend URL", () => { + expect(resolveBackendUrl(undefined)).toBe(import.meta.env.VITE_BACKEND_URL ?? ""); + }); + + it("preserves the client default when Vite has no configured backend", () => { + expect(backendBaseUrl).toBe(import.meta.env.VITE_BACKEND_URL ?? "http://localhost:8787"); + }); + + it.each(["/backend", "api", "//evil.test", "ftp://example.test", "https://user:pass@example.test"])( + "rejects unsupported backend URL %j", + (backendBaseUrl) => { + expect(() => resolveBackendUrl({ backendBaseUrl })).toThrow(/BACKEND_BASE_URL/); + }, + ); + + it.each(["", "/", "/api", "/api/", "http://localhost:8787", "https://api.example.test/v1"])( + "accepts supported backend URL %j", + (backendBaseUrl) => { + expect(resolveBackendUrl({ backendBaseUrl })).toBe(backendBaseUrl); + }, + ); + + it.each(cases)("applies the canonical policy to $value", ({ value, valid }) => { + const resolve = () => resolveBackendUrl({ backendBaseUrl: value }); + if (valid) expect(resolve()).toBe(value); + else expect(resolve).toThrow(/BACKEND_BASE_URL/); + }); +}); + +describe("joinBackendPath", () => { + it.each([ + ["", "/sessions/s1", "/sessions/s1"], + ["/", "/sessions/s1", "/sessions/s1"], + ["/api", "/sessions/s1", "/api/sessions/s1"], + ["/api/", "/sessions/s1", "/api/sessions/s1"], + ["https://example.test/api", "/sessions/s1", "https://example.test/api/sessions/s1"], + ["https://example.test/api/", "sessions/s1", "https://example.test/api/sessions/s1"], + ])("joins base %j and path %j", (base, path, expected) => { + expect(joinBackendPath(base, path)).toBe(expected); + }); +}); diff --git a/frontend/src/api/runtime-config.ts b/frontend/src/api/runtime-config.ts new file mode 100644 index 00000000..9cfab5f5 --- /dev/null +++ b/frontend/src/api/runtime-config.ts @@ -0,0 +1,41 @@ +import policy from "./backend-url-policy.json"; + +export interface RuntimeConfig { + backendBaseUrl?: string; +} + +declare global { + interface Window { + __THOTHII_CONFIG__?: RuntimeConfig; + } +} + +export function resolveBackendUrl(config: RuntimeConfig | undefined): string { + const value = config?.backendBaseUrl ?? import.meta.env.VITE_BACKEND_URL ?? ""; + if (policy.relativeBases.includes(value)) return value; + try { + if (!new RegExp(policy.absolutePattern).test(value)) throw new Error("syntax"); + const authority = value.replace(/^https?:\/\//, "").split("/", 1)[0]; + const suffix = authority.startsWith("[") + ? authority.slice(authority.indexOf("]") + 1) + : authority.slice(authority.lastIndexOf(":")); + const port = suffix.startsWith(":") ? suffix.slice(1) : ""; + if (port && (port.length > 5 || Number(port) > policy.maxPort)) throw new Error("port"); + return value; + } catch { + // Fall through to the single actionable runtime error below. + } + throw new Error( + "Invalid BACKEND_BASE_URL: use empty/root, /api, or a valid http(s) base without credentials, query, or fragment", + ); +} + +export function joinBackendPath(base: string, path: string): string { + const normalizedBase = base === "/" ? "" : base.replace(/\/+$/, ""); + const normalizedPath = path.replace(/^\/+/, ""); + return `${normalizedBase}/${normalizedPath}`; +} + +export const backendBaseUrl = + resolveBackendUrl(typeof window === "undefined" ? undefined : window.__THOTHII_CONFIG__) || + "http://localhost:8787"; diff --git a/frontend/src/stream/useSessionStream.test.tsx b/frontend/src/stream/useSessionStream.test.tsx index c77d8fcd..a5da11f5 100644 --- a/frontend/src/stream/useSessionStream.test.tsx +++ b/frontend/src/stream/useSessionStream.test.tsx @@ -13,7 +13,7 @@ beforeEach(() => { test("opens an EventSource and feeds NAMED events to the store", () => { renderHook(() => useSessionStream("s1")); const es = FakeEventSource.instances[0]; - expect(es.url).toContain("/sessions/s1/events"); + expect(es.url).toBe("http://localhost:8787/sessions/s1/events"); // Backend sends `event: ui_request` (named) — drive the addEventListener path // that production relies on, not the unnamed onmessage fallback. act(() => diff --git a/frontend/src/stream/useSessionStream.ts b/frontend/src/stream/useSessionStream.ts index 982cdf98..5289cdda 100644 --- a/frontend/src/stream/useSessionStream.ts +++ b/frontend/src/stream/useSessionStream.ts @@ -1,5 +1,6 @@ import { useEffect, useState } from "react"; import { BASE } from "../api/client"; +import { joinBackendPath } from "../api/runtime-config"; import { useSessionStore } from "../store/sessionStore"; import type { StreamEvent } from "../api/types"; @@ -10,7 +11,7 @@ export function useSessionStream(sessionId: string | null) { useEffect(() => { if (!sessionId) return; - const es = new EventSource(`${BASE}/sessions/${sessionId}/events`); + const es = new EventSource(joinBackendPath(BASE, `/sessions/${sessionId}/events`)); es.onopen = () => setConnected(true); es.onerror = () => setConnected(false); diff --git a/harness/pyproject.toml b/harness/pyproject.toml index 12b952be..5972c288 100644 --- a/harness/pyproject.toml +++ b/harness/pyproject.toml @@ -22,6 +22,7 @@ dependencies = [ tht = "tht.cli:app" [project.optional-dependencies] +s3 = ["boto3>=1.34,<2"] dev = [ "pytest>=8.0", "testcontainers[postgres]>=4.0", @@ -31,6 +32,9 @@ dev = [ [tool.setuptools.packages.find] include = ["tht*"] +[tool.setuptools.package-data] +tht = ["migrations/vector/*.sql"] + [tool.ruff] line-length = 100 diff --git a/harness/scripts/create_vector_reader_rpc.sql b/harness/scripts/create_vector_reader_rpc.sql index 913e3b4c..08897921 100644 --- a/harness/scripts/create_vector_reader_rpc.sql +++ b/harness/scripts/create_vector_reader_rpc.sql @@ -28,7 +28,9 @@ $$; create or replace function public.search_similar( table_name text, query_embedding vector, - limit_count integer + limit_count integer, + kinds text[] default null, + metadata_filter jsonb default null ) returns table(id bigint, similarity real, metadata jsonb) language plpgsql @@ -37,15 +39,33 @@ set search_path = public, vectors, extensions as $$ begin perform public._assert_vector_read_table(table_name); + if metadata_filter is not null and ( + table_name <> 'evidence' + or not (metadata_filter ? 'vector_generation') + or not (metadata_filter ? 'document_ids') + or not (metadata_filter ? 'workspace_id') + or jsonb_object_length(metadata_filter) <> 3 + or jsonb_typeof(metadata_filter->'document_ids') <> 'array' + ) then + raise exception 'invalid Evidence metadata filter'; + end if; return query execute format( 'select t.id, (1 - (t.embedding <=> $1))::real as similarity, t.metadata from vectors.%I t + where ($3 is null or t.kind = any ($3)) + and ($4 is null or ( + t.metadata->>''vector_generation'' = $4->>''vector_generation'' + and t.metadata->>''workspace_id'' = $4->>''workspace_id'' + and t.metadata->>''document_id'' in ( + select jsonb_array_elements_text($4->''document_ids'') + ) + )) order by t.embedding <=> $1 limit $2', table_name - ) using query_embedding, limit_count; + ) using query_embedding, limit_count, kinds, metadata_filter; end; $$; @@ -75,7 +95,7 @@ end; $$; revoke all on function public._assert_vector_read_table(text) from public; -revoke all on function public.search_similar(text, vector, integer) from public; +revoke all on function public.search_similar(text, vector, integer, text[], jsonb) from public; revoke all on function public.list_tables() from public; -- Revoca dai ruoli client generici, poi abilita solo il reader dedicato @@ -83,15 +103,16 @@ revoke all on function public.list_tables() from public; do $$ begin if exists (select 1 from pg_roles where rolname = 'anon') then - revoke all on function public.search_similar(text, vector, integer) from anon; + revoke all on function public.search_similar(text, vector, integer, text[], jsonb) from anon; revoke all on function public.list_tables() from anon; end if; if exists (select 1 from pg_roles where rolname = 'authenticated') then - revoke all on function public.search_similar(text, vector, integer) from authenticated; + revoke all on function public.search_similar(text, vector, integer, text[], jsonb) from authenticated; revoke all on function public.list_tables() from authenticated; end if; if exists (select 1 from pg_roles where rolname = 'vector_reader') then - grant execute on function public.search_similar(text, vector, integer) to vector_reader; + grant execute on function public.search_similar(text, vector, integer, text[], jsonb) + to vector_reader; grant execute on function public.list_tables() to vector_reader; end if; end $$; diff --git a/harness/scripts/create_vector_writer_rpc.sql b/harness/scripts/create_vector_writer_rpc.sql index 51a1aea5..538e215b 100644 --- a/harness/scripts/create_vector_writer_rpc.sql +++ b/harness/scripts/create_vector_writer_rpc.sql @@ -40,6 +40,30 @@ begin end; $$; +drop function if exists public.list_evidence_generations(text, text); + +create or replace function public.list_evidence_generations( + table_name text, kind text, workspace_id text +) +returns table(generation text) +language plpgsql +security definer +set search_path = public, vectors, extensions +as $$ +begin + if table_name <> 'evidence' or kind <> 'evidence' or workspace_id !~ '^[a-z][a-z0-9_-]{0,63}$' then + raise exception 'only exact Evidence generations may be listed'; + end if; + return query + select distinct e.metadata->>'vector_generation' + from vectors.evidence e + where e.kind = 'evidence' + and e.metadata->>'vector_generation' ~ '^gen:[0-9a-f]{32}$' + and e.metadata->>'workspace_id' = workspace_id + order by 1; +end; +$$; + create or replace function public.existing_vector_hashes(table_name text, kinds text[]) returns table(record_key text, content_hash text) language plpgsql @@ -101,9 +125,35 @@ begin end; $$; +drop function if exists public.delete_vector_generation(text, text, text); + +create or replace function public.delete_vector_generation( + table_name text, kind text, generation text, workspace_id text +) +returns jsonb +language plpgsql +security definer +set search_path = public, vectors, extensions +as $$ +declare affected integer; +begin + if table_name <> 'evidence' or kind <> 'evidence' or generation !~ '^gen:[0-9a-f]{32}$' + or workspace_id !~ '^[a-z][a-z0-9_-]{0,63}$' then + raise exception 'only an exact Evidence generation may be deleted'; + end if; + delete from vectors.evidence e + where e.kind = 'evidence' and e.metadata->>'vector_generation' = generation + and e.metadata->>'workspace_id' = workspace_id; + get diagnostics affected = row_count; + return jsonb_build_object('deleted', affected); +end; +$$; + revoke all on function public._assert_vector_write_table(text, text[]) from public; revoke all on function public.existing_vector_hashes(text, text[]) from public; revoke all on function public.upsert_vector_records(text, jsonb) from public; +revoke all on function public.delete_vector_generation(text, text, text, text) from public; +revoke all on function public.list_evidence_generations(text, text, text) from public; -- Su alcuni progetti Supabase le funzioni in `public` ricevono grant automatici: revoca -- esplicitamente dai ruoli client generici, poi abilita solo il writer dedicato. @@ -112,14 +162,20 @@ begin if exists (select 1 from pg_roles where rolname = 'anon') then revoke all on function public.existing_vector_hashes(text, text[]) from anon; revoke all on function public.upsert_vector_records(text, jsonb) from anon; + revoke all on function public.delete_vector_generation(text, text, text, text) from anon; + revoke all on function public.list_evidence_generations(text, text, text) from anon; end if; if exists (select 1 from pg_roles where rolname = 'authenticated') then revoke all on function public.existing_vector_hashes(text, text[]) from authenticated; revoke all on function public.upsert_vector_records(text, jsonb) from authenticated; + revoke all on function public.delete_vector_generation(text, text, text, text) from authenticated; + revoke all on function public.list_evidence_generations(text, text, text) from authenticated; end if; if exists (select 1 from pg_roles where rolname = 'vector_writer') then grant execute on function public.existing_vector_hashes(text, text[]) to vector_writer; grant execute on function public.upsert_vector_records(text, jsonb) to vector_writer; + grant execute on function public.delete_vector_generation(text, text, text, text) to vector_writer; + grant execute on function public.list_evidence_generations(text, text, text) to vector_writer; end if; end $$; diff --git a/harness/tests/l0/test_db_sampling.py b/harness/tests/l0/test_db_sampling.py index 66fe41ca..e15529d4 100644 --- a/harness/tests/l0/test_db_sampling.py +++ b/harness/tests/l0/test_db_sampling.py @@ -6,11 +6,27 @@ import pytest from tht.config import LshConfig from tht.db.introspect import introspect -from tht.db.sampling import is_text_type, unique_values_for_lsh +from tht.db.sampling import distinct_values, is_text_type, sample_column, unique_values_for_lsh pytestmark = [pytest.mark.l0] +@pytest.mark.parametrize("invalid_limit", [True, 1.5, 0, -1]) +def test_direct_sampling_rejects_non_positive_integer_limits(admin_engine, invalid_limit): + with pytest.raises(ValueError, match="positive integer"): + sample_column( + admin_engine, "dw", "fct_ricoveri", "reparto", limit=invalid_limit + ) + with pytest.raises(ValueError, match="positive integer"): + distinct_values( + admin_engine, + "dw", + "fct_ricoveri", + "reparto", + max_values=invalid_limit, + ) + + def test_is_text_type(): assert is_text_type("text") assert is_text_type("varchar(100)") @@ -52,3 +68,16 @@ def test_unique_values_for_lsh_truncation_reported(admin_engine): truncated_cols = {(t.table, t.column) for t in truncated} # fct_ricoveri has several eligible text columns with distinct values assert any(t[0] == "fct_ricoveri" for t in truncated_cols) + + +def test_adapter_sampling_is_distinct_and_frequency_ranked(admin_engine): + values = sample_column(admin_engine, "dw", "fct_ricoveri", "reparto", limit=2) + assert values == ["cardiologia", "pronto soccorso"] + + +def test_adapter_distinct_values_reports_truncation(admin_engine): + result = distinct_values( + admin_engine, "dw", "fct_ricoveri", "reparto", max_values=1 + ) + assert result.values == ["cardiologia"] + assert result.truncated is True diff --git a/harness/tests/l0/test_pgvector_corpus_lifecycle.py b/harness/tests/l0/test_pgvector_corpus_lifecycle.py new file mode 100644 index 00000000..9a06bc29 --- /dev/null +++ b/harness/tests/l0/test_pgvector_corpus_lifecycle.py @@ -0,0 +1,258 @@ +"""L0 gate for the complete durable Evidence/pgvector lifecycle.""" + +import hashlib +import json + +import pytest +from sqlalchemy import create_engine, text +from testcontainers.postgres import PostgresContainer + +from tht.adapters.evidence import FilesystemEvidenceSource +from tht.adapters.vector.pgvector import PgVectorStore +from tht.cli.vector_migrate_cmd import migrate +from tht.config import DatabaseConfig +from tht.corpus.chunk import ChunkPolicy +from tht.corpus.pipeline import CorpusPipeline +from tht.corpus.store import CorpusStore +from tht.ports.vector import VectorRecord, VectorWriteRecord +from tht.search import combined_search +from tht.search.evidence import ActiveEvidenceSearcher, resolve_evidence_file + + +DIMENSIONS = 768 + + +class DeterministicEmbedder: + def embed_documents(self, texts): + return [self.embed_query(text) for text in texts] + + def embed_query(self, text): + vector = [0.0] * DIMENSIONS + vector[0] = 0.8 + vector[1] = 0.6 + return vector + + +class EvidenceDelegate: + """Adapt the real multi-collection port to the runtime search protocol.""" + + def __init__(self, store): + self.store = store + + def search(self, embedding, top_n=10, kinds=None, metadata_filter=None): + return self.store.search( + ["evidence"], embedding, limit=top_n, kinds=kinds, + metadata_filter=metadata_filter, + ) + + +class InterruptAfterRealPartialUpsert: + """Crash after a committed real row, as a process death would.""" + + def __init__(self, store): + self.store = store + self.interrupt = True + + def __getattr__(self, name): + return getattr(self.store, name) + + def upsert(self, collection, records): + if self.interrupt and len(records) > 1: + self.interrupt = False + self.store.upsert(collection, records[:1]) + raise KeyboardInterrupt("injected process death after committed vector row") + return self.store.upsert(collection, records) + + +@pytest.fixture(scope="module") +def persistent_pgvector(): + with PostgresContainer("pgvector/pgvector:pg16") as postgres: + migrate(postgres.get_connection_url()) + admin = create_engine(postgres.get_connection_url()) + with admin.begin() as connection: + connection.exec_driver_sql( + "ALTER ROLE vector_reader LOGIN PASSWORD 'reader-lifecycle'" + ) + connection.exec_driver_sql( + "ALTER ROLE vector_writer LOGIN PASSWORD 'writer-lifecycle'" + ) + url = admin.url + common = dict( + host=url.host, port=url.port, database=url.database, schema="vectors" + ) + reader = DatabaseConfig( + **common, user="vector_reader", password="reader-lifecycle" + ) + writer = DatabaseConfig( + **common, user="vector_writer", password="writer-lifecycle" + ) + yield postgres, admin, reader, writer + admin.dispose() + + +def _pipeline(root, source_root, vectors): + return CorpusPipeline( + store=CorpusStore(root / "corpus"), + sources=[FilesystemEvidenceSource(source_root)], + embedder=DeterministicEmbedder(), + vector_store=vectors, + embedding_model="deterministic-l0", + embedding_dimensions=DIMENSIONS, + chunk_policy=ChunkPolicy(version="lifecycle-v1", max_chars=48), + pipeline_version="evidence-v1", + retain_published_generations=2, + ) + + +def _publish(pipeline, root, serial): + return pipeline.run_as_job( + workspace_id="pgvector-lifecycle", + workspace_root=root, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + f"{serial:x}" * 64, + ) + + +@pytest.mark.l0 +def test_real_pgvector_corpus_job_lifecycle(tmp_path, persistent_pgvector): + postgres, admin, reader_config, writer_config = persistent_pgvector + source_root = tmp_path / "sources" + source_root.mkdir() + kept = source_root / "kept.md" + stable = source_root / "stable.md" + stable.write_text("unchanged dependency evidence", encoding="utf-8") + removed = source_root / "removed.md" + removed.write_text("removed evidence generation zero", encoding="utf-8") + + vectors = PgVectorStore(reader_config, writer_config, expected_dimension=DIMENSIONS) + generations = [] + for serial in range(3): + kept.write_text(f"active evidence generation {serial}", encoding="utf-8") + result = _publish(_pipeline(tmp_path, source_root, vectors), tmp_path, serial + 1) + assert result.status == "succeeded" + generations.append(result.generation) + removed_document = next( + doc for doc in CorpusStore(tmp_path / "corpus").active_manifest().documents + if "removed.md" in doc.source_uri + ) + removed_document_id = removed_document.document_id + removed_ref = removed_document.document_id + + # A stale, closer row must not consume LIMIT before ACTIVE filtering. + stale_generation = generations[-2] + stale = VectorWriteRecord( + record=VectorRecord( + id="chunk:stale-perfect-match", kind="evidence", ref="doc:stale", + title="stale forbidden", content="stale forbidden", + metadata={ + "document_id": "doc:stale", + "vector_generation": stale_generation, + }, + ), + embedding=[1.0] + [0.0] * (DIMENSIONS - 1), + content_hash="sha256:" + "a" * 64, + ) + vectors.upsert("evidence", [stale]) + runtime = ActiveEvidenceSearcher(CorpusStore(tmp_path / "corpus"), EvidenceDelegate(vectors)) + query = DeterministicEmbedder().embed_query("active") + hits = runtime.search(query, top_n=1, kinds=["evidence"]) + assert len(hits) == 1 and hits[0].title != "stale forbidden" + packed = combined_search( + "active", lsh_hits=None, store=runtime, embedder=DeterministicEmbedder(), + top=1, rrf_k=60, kinds=["evidence"], query_vec=query, + ) + assert len(packed) == 1 and packed[0].label != "stale forbidden" + + # Fourth publication removes a document and creates multiple chunks for crash recovery. + removed.unlink() + kept.write_text("active fourth generation " * 8, encoding="utf-8") + crashing = InterruptAfterRealPartialUpsert(vectors) + candidate = _pipeline(tmp_path, source_root, crashing) + with pytest.raises(KeyboardInterrupt, match="injected process death"): + _publish(candidate, tmp_path, 4) + runs = tmp_path / ".tht-jobs" / "evidence" / "runs" + crashed_run = max(runs.iterdir(), key=lambda path: path.stat().st_mtime_ns).name + before = vectors.existing_hashes("evidence", ["evidence"]) + intent = json.loads( + (runs / crashed_run / "artifacts" / "vector-intent.json").read_text() + )["records"] + already_present = set(intent) & set(before) + assert len(already_present) == 1 + resumed = candidate.run_as_job( + workspace_id="pgvector-lifecycle", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "4" * 64, + resume_run_id=crashed_run, + ) + assert resumed.status == "succeeded" and resumed.resumed_from == crashed_run + generations.append(resumed.generation) + after = vectors.existing_hashes("evidence", ["evidence"]) + assert {key: after[key] for key in already_present} == { + key: before[key] for key in already_present + } + assert set(intent).issubset(after) + with admin.connect() as connection: + duplicate_count = connection.execute(text( + "SELECT count(*) - count(DISTINCT record_key) FROM vectors.evidence" + )).scalar_one() + assert duplicate_count == 0 + + runtime = ActiveEvidenceSearcher(CorpusStore(tmp_path / "corpus"), EvidenceDelegate(vectors)) + active_hits = runtime.search(query, top_n=20, kinds=["evidence"]) + assert active_hits + assert any("active fourth generation" in hit.content for hit in active_hits) + assert all(hit.ref not in {"doc:stale", removed_ref} for hit in active_hits) + assert all(hit.metadata.get("document_id") != removed_document_id for hit in active_hits) + active_pack = combined_search( + "active", lsh_hits=None, store=runtime, embedder=DeterministicEmbedder(), + top=20, rrf_k=60, kinds=["evidence"], query_vec=query, + ) + assert active_pack + assert any("active fourth generation" in result.content for result in active_pack) + assert all(removed_document.content not in result.content for result in active_pack) + assert any("unchanged dependency evidence" in result.content for result in active_pack) + manifest = CorpusStore(tmp_path / "corpus").active_manifest() + assert resolve_evidence_file( + CorpusStore(tmp_path / "corpus"), removed_document_id, + materialized_root=tmp_path / "session", + ) == "" + + active_document = manifest.documents[0] + owned = CorpusStore(tmp_path / "corpus").materialize_document( + active_document.document_id, tmp_path / "session" / "active-evidence.md" + ) + assert owned.read_bytes() == active_document.content.encode() + assert hashlib.sha256(owned.read_bytes()).hexdigest() == active_document.content_hash[7:] + + # Recreate engines and stores against the same persisted database. + vectors._reader.dispose() + vectors._writer.dispose() + recreated = PgVectorStore(reader_config, writer_config, expected_dimension=DIMENSIONS) + assert recreated.health().ok is True + recreated_hits = ActiveEvidenceSearcher( + CorpusStore(tmp_path / "corpus"), EvidenceDelegate(recreated) + ).search(query, top_n=2, kinds=["evidence"]) + active_dependencies = set(manifest.metadata["document_generations"].values()) + assert recreated_hits + assert all(hit.metadata["vector_generation"] in active_dependencies for hit in recreated_hits) + + orphan = "gen:" + "f" * 32 + recreated.upsert("evidence", [VectorWriteRecord( + record=VectorRecord( + id="chunk:exact-vector-orphan", kind="evidence", ref="doc:orphan", + title="orphan", content="orphan", + metadata={"document_id": "doc:orphan", "vector_generation": orphan, + "workspace_id": "pgvector-lifecycle"}, + ), + embedding=query, content_hash="sha256:" + "f" * 64, + )]) + final_pipeline = _pipeline(tmp_path, source_root, recreated) + report = final_pipeline.gc(workspace_root=tmp_path) + assert report["evicted"] == [orphan] + expected_fs = set(generations[-2:]) + assert set(CorpusStore(tmp_path / "corpus").list_generations()) == expected_fs + expected_vectors = expected_fs | {generations[0]} + assert set(recreated.list_evidence_generations( + "evidence", "pgvector-lifecycle" + )) == expected_vectors + assert final_pipeline.gc(workspace_root=tmp_path)["evicted"] == [] diff --git a/harness/tests/l0/test_pgvector_store.py b/harness/tests/l0/test_pgvector_store.py new file mode 100644 index 00000000..83c77b66 --- /dev/null +++ b/harness/tests/l0/test_pgvector_store.py @@ -0,0 +1,406 @@ +import pytest +from psycopg2.errors import InsufficientPrivilege +from sqlalchemy import create_engine, text +from sqlalchemy.exc import ProgrammingError +from testcontainers.postgres import PostgresContainer + +from tht.adapters.vector.thoth_http import ThothHttpVectorStore +from tht.config import DatabaseConfig +from tht.ports.vector import ( + VectorReadUnavailable, + VectorRecord, + VectorStoreError, + VectorWriteRecord, + VectorWriteUnavailable, +) + + +def _record(content_hash: str, embedding: list[float], *, kind: str = "memory"): + return VectorWriteRecord( + record=VectorRecord( + id=f"record:{content_hash}", + kind=kind, + ref="session:test", + title=content_hash, + content=f"content {content_hash}", + metadata={"content_hash": content_hash}, + ), + embedding=embedding, + content_hash=content_hash, + ) + + +@pytest.fixture(scope="module") +def vector_configs(): + with PostgresContainer("pgvector/pgvector:pg16") as pg: + host = pg.get_container_host_ip() + port = int(pg.get_exposed_port(5432)) + admin_config = DatabaseConfig( + host=host, + port=port, + database=pg.dbname, + schema="vectors", + user=pg.username, + password=pg.password, + ) + engine = create_engine(pg.get_connection_url()) + with engine.begin() as connection: + connection.exec_driver_sql("CREATE SCHEMA vectors") + connection.exec_driver_sql("CREATE EXTENSION vector WITH SCHEMA vectors") + for table in ("schema_records", "evidence", "memory"): + connection.exec_driver_sql(f""" + CREATE TABLE vectors.{table} ( + id bigserial PRIMARY KEY, + record_key text UNIQUE NOT NULL, + kind text NOT NULL, + content_hash text NOT NULL, + metadata jsonb NOT NULL, + embedding vectors.vector(2) NOT NULL, + indexed_at timestamptz NOT NULL DEFAULT now() + ) + """) + connection.exec_driver_sql("CREATE ROLE vector_l0_reader LOGIN PASSWORD 'reader'") + connection.exec_driver_sql("CREATE ROLE vector_l0_writer LOGIN PASSWORD 'writer'") + connection.exec_driver_sql( + "CREATE ROLE vector_l0_no_sequence LOGIN PASSWORD 'no_sequence'" + ) + connection.exec_driver_sql( + "GRANT USAGE ON SCHEMA vectors TO vector_l0_reader, vector_l0_writer, " + "vector_l0_no_sequence" + ) + connection.exec_driver_sql( + "GRANT SELECT ON ALL TABLES IN SCHEMA vectors TO vector_l0_reader" + ) + connection.exec_driver_sql( + "GRANT USAGE, SELECT ON ALL SEQUENCES IN SCHEMA vectors TO vector_l0_writer" + ) + for table in ("schema_records", "evidence", "memory"): + connection.exec_driver_sql( + f"GRANT INSERT, UPDATE ON vectors.{table} " + "TO vector_l0_writer, vector_l0_no_sequence" + ) + if table == "evidence": + connection.exec_driver_sql( + "GRANT DELETE ON vectors.evidence TO vector_l0_writer" + ) + connection.exec_driver_sql( + "GRANT SELECT (kind, metadata) ON vectors.evidence TO vector_l0_writer" + ) + connection.exec_driver_sql( + f"GRANT SELECT (record_key, kind, content_hash) " + f"ON vectors.{table} TO vector_l0_writer, vector_l0_no_sequence" + ) + engine.dispose() + reader_config = admin_config.model_copy( + update={"user": "vector_l0_reader", "password": "reader"} + ) + writer_config = admin_config.model_copy( + update={"user": "vector_l0_writer", "password": "writer"} + ) + no_sequence_config = admin_config.model_copy( + update={"user": "vector_l0_no_sequence", "password": "no_sequence"} + ) + yield admin_config, reader_config, writer_config, no_sequence_config + + +@pytest.fixture +def store(vector_configs): + from tht.adapters.vector.pgvector import PgVectorStore + + _, reader_config, writer_config, _ = vector_configs + store = PgVectorStore(reader_config, writer_config, expected_dimension=2) + store.upsert("memory", [_record("reset", [0.0, 1.0])]) + yield store + + +def test_pgvector_round_trip_hash_and_upsert(store): + assert store.upsert("memory", [_record("a", [1.0, 0.0])]) == 1 + assert store.existing_hashes("memory", ["memory"])["record:a"] == "a" + + hits = store.search(["memory"], [1.0, 0.0], limit=5, kinds=["memory"]) + assert hits[0].metadata["content_hash"] == "a" + assert hits[0].id == "record:a" + + assert store.upsert("memory", [_record("a", [0.8, 0.2])]) == 1 + assert store.search(["memory"], [0.8, 0.2], limit=1)[0].id == "record:a" + + +def test_pgvector_lists_and_deletes_exact_evidence_generation(store): + generation = "gen:" + "a" * 32 + value = VectorWriteRecord( + record=VectorRecord( + id="evidence-generation-a", kind="evidence", ref="doc:a", title="a", + content="content", metadata={"vector_generation": generation, "workspace_id": "default"}, + ), + embedding=[1.0, 0.0], content_hash="sha256:" + "a" * 64, + ) + store.upsert("evidence", [value]) + assert generation in store.list_evidence_generations("evidence", "default") + assert store.delete_generation("evidence", generation, "default") == 1 + assert generation not in store.list_evidence_generations("evidence", "default") + + +def test_pgvector_generation_cleanup_isolated_between_workspaces(store): + generation = "gen:" + "b" * 32 + records = [VectorWriteRecord( + record=VectorRecord( + id=f"evidence-{workspace}", kind="evidence", ref=f"doc:{workspace}", + title=workspace, content=workspace, + metadata={"vector_generation": generation, "workspace_id": workspace}, + ), embedding=[1.0, 0.0], content_hash="sha256:" + key * 64, + ) for workspace, key in (("workspace-a", "b"), ("workspace-b", "c"))] + store.upsert("evidence", records) + assert store.delete_generation("evidence", generation, "workspace-a") == 1 + assert generation not in store.list_evidence_generations("evidence", "workspace-a") + assert generation in store.list_evidence_generations("evidence", "workspace-b") + + +def test_pgvector_search_filters_kinds_before_limit(store): + store.upsert("memory", [_record("solved", [1.0, 0.0], kind="solved_question")]) + hits = store.search("memory".split(), [1.0, 0.0], limit=1, kinds=["memory"]) + assert len(hits) == 1 + assert hits[0].kind == "memory" + + +def test_pgvector_multi_collection_search_skips_collections_unrelated_to_kinds(store): + store.upsert("evidence", [_record("evidence", [1.0, 0.0], kind="evidence")]) + + hits = store.search(["evidence", "memory"], [1.0, 0.0], limit=3, kinds=["memory"]) + + assert hits + assert {hit.kind for hit in hits} == {"memory"} + + +def test_pgvector_multi_collection_kind_filter_matches_http_adapter(store): + class Reader: + def search_similar(self, collection, embedding, limit, kinds=None): + if collection != "memory" or "memory" not in (kinds or []): + return [] + return [ + { + "similarity": 1.0, + "metadata": { + "record_key": "record:a", + "kind": "memory", + "ref": "session:test", + "title": "a", + "content": "content a", + "content_hash": "a", + }, + } + ] + + direct = store.search(["evidence", "memory"], [1.0, 0.0], limit=1, kinds=["memory"]) + http = ThothHttpVectorStore(Reader(), None).search( + ["evidence", "memory"], [1.0, 0.0], limit=1, kinds=["memory"] + ) + assert [(hit.id, hit.kind) for hit in direct] == [(hit.id, hit.kind) for hit in http] + + +def test_pgvector_search_rejects_unknown_kind_globally(store): + with pytest.raises(VectorStoreError, match="Kind not allowed"): + store.search(["memory"], [1.0, 0.0], limit=1, kinds=["unknown"]) + + +@pytest.mark.parametrize("limit", [True, False, 1.0, 0, -1]) +def test_pgvector_search_requires_strict_positive_limit(store, limit): + with pytest.raises(ValueError, match="positive integer"): + store.search(["memory"], [1.0, 0.0], limit=limit) + + +def test_pgvector_allowlists_collections(store): + with pytest.raises(VectorStoreError, match="Collection not allowed"): + store.search(["memory; DROP SCHEMA vectors"], [1.0, 0.0], limit=1) + with pytest.raises(VectorStoreError, match="Collection not allowed"): + store.upsert("unknown", []) + + +def test_pgvector_rejects_kinds_not_belonging_to_collection(store): + with pytest.raises(VectorStoreError, match="Kind not allowed"): + store.existing_hashes("evidence", ["memory"]) + with pytest.raises(VectorStoreError, match="Kind not allowed"): + store.upsert("evidence", [_record("wrong", [1.0, 0.0])]) + + +def test_pgvector_separates_read_and_write_credentials(vector_configs): + from tht.adapters.vector.pgvector import PgVectorStore + + _, reader_config, writer_config, _ = vector_configs + reader = PgVectorStore(reader_config, expected_dimension=2) + assert reader.capabilities.search is True + assert reader.capabilities.upsert is False + with pytest.raises(VectorWriteUnavailable): + reader.upsert("memory", []) + + writer = PgVectorStore(None, writer_config, expected_dimension=2) + assert writer.capabilities.search is False + assert writer.capabilities.upsert is True + with pytest.raises(VectorReadUnavailable): + writer.search(["memory"], [1.0, 0.0], limit=1) + + +def test_pgvector_database_roles_are_least_privilege(vector_configs): + _, reader_config, writer_config, _ = vector_configs + reader_engine = create_engine( + f"postgresql+psycopg2://{reader_config.user}:{reader_config.password}" + f"@{reader_config.host}:{reader_config.port}/{reader_config.database}" + ) + writer_engine = create_engine( + f"postgresql+psycopg2://{writer_config.user}:{writer_config.password}" + f"@{writer_config.host}:{writer_config.port}/{writer_config.database}" + ) + with pytest.raises(ProgrammingError): + with reader_engine.begin() as connection: + connection.execute( + text( + "INSERT INTO vectors.memory " + "(record_key, kind, content_hash, metadata, embedding) " + "VALUES ('forbidden', 'memory', 'x', '{}', '[1,0]')" + ) + ) + with pytest.raises(ProgrammingError): + with writer_engine.connect() as connection: + connection.execute( + text( + "SELECT metadata, 1 - (embedding <=> '[1,0]'::vector) AS similarity " + "FROM vectors.memory ORDER BY embedding <=> '[1,0]'::vector LIMIT 1" + ) + ) + reader_engine.dispose() + writer_engine.dispose() + + +def test_pgvector_writer_health_requires_sequence_usage(vector_configs): + from tht.adapters.vector.pgvector import PgVectorStore + + admin_config, _, _, no_sequence_config = vector_configs + store = PgVectorStore(None, no_sequence_config, expected_dimension=2) + + health = store.health() + assert health.ok is False + assert health.write_reachable is False + assert health.write_detail == ( + "vector schema incomplete: missing sequence privileges evidence, memory, schema_records" + ) + with pytest.raises(VectorWriteUnavailable) as error: + store.upsert("memory", [_record("needs-sequence", [1.0, 0.0])]) + assert isinstance(error.value.__cause__, InsufficientPrivilege) + + admin_engine = create_engine( + f"postgresql+psycopg2://{admin_config.user}:{admin_config.password}" + f"@{admin_config.host}:{admin_config.port}/{admin_config.database}" + ) + with admin_engine.begin() as connection: + connection.exec_driver_sql( + "GRANT USAGE ON ALL SEQUENCES IN SCHEMA vectors TO vector_l0_no_sequence" + ) + admin_engine.dispose() + + assert store.health().ok is True + assert store.upsert("memory", [_record("has-sequence", [1.0, 0.0])]) == 1 + + +def test_pgvector_health_requires_schema_usage_for_reader_and_writer(vector_configs): + from tht.adapters.vector.pgvector import PgVectorStore + + admin_config, reader_config, writer_config, _ = vector_configs + admin_engine = create_engine( + f"postgresql+psycopg2://{admin_config.user}:{admin_config.password}" + f"@{admin_config.host}:{admin_config.port}/{admin_config.database}" + ) + store = PgVectorStore(reader_config, writer_config, expected_dimension=2) + with admin_engine.begin() as connection: + connection.exec_driver_sql( + f"REVOKE USAGE ON SCHEMA vectors FROM {reader_config.user}, {writer_config.user}" + ) + health = store.health() + assert health.read_reachable is False and health.write_reachable is False + assert "missing schema usage" in health.read_detail + assert "missing schema usage" in health.write_detail + with pytest.raises(VectorReadUnavailable, match="Vector read operation unavailable"): + store.search(["memory"], [1.0, 0.0], limit=1) + with pytest.raises(VectorWriteUnavailable, match="Vector write operation unavailable"): + store.upsert("memory", [_record("blocked", [1.0, 0.0])]) + with admin_engine.begin() as connection: + connection.exec_driver_sql( + f"GRANT USAGE ON SCHEMA vectors TO {reader_config.user}, {writer_config.user}" + ) + admin_engine.dispose() + assert store.health().ok is True + + +def test_pgvector_maps_unavailable_connections_without_leaking_password(vector_configs): + from tht.adapters.vector.pgvector import PgVectorStore + + _, reader_config, writer_config, _ = vector_configs + password = "never-leak-this" + reader = reader_config.model_copy(update={"port": 1, "password": password}) + writer = writer_config.model_copy(update={"port": 1, "password": password}) + with pytest.raises(VectorReadUnavailable) as read_error: + PgVectorStore(reader, None).search(["memory"], [1.0, 0.0], limit=1) + with pytest.raises(VectorWriteUnavailable) as hash_error: + PgVectorStore(None, writer).existing_hashes("memory", ["memory"]) + with pytest.raises(VectorWriteUnavailable) as write_error: + PgVectorStore(None, writer).upsert("memory", [_record("x", [1.0, 0.0])]) + assert password not in str(read_error.value) + assert password not in str(hash_error.value) + assert password not in str(write_error.value) + + +def test_pgvector_health_reports_dimension_and_each_connection(vector_configs): + from tht.adapters.vector.pgvector import PgVectorStore + + _, reader_config, writer_config, _ = vector_configs + health = PgVectorStore(reader_config, writer_config, expected_dimension=2).health() + assert health.ok is True + assert health.read_reachable is True + assert health.write_reachable is True + assert health.observed_dimensions == (2,) + assert health.dimension_compatible is True + + mismatch = PgVectorStore(reader_config, None, expected_dimension=3).health() + assert mismatch.ok is False + assert mismatch.read_reachable is False + assert mismatch.read_detail == ( + "embedding dimension mismatch: evidence=2, memory=2, schema_records=2" + ) + assert mismatch.dimension_compatible is False + + +def test_pgvector_health_rejects_clean_and_partial_schemas(vector_configs): + from tht.adapters.vector.pgvector import PgVectorStore + + admin_config, _, _, _ = vector_configs + engine = create_engine( + f"postgresql+psycopg2://{admin_config.user}:{admin_config.password}" + f"@{admin_config.host}:{admin_config.port}/{admin_config.database}" + ) + with engine.begin() as connection: + connection.exec_driver_sql("CREATE SCHEMA clean_vectors") + connection.exec_driver_sql("CREATE SCHEMA partial_vectors") + connection.exec_driver_sql( + "CREATE TABLE partial_vectors.memory " + "(record_key text, kind text, content_hash text, metadata jsonb)" + ) + engine.dispose() + + clean = PgVectorStore( + admin_config.model_copy(update={"db_schema": "clean_vectors"}), + expected_dimension=2, + ).health() + assert clean.ok is False + assert clean.read_reachable is False + assert clean.read_detail == ( + "vector schema incomplete: missing tables evidence, memory, schema_records" + ) + + partial = PgVectorStore( + admin_config.model_copy(update={"db_schema": "partial_vectors"}), + expected_dimension=2, + ).health() + assert partial.ok is False + assert partial.read_reachable is False + assert partial.read_detail == ( + "vector schema incomplete: missing tables evidence, schema_records; " + "missing embedding columns memory" + ) diff --git a/harness/tests/l0/test_vector_adapter_parity.py b/harness/tests/l0/test_vector_adapter_parity.py new file mode 100644 index 00000000..ec5899ad --- /dev/null +++ b/harness/tests/l0/test_vector_adapter_parity.py @@ -0,0 +1,288 @@ +import math + +import pytest +from sqlalchemy import create_engine +from testcontainers.postgres import PostgresContainer + +from tht.adapters.vector.pgvector import PgVectorStore +from tht.adapters.vector.thoth_http import ThothHttpVectorStore +from tht.config import DatabaseConfig, RestConfig +from tht.ports.vector import VectorRecord, VectorStoreError, VectorWriteRecord +from tht.vectorstore.rest_client import VectorRestClient, VectorRestError + + +def _write(record_id, kind, embedding, content_hash): + return VectorWriteRecord( + VectorRecord( + id=record_id, + kind=kind, + ref="fixture", + title=record_id, + content=f"content {record_id}", + metadata={"fixture": True}, + ), + embedding, + content_hash, + ) + + +FIXTURE = [ + _write("memory:a", "memory", [1.0, 0.0], "hash-a"), + _write("memory:b", "memory", [1.0, 0.0], "hash-b"), + _write("solved:a", "solved_question", [0.8, 0.2], "hash-solved"), +] + + +class Response: + def __init__(self, payload=None, status=200): + self.status_code = status + self.payload = payload + self.text = "" if payload is None else "json" + + @property + def ok(self): + return self.status_code < 400 + + def json(self): + return self.payload + + +class FixtureHttpTransport: + def __init__(self): + self.rows = {} + self.calls = [] + + def post(self, url, json, headers, **kwargs): + assert headers == {"X-API-Key": "parity-key"} + self.calls.append((url.rsplit("/", 1)[-1], json)) + function = self.calls[-1][0] + if function == "list_tables": + return Response([{"table_name": "memory", "vector_dimensions": 2}]) + if function == "upsert_vector_records": + for row in json["rows"]: + self.rows[(json["table_name"], row["record_key"])] = row + return Response({"upserted": len(json["rows"])}) + if function == "existing_vector_hashes": + return Response([ + {"record_key": row["record_key"], "content_hash": row["content_hash"]} + for (table, _), row in self.rows.items() + if table == json["table_name"] and row["kind"] in json["kinds"] + ]) + assert function == "search_similar" + table_name = json["table_name"] + embedding = json["query_embedding"] + kinds = json.get("kinds") + + def similarity(row): + left, right = row["embedding"], embedding + return sum(a * b for a, b in zip(left, right)) / ( + math.sqrt(sum(a * a for a in left)) + * math.sqrt(sum(b * b for b in right)) + ) + + rows = [ + {"metadata": row["metadata"], "similarity": similarity(row)} + for (table, _), row in self.rows.items() + if table == table_name and (not kinds or row["kind"] in kinds) + ] + payload = sorted( + rows, + key=lambda row: (-row["similarity"], row["metadata"]["record_key"]), + )[: json["limit_count"]] + return Response(payload) + + +@pytest.fixture +def direct_store(): + with PostgresContainer("pgvector/pgvector:pg16") as postgres: + config = DatabaseConfig( + host=postgres.get_container_host_ip(), + port=int(postgres.get_exposed_port(5432)), + database=postgres.dbname, + schema="vectors", + user=postgres.username, + password=postgres.password, + ) + engine = create_engine(postgres.get_connection_url()) + with engine.begin() as connection: + connection.exec_driver_sql("CREATE SCHEMA vectors") + connection.exec_driver_sql("CREATE EXTENSION vector WITH SCHEMA vectors") + connection.exec_driver_sql( + "CREATE TABLE vectors.memory (" + "id bigserial PRIMARY KEY, record_key text UNIQUE NOT NULL, " + "kind text NOT NULL, content_hash text NOT NULL, metadata jsonb NOT NULL, " + "embedding vectors.vector(2) NOT NULL, indexed_at timestamptz NOT NULL " + "DEFAULT now())" + ) + engine.dispose() + reader, writer = config, config + store = PgVectorStore(reader, writer, expected_dimension=2) + store.upsert("memory", FIXTURE) + yield store + + +@pytest.fixture +def http_store(monkeypatch): + transport = FixtureHttpTransport() + monkeypatch.setattr("tht.vectorstore.rest_client.requests.post", transport.post) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key")) + store = ThothHttpVectorStore(client, client, expected_dimension=2) + store.upsert("memory", FIXTURE) + store.transport = transport + return store + + +@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"]) +def test_kind_filtered_search_has_identical_order(request, store_fixture): + store = request.getfixturevalue(store_fixture) + hits = store.search(["memory"], [1.0, 0.0], limit=3, kinds=["memory"]) + assert [(hit.id, hit.kind, round(hit.similarity, 6)) for hit in hits] == [ + ("memory:a", "memory", 1.0), + ("memory:b", "memory", 1.0), + ] + + +@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"]) +def test_hash_and_upsert_parity(request, store_fixture): + store = request.getfixturevalue(store_fixture) + assert store.existing_hashes("memory", ["memory"]) == { + "memory:a": "hash-a", + "memory:b": "hash-b", + } + replacement = _write("memory:a", "memory", [0.0, 1.0], "hash-a-2") + assert store.upsert("memory", [replacement]) == 1 + assert store.existing_hashes("memory", ["memory"])["memory:a"] == "hash-a-2" + assert store.search(["memory"], [0.0, 1.0], limit=1, kinds=["memory"])[0].id == "memory:a" + + +@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"]) +def test_validation_error_parity(request, store_fixture): + store = request.getfixturevalue(store_fixture) + with pytest.raises(VectorStoreError, match="Collection not allowed"): + store.search(["not_allowed"], [1.0, 0.0], limit=1) + with pytest.raises(VectorStoreError, match="Kind not allowed"): + store.search(["memory"], [1.0, 0.0], limit=1, kinds=["not_allowed"]) + + +@pytest.mark.parametrize("store_fixture", ["direct_store", "http_store"]) +def test_dimension_error_parity(request, store_fixture): + store = request.getfixturevalue(store_fixture) + with pytest.raises(VectorStoreError, match="Query embedding dimension"): + store.search(["memory"], [1.0], limit=1) + with pytest.raises(VectorStoreError, match="Embedding dimension"): + store.upsert("memory", [_write("bad", "memory", [1.0], "bad")]) + + +def test_http_parity_exercises_rpc_kinds_payload(http_store): + http_store.search(["memory"], [1.0, 0.0], limit=2, kinds=["memory"]) + search_calls = [payload for function, payload in http_store.transport.calls if function == "search_similar"] + assert search_calls[-1] == { + "query_embedding": [1.0, 0.0], + "limit_count": 2, + "table_name": "memory", + "kinds": ["memory"], + } + + +def test_http_adapter_maps_transport_error(monkeypatch): + monkeypatch.setattr( + "tht.vectorstore.rest_client.requests.post", + lambda *args, **kwargs: Response({"message": "server broke"}, status=500), + ) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key")) + store = ThothHttpVectorStore(client, client, expected_dimension=2) + with pytest.raises(VectorStoreError, match="HTTP 500"): + store.search(["memory"], [1.0, 0.0], limit=1, kinds=["memory"]) + + +def test_http_adapter_tolerates_malformed_metadata(monkeypatch): + monkeypatch.setattr( + "tht.vectorstore.rest_client.requests.post", + lambda *args, **kwargs: Response([{"similarity": 0.5, "metadata": None}]), + ) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key")) + hit = ThothHttpVectorStore(client, None, expected_dimension=2).search( + ["memory"], [1.0, 0.0], limit=1 + )[0] + assert (hit.id, hit.kind, hit.metadata) == ("", "", {}) + + +def test_http_adapter_legacy_fallback_preserves_kind_semantics(monkeypatch): + calls = [] + + def post(url, json, **kwargs): + calls.append(json) + if "kinds" in json: + return Response({"message": "function not found"}, status=404) + return Response([ + {"similarity": 1.0, "metadata": {"record_key": "wrong", "kind": "solved_question"}}, + {"similarity": 0.9, "metadata": {"record_key": "right", "kind": "memory"}}, + ]) + + monkeypatch.setattr("tht.vectorstore.rest_client.requests.post", post) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="parity-key")) + hits = ThothHttpVectorStore(client, None, expected_dimension=2).search( + ["memory"], [1.0, 0.0], limit=2, kinds=["memory"] + ) + assert [hit.id for hit in hits] == ["right"] + assert "kinds" in calls[0] and "kinds" not in calls[1] + + +def test_http_delete_generation_uses_exact_allowlisted_rpc_payload(monkeypatch): + calls = [] + monkeypatch.setattr( + "tht.vectorstore.rest_client.requests.post", + lambda url, json, **kwargs: calls.append((url, json)) or Response({"deleted": 2}), + ) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer")) + assert client.delete_generation("evidence", "gen:" + "a" * 32, "default") == 2 + assert calls == [("https://vectors.test/rpc/delete_vector_generation", { + "table_name": "evidence", "kind": "evidence", "generation": "gen:" + "a" * 32, + "workspace_id": "default", + })] + + +def test_http_delete_generation_legacy_404_fails_closed_without_body_leak(monkeypatch): + monkeypatch.setattr( + "tht.vectorstore.rest_client.requests.post", + lambda *args, **kwargs: Response({"message": "secret legacy endpoint detail"}, status=404), + ) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer")) + with pytest.raises(VectorRestError, match="delete_vector_generation RPC is unavailable") as error: + client.delete_generation("evidence", "gen:" + "a" * 32, "default") + assert "secret" not in str(error.value) + + +def test_http_list_evidence_generations_exact_rpc_and_legacy_fail_closed(monkeypatch): + calls = [] + monkeypatch.setattr( + "tht.vectorstore.rest_client.requests.post", + lambda url, json, **kwargs: calls.append((url, json)) or Response([ + {"generation": "gen:" + "a" * 32} + ]), + ) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer")) + assert client.list_evidence_generations("evidence", "default") == ["gen:" + "a" * 32] + assert calls[0][0].endswith("/rpc/list_evidence_generations") + assert calls[0][1] == {"table_name": "evidence", "kind": "evidence", "workspace_id": "default"} + + +@pytest.mark.parametrize("generation", ["gen:a", "gen:" + "A" * 32, "gen:" + "a" * 33]) +def test_http_generation_operations_reject_noncanonical_values(monkeypatch, generation): + monkeypatch.setattr( + "tht.vectorstore.rest_client.requests.post", + lambda *args, **kwargs: pytest.fail("invalid generation reached transport"), + ) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer")) + with pytest.raises(ValueError, match="canonical"): + client.delete_generation("evidence", generation, "default") + + +def test_http_inventory_rejects_malformed_rpc_output(monkeypatch): + monkeypatch.setattr( + "tht.vectorstore.rest_client.requests.post", + lambda *args, **kwargs: Response([{"generation": "gen:../escape"}]), + ) + client = VectorRestClient(RestConfig(base_url="https://vectors.test", api_key="writer")) + with pytest.raises(VectorRestError, match="malformed"): + client.list_evidence_generations("evidence", "default") diff --git a/harness/tests/l0/test_vector_migrations.py b/harness/tests/l0/test_vector_migrations.py new file mode 100644 index 00000000..72640003 --- /dev/null +++ b/harness/tests/l0/test_vector_migrations.py @@ -0,0 +1,303 @@ +import json +from pathlib import Path + +import pytest +from sqlalchemy import create_engine, text +from sqlalchemy.exc import ProgrammingError +from testcontainers.postgres import PostgresContainer +from typer.testing import CliRunner + +from tht.cli import app +from tht.config import DatabaseConfig +from tht.ports.vector import VectorRecord, VectorWriteRecord + + +@pytest.fixture(scope="module") +def database_url(): + with PostgresContainer("pgvector/pgvector:pg16") as postgres: + yield postgres.get_connection_url() + + +def test_migrations_are_clean_and_idempotent(database_url): + from tht.cli.vector_migrate_cmd import migrate, migration_status + + before = migration_status(database_url) + assert [item.version for item in before.pending] == ["001", "002", "003", "004"] + + migrate(database_url) + migrate(database_url) + + status = migration_status(database_url) + assert status.pending == () + assert status.drifted == () + assert [item.version for item in status.applied] == ["001", "002", "003", "004"] + + +def test_schema_matches_direct_adapter_contract(database_url): + engine = create_engine(database_url) + with engine.connect() as connection: + rows = connection.execute( + text( + "SELECT table_name, column_name, data_type, udt_name " + "FROM information_schema.columns WHERE table_schema = 'vectors' " + "ORDER BY table_name, ordinal_position" + ) + ).all() + vector_types = connection.execute( + text( + "SELECT c.relname, format_type(a.atttypid, a.atttypmod) " + "FROM pg_class c JOIN pg_namespace n ON n.oid = c.relnamespace " + "JOIN pg_attribute a ON a.attrelid = c.oid AND a.attname = 'embedding' " + "WHERE n.nspname = 'vectors' ORDER BY c.relname" + ) + ).all() + engine.dispose() + + tables = {row.table_name for row in rows} + assert tables == {"evidence", "memory", "schema_records"} + required = {"id", "record_key", "kind", "content_hash", "metadata", "embedding", "indexed_at"} + for table in tables: + assert {row.column_name for row in rows if row.table_name == table} == required + assert vector_types == [ + ("evidence", "vectors.vector(768)"), + ("memory", "vectors.vector(768)"), + ("schema_records", "vectors.vector(768)"), + ] + + +def test_roles_have_runtime_privileges_only(database_url): + from tht.cli.vector_migrate_cmd import migrate + + migrate(database_url) + admin = create_engine(database_url) + with admin.begin() as connection: + connection.exec_driver_sql("ALTER ROLE vector_reader LOGIN PASSWORD 'reader-test-only'") + connection.exec_driver_sql("ALTER ROLE vector_writer LOGIN PASSWORD 'writer-test-only'") + url = admin.url + reader = create_engine(url.set(username="vector_reader", password="reader-test-only")) + writer = create_engine(url.set(username="vector_writer", password="writer-test-only")) + + from tht.adapters.vector.pgvector import PgVectorStore + + common = { + "host": url.host, + "port": url.port, + "database": url.database, + "schema": "vectors", + } + reader_config = DatabaseConfig( + **common, user="vector_reader", password="reader-test-only" + ) + writer_config = DatabaseConfig( + **common, user="vector_writer", password="writer-test-only" + ) + store = PgVectorStore(reader_config, writer_config, expected_dimension=768) + assert store.health().ok is True + assert store.upsert( + "memory", + [ + VectorWriteRecord( + record=VectorRecord( + id="adapter-write", + kind="memory", + ref="session:test", + title="test", + content="test", + ), + embedding=[0.0] * 768, + content_hash="adapter-hash", + ) + ], + ) == 1 + + with reader.connect() as connection: + connection.execute(text("SELECT metadata, embedding FROM vectors.memory")).all() + with pytest.raises(ProgrammingError): + with reader.begin() as connection: + connection.execute( + text( + "INSERT INTO vectors.memory " + "(record_key, kind, content_hash, metadata, embedding) " + "VALUES ('reader-write', 'memory', 'x', '{}', " + "array_fill(0, ARRAY[768])::vectors.vector)" + ) + ) + + with writer.begin() as connection: + connection.execute( + text( + "INSERT INTO vectors.memory " + "(record_key, kind, content_hash, metadata, embedding) " + "VALUES ('writer-ok', 'memory', 'x', '{}', " + "array_fill(0, ARRAY[768])::vectors.vector)" + ) + ) + assert connection.execute( + text("SELECT content_hash FROM vectors.memory WHERE record_key = 'writer-ok'") + ).scalar_one() == "x" + connection.execute( + text("UPDATE vectors.memory SET content_hash = 'y' WHERE record_key = 'writer-ok'") + ) + with pytest.raises(ProgrammingError): + with writer.connect() as connection: + connection.execute(text("SELECT metadata FROM vectors.memory")).all() + with pytest.raises(ProgrammingError): + with writer.begin() as connection: + connection.execute(text("DELETE FROM vectors.memory WHERE record_key = 'writer-ok'")) + + reader.dispose() + writer.dispose() + admin.dispose() + + +def test_status_json_is_pristine(database_url, monkeypatch): + monkeypatch.setenv("THT_VECTOR_ADMIN_URL", database_url) + result = CliRunner().invoke(app, ["vector", "migrate", "--status", "--json"]) + + assert result.exit_code == 0, result.output + assert json.loads(result.stdout) == { + "applied": ["001", "002", "003", "004"], + "drifted": [], + "pending": [], + } + assert result.stderr == "" + + +def test_checksum_drift_is_reported_and_refused(database_url, tmp_path): + from tht.cli.vector_migrate_cmd import MigrationError, migrate, migration_status + + migrations = _copy_migrations(tmp_path) + migrate(database_url, migrations) + (migrations / "002_schema_tables.sql").write_text("SELECT 2;\n") + + assert [item.version for item in migration_status(database_url, migrations).drifted] == [ + "002" + ] + with pytest.raises(MigrationError, match="checksum drift"): + migrate(database_url, migrations) + + +def test_unknown_applied_version_is_downgrade_drift(database_url): + from tht.cli.vector_migrate_cmd import MigrationError, migrate, migration_status + + migrate(database_url) + engine = create_engine(database_url) + with engine.begin() as connection: + connection.execute( + text( + "INSERT INTO public.tht_vector_migrations (version, name, checksum) " + "VALUES ('999', 'future', 'future-checksum'), " + "('future_x', 'future_named', 'future-checksum')" + ) + ) + try: + with pytest.raises( + MigrationError, match="absent from local manifest: 999, future_x" + ): + migration_status(database_url) + with pytest.raises( + MigrationError, match="absent from local manifest: 999, future_x" + ): + migrate(database_url) + finally: + with engine.begin() as connection: + connection.execute( + text( + "DELETE FROM public.tht_vector_migrations " + "WHERE version IN ('999', 'future_x')" + ) + ) + engine.dispose() + + +def test_migration_versions_sort_numerically_and_reject_numeric_duplicates(tmp_path): + from tht.cli.vector_migrate_cmd import MigrationError, _discover + + migrations = tmp_path / "ordered" + migrations.mkdir() + (migrations / "10_tenth.sql").write_text("SELECT 10;\n") + (migrations / "2_second.sql").write_text("SELECT 2;\n") + assert [item.version for item in _discover(migrations)] == ["2", "10"] + + (migrations / "02_duplicate.sql").write_text("SELECT 2;\n") + with pytest.raises(MigrationError, match="Duplicate migration version: 2"): + _discover(migrations) + + +def test_hostile_admin_search_path_cannot_shadow_migration_objects(database_url): + from tht.cli.vector_migrate_cmd import migrate + + admin = create_engine(database_url, isolation_level="AUTOCOMMIT") + with admin.connect() as connection: + connection.exec_driver_sql("DROP DATABASE IF EXISTS vector_hostile") + connection.exec_driver_sql("CREATE DATABASE vector_hostile") + hostile_url = admin.url.set(database="vector_hostile") + hostile = create_engine(hostile_url) + try: + with hostile.begin() as connection: + connection.exec_driver_sql("CREATE SCHEMA shadow") + connection.exec_driver_sql( + "CREATE TABLE shadow.tht_vector_migrations " + "(version text, checksum text, poisoned boolean DEFAULT true)" + ) + connection.exec_driver_sql("ALTER ROLE test SET search_path = shadow, public") + hostile.dispose() + + migrate(hostile_url.render_as_string(hide_password=False)) + + verification = create_engine(hostile_url) + with verification.connect() as connection: + assert connection.execute( + text("SELECT count(*) FROM public.tht_vector_migrations") + ).scalar_one() == 4 + assert connection.execute( + text("SELECT count(*) FROM shadow.tht_vector_migrations") + ).scalar_one() == 0 + assert connection.execute( + text( + "SELECT format_type(a.atttypid, a.atttypmod) " + "FROM pg_catalog.pg_attribute a " + "WHERE a.attrelid = 'vectors.memory'::pg_catalog.regclass " + "AND a.attname = 'embedding'" + ) + ).scalar_one() == "vectors.vector(768)" + verification.dispose() + finally: + cleanup = create_engine(database_url, isolation_level="AUTOCOMMIT") + with cleanup.connect() as connection: + connection.exec_driver_sql("ALTER ROLE test RESET search_path") + connection.exec_driver_sql( + "SELECT pg_catalog.pg_terminate_backend(pid) FROM pg_catalog.pg_stat_activity " + "WHERE datname = 'vector_hostile' AND pid <> pg_catalog.pg_backend_pid()" + ) + connection.exec_driver_sql("DROP DATABASE IF EXISTS vector_hostile") + cleanup.dispose() + admin.dispose() + + +def test_failed_batch_rolls_back_schema_and_ledger(database_url, tmp_path): + from tht.cli.vector_migrate_cmd import MigrationError, migrate, migration_status + + migrations = _copy_migrations(tmp_path) + (migrations / "005_first.sql").write_text("CREATE TABLE public.must_rollback (id int);\n") + (migrations / "006_broken.sql").write_text("THIS IS NOT SQL;\n") + + with pytest.raises(MigrationError, match="006_broken.sql"): + migrate(database_url, migrations) + + engine = create_engine(database_url) + with engine.connect() as connection: + assert connection.execute(text("SELECT to_regclass('public.must_rollback')")).scalar() is None + engine.dispose() + status = migration_status(database_url, migrations) + assert [item.version for item in status.applied] == ["001", "002", "003", "004"] + assert [item.version for item in status.pending] == ["005", "006"] + + +def _copy_migrations(tmp_path: Path) -> Path: + source = Path(__file__).parents[2] / "tht" / "migrations" / "vector" + target = tmp_path / "migrations" + target.mkdir() + for migration in source.glob("*.sql"): + (target / migration.name).write_bytes(migration.read_bytes()) + return target diff --git a/harness/tests/l2/test_memory_save_one_real.py b/harness/tests/l2/test_memory_save_one_real.py index bc98c283..c7e2aa74 100644 --- a/harness/tests/l2/test_memory_save_one_real.py +++ b/harness/tests/l2/test_memory_save_one_real.py @@ -39,7 +39,10 @@ def test_save_one_upserts_to_real_pgvector(l2_env): detail="ablazione", rationale="L2 self-test (idempotent)", question_context="ablazione 2025", tables=["fct_ricoveri"], concepts=[], ) - upserted = save_one_memory([record], decision_seq=999, writer=writer, embedder=embedder) + from tht.adapters.vector import ThothHttpVectorStore + + store = ThothHttpVectorStore(reader=writer, writer=writer) + upserted = save_one_memory([record], decision_seq=999, store=store, embedder=embedder) assert upserted >= 0 # idempotent: 0 on unchanged, >=1 on new/updated # read it back via the READER key (vector_rest, path /vector/v1/) diff --git a/harness/tests/test_adapter_command_regressions.py b/harness/tests/test_adapter_command_regressions.py new file mode 100644 index 00000000..21a67802 --- /dev/null +++ b/harness/tests/test_adapter_command_regressions.py @@ -0,0 +1,116 @@ +from types import SimpleNamespace + +import pytest +import typer + +from tht.cli import db_cmd +from tht.cli.lsh_cmd import _extract_lsh_values +from tht.mschema.models import Annotations, ColumnPhysical, PhysicalSchema, TablePhysical +from tht.ports.dwh import DistinctValues, DwhHealth +from tht.cli import memory_cmd +from tht.memory import MemoryRecord +from datetime import datetime + + +def _ping(monkeypatch, health, capsys): + monkeypatch.setattr(db_cmd, "load_config", lambda path: SimpleNamespace(database=SimpleNamespace(user="u"))) + monkeypatch.setattr(db_cmd, "build_dwh", lambda cfg: SimpleNamespace(health=lambda: health)) + try: + db_cmd.ping_cmd() + except typer.Exit as exc: + code = exc.exit_code + else: + code = 0 + return code, capsys.readouterr() + + +def test_db_ping_public_health_success(monkeypatch, capsys): + code, output = _ping(monkeypatch, DwhHealth(ok=True, database="d", schema="s", read_only=True), capsys) + assert code == 0 + assert "OK: connesso a d (schema s)" in output.out + + +def test_db_ping_rest_inaccessible_historical_wording(monkeypatch, capsys): + code, output = _ping(monkeypatch, DwhHealth(ok=False, detail="{'db_connected': False}", error_kind="inaccessible"), capsys) + assert code == 1 + assert "ERRORE: DWH non accessibile via REST (risposta: {'db_connected': False})." in output.err + + +def test_db_ping_direct_connection_historical_wording(monkeypatch, capsys): + code, output = _ping(monkeypatch, DwhHealth(ok=False, detail="connection refused", error_kind="connection"), capsys) + assert code == 1 + assert "ERRORE di connessione: connection refused" in output.err + + +@pytest.mark.parametrize(("limit", "truncated"), [(7, False), (1201, True)]) +def test_lsh_extraction_honors_configured_limit(limit, truncated): + physical = PhysicalSchema(database="d", schema="s", introspected_at=datetime(2026, 1, 1), tables={ + "t": TablePhysical(columns={"c": ColumnPhysical(type="text", eligible=True)}) + }) + calls = [] + class Dwh: + def distinct_values(self, table, column, *, limit): + calls.append(limit) + return DistinctValues(values=list(range(limit)), truncated=truncated) + values, _, reports = _extract_lsh_values(Dwh(), physical, Annotations(), limit) + assert calls == [limit] + assert len(values["t"]["c"]) == limit + assert [report.indexed for report in reports] == ([limit] if truncated else []) + + +def test_memory_command_writes_through_factory_vector_store(monkeypatch): + store = SimpleNamespace(existing_hashes=lambda *args: {}, upsert=lambda table, rows: 1) + captured = [] + original_upsert = store.upsert + store.upsert = lambda table, rows: captured.extend(rows) or original_upsert(table, rows) + cfg = SimpleNamespace(embeddings=object(), vector_write_rest=object()) + manifest = SimpleNamespace(id="s1") + record = MemoryRecord(id="m1", ts=datetime(2026, 1, 1), session_id="s1", + decision_seq=7, type="table_promoted", subject="t", + question_context="q") + monkeypatch.setattr(memory_cmd, "_load_config_or_exit", lambda path: cfg) + monkeypatch.setattr(memory_cmd, "load_session_or_exit", lambda cfg, session: manifest) + monkeypatch.setattr(memory_cmd, "require_vector_write_allowed", lambda *args: None) + monkeypatch.setattr(memory_cmd, "has_vector_write_rest", lambda cfg: True) + monkeypatch.setattr(memory_cmd, "session_dir", lambda *args: None) + monkeypatch.setattr(memory_cmd, "registry_path", lambda cfg: None) + monkeypatch.setattr("tht.adapters.factory.build_vector_store", lambda cfg, require_write: store) + monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", + lambda cfg: SimpleNamespace(embed_documents=lambda texts: [[0.1]])) + monkeypatch.setattr("tht.memory.promote", lambda *args, **kwargs: None) + monkeypatch.setattr("tht.memory.load_registry", lambda path: [record]) + memory_cmd.save_one_cmd(session="s1", decision=7, json_out=True) + from tht.ports.vector import VectorWriteRecord + assert len(captured) == 1 and isinstance(captured[0], VectorWriteRecord) + + +def test_solved_index_writes_through_writer_only_factory_store(monkeypatch): + writer_only_store = SimpleNamespace( + capabilities=SimpleNamespace(search=False, upsert=True), + existing_hashes=lambda *args: {}, + upsert=lambda table, rows: 1, + ) + cfg = SimpleNamespace(embeddings=object(), vector_write_rest=object()) + manifest = SimpleNamespace(id="s1") + solved_record = object() + calls = [] + + monkeypatch.setattr(memory_cmd, "has_vector_write_rest", lambda cfg: True) + monkeypatch.setattr(memory_cmd, "load_session_or_exit", lambda cfg, session: manifest) + monkeypatch.setattr(memory_cmd, "session_dir", lambda *args: None) + monkeypatch.setattr( + "tht.adapters.factory.build_vector_store", + lambda cfg, require_write: calls.append(require_write) or writer_only_store, + ) + monkeypatch.setattr("tht.cli.sql_cmd.promoted_tables_for", lambda *args: []) + monkeypatch.setattr("tht.solved.build_solved_record", lambda *args: solved_record) + monkeypatch.setattr( + "tht.solved.save_solved_question", + lambda record, *, store, embedder: int( + record is solved_record and store is writer_only_store + ), + ) + monkeypatch.setattr("tht.cli.vector_cmd.make_embedder", lambda cfg: object()) + + assert memory_cmd.index_solved_session(cfg, "s1") == 1 + assert calls == [True] diff --git a/harness/tests/test_adapter_factory.py b/harness/tests/test_adapter_factory.py new file mode 100644 index 00000000..ba417cf2 --- /dev/null +++ b/harness/tests/test_adapter_factory.py @@ -0,0 +1,131 @@ +import pytest + +from tht.adapters.dwh import PostgresDwhAdapter, ThothRestDwhAdapter +from tht.adapters.vector import PgVectorStore, ThothHttpVectorStore +from tht.adapters.factory import build_dwh, build_vector_store +from tht.config import Config, ConfigError + + +def _config(*, dwh_type="thoth_rest", vector_type="thoth_vector_http", reader=True, writer=True): + dwh = ( + { + "type": "thoth_rest", + "database": {"database": "analytics", "schema": "mart"}, + "endpoint": {"base_url": "https://dwh.test/", "api_key": "reader"}, + } + if dwh_type == "thoth_rest" + else { + "type": "postgres_direct", + "connection": { + "host": "db", + "database": "analytics", + "schema": "mart", + "user": "reader", + "password": "secret", + }, + } + ) + vectors = ( + { + "type": "thoth_vector_http", + **( + {"reader": {"base_url": "https://vectors.test/", "api_key": "reader"}} + if reader + else {} + ), + **( + {"writer": {"base_url": "https://vectors.test/", "api_key": "writer"}} + if writer + else {} + ), + } + if vector_type == "thoth_vector_http" + else { + "type": "pgvector_direct", + **( + { + "reader": { + "host": "vector-db", + "database": "postgres", + "schema": "vectors", + "user": "reader", + "password": "secret", + } + } + if reader + else {} + ), + **( + { + "writer": { + "host": "vector-db", + "database": "postgres", + "schema": "vectors", + "user": "writer", + "password": "secret", + } + } + if writer + else {} + ), + } + ) + legacy_database = ( + dwh["connection"] + if dwh_type == "postgres_direct" + else { + **dwh["database"], + "user": "rest", + "password": "", + "transport": "rest", + } + ) + return Config.model_validate({"dwh": dwh, "vectors": vectors, "database": legacy_database}) + + +@pytest.mark.parametrize( + ("dwh_type", "adapter_type"), + [("postgres_direct", PostgresDwhAdapter), ("thoth_rest", ThothRestDwhAdapter)], +) +def test_factory_selects_dwh_adapter(dwh_type, adapter_type): + assert isinstance(build_dwh(_config(dwh_type=dwh_type)), adapter_type) + + +def test_factory_selects_http_vector_and_requires_writer(): + config = _config(writer=False) + + assert isinstance(build_vector_store(config), ThothHttpVectorStore) + with pytest.raises(ConfigError, match="writer"): + build_vector_store(config, require_write=True) + + +def test_factory_builds_writer_only_http_vector_when_write_is_required(): + config = _config(reader=False, writer=True) + + store = build_vector_store(config, require_write=True) + assert isinstance(store, ThothHttpVectorStore) + assert store.capabilities.search is False + assert store.capabilities.upsert is True + + +def test_factory_selects_direct_vector_store_and_requires_writer(): + config = _config(vector_type="pgvector_direct", writer=False) + + assert isinstance(build_vector_store(config), PgVectorStore) + with pytest.raises(ConfigError, match="writer"): + build_vector_store(config, require_write=True) + + +def test_factory_builds_writer_only_direct_vector_when_write_is_required(): + store = build_vector_store( + _config(vector_type="pgvector_direct", reader=False), require_write=True + ) + assert isinstance(store, PgVectorStore) + assert store.capabilities.search is False + assert store.capabilities.upsert is True + + +def test_factory_propagates_non_default_statement_timeout(): + config = _config(dwh_type="postgres_direct") + config.execution.statement_timeout_ms = 12_345 + assert build_dwh(config)._statement_timeout_ms == 12_345 diff --git a/harness/tests/test_config_legacy_compat.py b/harness/tests/test_config_legacy_compat.py new file mode 100644 index 00000000..2166ad09 --- /dev/null +++ b/harness/tests/test_config_legacy_compat.py @@ -0,0 +1,148 @@ +import json +import os +import subprocess +from pathlib import Path + +import pytest +from typer.testing import CliRunner + +from tht.cli import app +from tht.config import load_config +from tht.adapters.factory import build_vector_store + + +def _write_old_workspace(tmp_path): + path = tmp_path / "old.yaml" + path.write_text( + """ +database: + host: ignored-for-rest + database: analytics + schema: mart + user: legacy-user + password: legacy-password + transport: rest +rest: + base_url: https://dwh.example.test/ + api_key: dwh-reader +vector_db: + host: vector-db + database: postgres + schema: vectors + user: vector-user + password: vector-password +vector_rest: + base_url: https://vectors.example.test/ + api_key: vector-reader +vector_write_rest: + base_url: https://vectors.example.test/ + api_key: vector-writer +paths: + artifacts: build/artifacts + indexes: build/indexes + sessions: build/sessions +""" + ) + return path + + +def _write_new_workspace(tmp_path): + path = tmp_path / "new.yaml" + path.write_text( + """ +dwh: + type: thoth_rest + database: + database: analytics + schema: mart + endpoint: + base_url: https://dwh.example.test/ + api_key: dwh-reader +vectors: + type: thoth_vector_http + reader: + base_url: https://vectors.example.test/ + api_key: vector-reader + writer: + base_url: https://vectors.example.test/ + api_key: vector-writer + direct: + host: vector-db + database: postgres + schema: vectors + user: vector-user + password: vector-password +roots: + artifacts: build/artifacts + indexes: build/indexes + sessions: build/sessions +""" + ) + return path + + +def test_legacy_rest_workspace_equals_new_resource_schema(tmp_path, capsys): + with pytest.warns(FutureWarning, match="DEPRECATION") as warnings: + old = load_config(_write_old_workspace(tmp_path)) + captured = capsys.readouterr() + new = load_config(_write_new_workspace(tmp_path)) + + assert old.dwh.model_dump() == new.dwh.model_dump() + assert old.vectors.model_dump() == new.vectors.model_dump() + assert old.roots.model_dump() == new.roots.model_dump() + assert captured.out == "" + assert captured.err == "" + assert len(warnings) == 1 + + +def test_legacy_warning_does_not_contaminate_cli_json(tmp_path): + with pytest.warns(FutureWarning, match="DEPRECATION") as warnings: + result = CliRunner().invoke( + app, + ["session", "list", "--json", "-c", str(_write_old_workspace(tmp_path))], + ) + + assert result.exit_code == 0 + json.loads(result.stdout) + assert "DEPRECATION" not in result.stdout + assert result.stderr == "" + assert len(warnings) == 1 + + +def test_legacy_cli_subprocess_warns_once_on_stderr_and_keeps_json_stdout(tmp_path): + workspace = _write_old_workspace(tmp_path) + result = subprocess.run( + [ + str(Path(__file__).parents[1] / ".venv" / "bin" / "tht"), + "session", + "list", + "--json", + "-c", + str(workspace), + ], + cwd=tmp_path, + env={**os.environ, "PYTHONWARNINGS": "default"}, + text=True, + capture_output=True, + check=False, + ) + + assert result.returncode == 0 + json.loads(result.stdout) + assert "DEPRECATION" not in result.stdout + assert result.stderr.count("DEPRECATION") == 1 + + +def test_legacy_writer_only_vector_config_builds_for_targeted_writes(tmp_path): + workspace = _write_old_workspace(tmp_path) + content = workspace.read_text().replace( + "vector_rest:\n base_url: https://vectors.example.test/\n api_key: vector-reader\n", + "", + ) + workspace.write_text(content) + + with pytest.warns(FutureWarning): + cfg = load_config(workspace) + store = build_vector_store(cfg, require_write=True) + assert store.capabilities.search is False + assert store.capabilities.upsert is True diff --git a/harness/tests/test_config_resources.py b/harness/tests/test_config_resources.py new file mode 100644 index 00000000..ab13f271 --- /dev/null +++ b/harness/tests/test_config_resources.py @@ -0,0 +1,171 @@ +import pytest + +from tht.config import ( + ConfigError, + PgvectorDirectConfig, + PostgresDwhConfig, + ThothRestDwhConfig, + ThothVectorHttpConfig, + load_config, +) +from tht.adapters.evidence import FilesystemEvidenceSource, HttpManifestEvidenceSource +from tht.adapters.factory import build_evidence_sources + + +def test_direct_vector_passwords_load_from_file_references(monkeypatch, tmp_path): + reader = tmp_path / "reader" + writer = tmp_path / "writer" + reader.write_text("reader-secret") + writer.write_text("writer-secret") + monkeypatch.setenv("READER_FILE", str(reader)) + monkeypatch.setenv("WRITER_FILE", str(writer)) + workspace = tmp_path / "workspace.yaml" + workspace.write_text(""" +dwh: + type: postgres_direct + connection: {database: d, schema: public, user: u, password: p} +vectors: + type: pgvector_direct + reader: {database: d, schema: vectors, user: r, password_file: '${READER_FILE}'} + writer: {database: d, schema: vectors, user: w, password_file: '${WRITER_FILE}'} +""") + config = load_config(workspace) + assert config.vectors.reader.password == "reader-secret" + assert config.vectors.writer.password == "writer-secret" + + +def test_direct_vector_secret_file_rejects_whitespace(tmp_path): + secret = tmp_path / "reader" + secret.write_text("bad secret") + workspace = tmp_path / "workspace.yaml" + workspace.write_text(f""" +dwh: + type: postgres_direct + connection: {{database: d, schema: public, user: u, password: p}} +vectors: + type: pgvector_direct + reader: {{database: d, schema: vectors, user: r, password_file: {secret}}} +""") + with pytest.raises(ConfigError, match="secret file"): + load_config(workspace) + + +def test_loads_discriminated_dwh_and_vector_resources(tmp_path): + workspace = tmp_path / "workspace.yaml" + workspace.write_text( + """ +dwh: + type: thoth_rest + database: + database: analytics + schema: mart + endpoint: + base_url: https://dwh.example.test/ + api_key: dwh-reader +vectors: + type: thoth_vector_http + reader: + base_url: https://vectors.example.test/ + api_key: vector-reader + writer: + base_url: https://vectors.example.test/ + api_key: vector-writer +roots: + artifacts: build/artifacts + indexes: build/indexes + sessions: build/sessions +""" + ) + + cfg = load_config(workspace) + + assert isinstance(cfg.dwh, ThothRestDwhConfig) + assert cfg.dwh.database.db_schema == "mart" + assert isinstance(cfg.vectors, ThothVectorHttpConfig) + assert cfg.vectors.writer.api_key == "vector-writer" + assert cfg.roots.sessions.as_posix() == "build/sessions" + + +def test_loads_direct_discriminated_resources(tmp_path): + workspace = tmp_path / "workspace.yaml" + workspace.write_text( + """ +dwh: + type: postgres_direct + connection: &database + host: db + database: analytics + schema: mart + user: reader + password: secret +vectors: + type: pgvector_direct + connection: + <<: *database + schema: vectors +""" + ) + + cfg = load_config(workspace) + + assert isinstance(cfg.dwh, PostgresDwhConfig) + assert cfg.database.transport == "direct" + assert isinstance(cfg.vectors, PgvectorDirectConfig) + assert cfg.vector_db.db_schema == "vectors" + + +def test_loads_writer_only_http_vector_resource(tmp_path): + workspace = tmp_path / "workspace.yaml" + workspace.write_text( + """ +dwh: + type: thoth_rest + database: {database: analytics, schema: mart} + endpoint: {base_url: https://dwh.test/, api_key: reader} +vectors: + type: thoth_vector_http + writer: {base_url: https://vectors.test/, api_key: writer} +embeddings: {base_url: http://ollama:11434, dim: 768} +""" + ) + + cfg = load_config(workspace) + assert cfg.vectors.reader is None + assert cfg.vectors.writer.api_key == "writer" + + +def test_builds_typed_evidence_sources_and_keeps_legacy_compatible(tmp_path): + common = """ +dwh: + type: postgres_direct + connection: {database: d, schema: public, user: u, password: p} +""" + modern = tmp_path / "modern.yaml" + modern.write_text(common + f""" +evidence: + sources: + - type: filesystem + root: {tmp_path} + max_bytes: 123 + - type: http + urls: ['https://example.test/doc.md?token=transport-only'] +""") + cfg = load_config(modern) + assert "transport-only" not in repr(cfg.evidence) + assert "transport-only" not in cfg.evidence.model_dump_json() + assert cfg.evidence.sources[1].allow_private_hosts is False + sources = build_evidence_sources(cfg) + assert isinstance(sources[0], FilesystemEvidenceSource) + assert isinstance(sources[1], HttpManifestEvidenceSource) + assert "transport-only" not in repr(sources[1]) + + legacy = tmp_path / "legacy.yaml" + (tmp_path / "curated").mkdir() + legacy.write_text(common + f""" +evidence: + source_root: {tmp_path} + evidence_dir: curated +""") + legacy_source = build_evidence_sources(load_config(legacy))[0] + assert isinstance(legacy_source, FilesystemEvidenceSource) + assert legacy_source.root == (tmp_path / "curated").resolve() diff --git a/harness/tests/test_corpus_chunk.py b/harness/tests/test_corpus_chunk.py new file mode 100644 index 00000000..341e13bb --- /dev/null +++ b/harness/tests/test_corpus_chunk.py @@ -0,0 +1,118 @@ +import hashlib + +import pytest + +from tht.corpus.chunk import ChunkPolicy, chunk +from tht.corpus.models import CanonicalDocument, CorpusManifest + + +def document(content: str) -> CanonicalDocument: + normalized = content.replace("\r\n", "\n").replace("\r", "\n") + digest = hashlib.sha256(normalized.encode()).hexdigest() + return CanonicalDocument( + document_id="doc:abc", + source_id="source:a", + source_uri="https://host/a.md", + source_fingerprint="etag:abc", + content_hash=f"sha256:{digest}", + title="A", + content=normalized, + media_type="text/markdown", + pipeline_version="pipe:v1", + metadata={"owner": "docs"}, + ) + + +def other_document(content: str) -> CanonicalDocument: + return document(content).model_copy( + update={ + "document_id": "doc:def", + "source_id": "source:b", + "source_uri": "https://host/b.md", + } + ) + + +def test_chunk_ids_are_stable_for_same_content_and_repeat_runs(): + policy = ChunkPolicy(version="paragraph:v1", max_chars=8) + first = chunk(document("A\n\nB"), policy) + second = chunk(document("A\r\n\r\nB"), policy) + repeated = chunk(document("A\n\nB"), policy) + assert first == second == repeated + + +def test_policy_version_changes_ids_without_changing_boundaries(): + doc = document("alpha\n\nbeta") + first = chunk(doc, ChunkPolicy(version="paragraph:v1", max_chars=6)) + second = chunk(doc, ChunkPolicy(version="paragraph:v2", max_chars=6)) + assert [item.content for item in first] == [item.content for item in second] + assert [item.chunk_id for item in first] != [item.chunk_id for item in second] + + +def test_same_policy_version_with_different_boundary_config_changes_ids(): + doc = document("alpha beta") + first = chunk(doc, ChunkPolicy(version="paragraph:v1", max_chars=6)) + second = chunk(doc, ChunkPolicy(version="paragraph:v1", max_chars=7)) + assert first[0].chunk_id != second[0].chunk_id + + +def test_identical_content_in_different_documents_cannot_collide_in_manifest(): + policy = ChunkPolicy(version="paragraph:v1", max_chars=20) + first = document("same") + second = other_document("same") + chunks = [*chunk(first, policy), *chunk(second, policy)] + manifest = CorpusManifest( + pipeline_version="pipe:v1", documents=[first, second], chunks=chunks + ) + assert len({item.chunk_id for item in manifest.chunks}) == 2 + + +def test_long_non_ascii_tokens_are_hard_split_by_unicode_characters(): + chunks = chunk(document("ééééé世界"), ChunkPolicy(version="chars:v1", max_chars=3)) + assert [item.content for item in chunks] == ["ééé", "éé世", "界"] + assert all(len(item.content) <= 3 for item in chunks) + + +@pytest.mark.parametrize( + "content", + [ + "alpha beta\tgamma\n\ndelta", + "line with markdown hard break \nnext line\n```\na b\n```", + " \t\n\n \n", + "supercalifragilisticexpialidocious", + "é 世界\r\nnext", + ], +) +def test_chunks_preserve_every_character_and_respect_max_chars(content): + doc = document(content) + chunks = chunk(doc, ChunkPolicy(version="exact:v1", max_chars=9)) + assert "".join(item.content for item in chunks) == doc.content + assert all(0 < len(item.content) <= 9 for item in chunks) + + +def test_chunks_have_contiguous_ordinals_hashes_and_provenance_metadata(): + doc = document("alpha beta gamma") + chunks = chunk(doc, ChunkPolicy(version="words:v1", max_chars=7)) + assert [item.ordinal for item in chunks] == list(range(len(chunks))) + assert len({item.chunk_id for item in chunks}) == len(chunks) + for item in chunks: + assert item.source_uri == doc.source_uri + assert item.document_id == doc.document_id + assert item.pipeline_version == doc.pipeline_version + assert item.metadata["chunk_policy"]["max_chars"] == 7 + assert item.metadata["chunk_policy"]["version"] == "words:v1" + assert item.metadata["chunk_policy"]["fingerprint"].startswith("sha256:") + assert item.metadata["document"] == {"owner": "docs"} + assert item.content_hash == "sha256:" + hashlib.sha256(item.content.encode()).hexdigest() + + +def test_duplicate_chunk_content_cannot_collide_across_ordinals(): + chunks = chunk(document("samesame"), ChunkPolicy(version="paragraph:v1", max_chars=4)) + assert [item.content for item in chunks] == ["same", "same"] + assert chunks[0].chunk_id != chunks[1].chunk_id + + +def test_empty_document_has_no_chunks_and_invalid_policy_is_rejected(): + assert chunk(document(""), ChunkPolicy(version="v1", max_chars=4)) == [] + with pytest.raises(ValueError): + ChunkPolicy(version="v1", max_chars=0) diff --git a/harness/tests/test_corpus_models.py b/harness/tests/test_corpus_models.py new file mode 100644 index 00000000..0e9ee673 --- /dev/null +++ b/harness/tests/test_corpus_models.py @@ -0,0 +1,217 @@ +import hashlib +from datetime import UTC, datetime, timedelta, timezone + +import pytest +from pydantic import ValidationError + +from tht.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest + + +def document(source_uri: str = "https://host/a.md") -> CanonicalDocument: + content = "# A" + return CanonicalDocument( + document_id="doc:abc", + source_id="source:a", + source_uri=source_uri, + source_fingerprint="etag:abc", + content_hash=f"sha256:{hashlib.sha256(content.encode()).hexdigest()}", + title="A", + content=content, + media_type="text/markdown", + pipeline_version="evidence-v1", + ) + + +def chunk() -> CanonicalChunk: + content = "# A" + return CanonicalChunk( + chunk_id="chunk:abc:0", + document_id="doc:abc", + ordinal=0, + content=content, + content_hash=f"sha256:{hashlib.sha256(content.encode()).hexdigest()}", + source_uri="https://host/a.md", + pipeline_version="evidence-v1", + ) + + +def test_manifest_contains_provenance_without_credentials(): + manifest = CorpusManifest( + manifest_id="manifest:abc", + created_at=datetime(2026, 7, 12, tzinfo=UTC), + pipeline_version="evidence-v1", + embedding_model="nomic-embed-text", + embedding_dimensions=768, + documents=[document()], + chunks=[chunk()], + ) + + payload = manifest.model_dump_json() + + assert "https://host/a.md" in payload + assert "etag:abc" in payload + assert "evidence-v1" in payload + assert "nomic-embed-text" in payload + assert "api_key" not in payload + + +def test_manifest_can_be_assembled_before_publish_identifiers_are_assigned(): + manifest = CorpusManifest(documents=[document()]) + + assert manifest.documents[0].source_uri == "https://host/a.md" + assert manifest.manifest_id is None + + +@pytest.mark.parametrize("model", [document(), chunk()]) +def test_canonical_records_are_frozen(model): + with pytest.raises(ValidationError): + model.pipeline_version = "changed" # type: ignore[misc] + + +def test_manifest_collections_are_immutable_tuples_with_json_arrays(): + first = CorpusManifest( + manifest_id="manifest:one", created_at=datetime.now(UTC), pipeline_version="v1" + ) + second = CorpusManifest( + manifest_id="manifest:two", created_at=datetime.now(UTC), pipeline_version="v1" + ) + + with pytest.raises(AttributeError): + first.documents.append(document()) + assert first.documents == () + assert second.documents == () + assert '"documents":[]' in first.model_dump_json() + + +def test_manifest_validates_embedding_compatibility_fields(): + with pytest.raises(ValidationError): + CorpusManifest( + manifest_id="bad", + created_at=datetime.now(UTC), + pipeline_version="v1", + embedding_dimensions=0, + ) + + +def test_canonical_metadata_rejects_secrets_and_non_json_values(): + with pytest.raises(ValidationError, match="credential-like"): + CanonicalDocument.model_validate( + {**document().model_dump(), "metadata": {"password": "secret"}} + ) + with pytest.raises(ValidationError): + CanonicalChunk( + chunk_id="c", + document_id="d", + ordinal=0, + content="x", + content_hash="sha256:x", + source_uri="file:///x", + pipeline_version="v1", + metadata={"bad": object()}, + ) + + +def test_manifest_rejects_duplicate_ids_and_source_ids(): + first = document() + duplicate_source = first.model_copy( + update={"document_id": "doc:other", "source_uri": "https://host/b.md"} + ) + with pytest.raises(ValidationError, match="source_id"): + CorpusManifest(pipeline_version="evidence-v1", documents=[first, duplicate_source]) + + with pytest.raises(ValidationError, match="chunk_id"): + CorpusManifest( + pipeline_version="evidence-v1", documents=[first], chunks=[chunk(), chunk()] + ) + + +def test_manifest_rejects_orphan_noncontiguous_and_inconsistent_chunks(): + with pytest.raises(ValidationError, match="unknown document"): + CorpusManifest(pipeline_version="evidence-v1", chunks=[chunk()]) + + second = chunk().model_copy(update={"chunk_id": "chunk:abc:2", "ordinal": 2}) + with pytest.raises(ValidationError, match="contiguous"): + CorpusManifest( + pipeline_version="evidence-v1", documents=[document()], chunks=[chunk(), second] + ) + + wrong_uri = chunk().model_copy(update={"source_uri": "https://host/wrong.md"}) + with pytest.raises(ValidationError, match="source_uri"): + CorpusManifest( + pipeline_version="evidence-v1", documents=[document()], chunks=[wrong_uri] + ) + + +def test_manifest_rejects_inconsistent_pipeline_versions(): + wrong = document().model_copy(update={"pipeline_version": "other-v1"}) + with pytest.raises(ValidationError, match="pipeline_version"): + CorpusManifest(pipeline_version="evidence-v1", documents=[wrong]) + + +def test_vector_generation_requires_embedding_compatibility(): + with pytest.raises(ValidationError, match="vector_generation"): + CorpusManifest(pipeline_version="evidence-v1", vector_generation="generation:one") + + +@pytest.mark.parametrize( + ("field", "value"), + [ + ("document_id", "not-namespaced"), + ("content_hash", "sha256:not-hex"), + ("source_uri", "https://user:pass@host/a"), + ], +) +def test_canonical_document_rejects_malformed_or_sensitive_provenance(field, value): + with pytest.raises(ValidationError): + CanonicalDocument.model_validate({**document().model_dump(), field: value}) + + +@pytest.mark.parametrize( + "source_uri", + [ + "https://host/a?X-Amz-Credential=abc&X-Amz-Signature=secret#access_token=bad", + "https://host/a?sig=sas-secret&sp=r#section", + ], +) +def test_canonical_provenance_strips_query_and_fragment(source_uri): + doc = document(source_uri=source_uri) + canonical_chunk = chunk().model_copy(update={"source_uri": source_uri}) + manifest = CorpusManifest( + pipeline_version="evidence-v1", documents=[doc], chunks=[canonical_chunk] + ) + + assert doc.source_uri == "https://host/a" + assert canonical_chunk.source_uri == "https://host/a" + payload = manifest.model_dump_json() + assert "X-Amz" not in payload + assert "sas-secret" not in payload + assert "access_token" not in payload + + +@pytest.mark.parametrize("factory", [document, chunk]) +def test_content_hash_must_match_exact_canonical_utf8(factory): + record = factory() + with pytest.raises(ValidationError, match="exact canonical UTF-8 content"): + type(record).model_validate({**record.model_dump(), "content": record.content + "\n"}) + + +def test_model_copy_revalidates_records_and_manifests(): + with pytest.raises(ValidationError, match="namespaced"): + document().model_copy(update={"document_id": "invalid"}) + manifest = CorpusManifest( + pipeline_version="evidence-v1", + embedding_model="embed-v1", + embedding_dimensions=768, + ) + with pytest.raises(ValidationError, match="set together"): + manifest.model_copy(update={"embedding_dimensions": None}) + + +def test_manifest_datetimes_are_aware_and_normalized_to_utc(): + with pytest.raises(ValidationError, match="timezone-aware"): + CorpusManifest(created_at=datetime(2026, 7, 12), pipeline_version="evidence-v1") + + plus_two = datetime(2026, 7, 12, 12, tzinfo=timezone(timedelta(hours=2))) + manifest = CorpusManifest(created_at=plus_two, pipeline_version="evidence-v1") + assert manifest.created_at.tzinfo is UTC + assert manifest.created_at.hour == 10 diff --git a/harness/tests/test_corpus_normalize.py b/harness/tests/test_corpus_normalize.py new file mode 100644 index 00000000..67bc41e7 --- /dev/null +++ b/harness/tests/test_corpus_normalize.py @@ -0,0 +1,90 @@ +import hashlib +from datetime import UTC, datetime + +import pytest + +from tht.corpus.normalize import MAX_DOCUMENT_BYTES, PermanentNormalizationError, normalize +from tht.ports.evidence import AcquiredDocument, SourceObject + + +def acquired(content: bytes, *, media_type: str = "text/markdown") -> AcquiredDocument: + return AcquiredDocument( + source=SourceObject( + source_id="source:guide", + uri="https://host/guide.md?signature=transport#part", + fingerprint="etag:abc", + modified_at=datetime(2026, 7, 12, 12, 0, tzinfo=UTC), + metadata={"owner": "docs"}, + ), + content=content, + media_type=media_type, + metadata={"transport": "http"}, + ) + + +def test_normalize_utf8_bom_newlines_unicode_and_frontmatter(): + raw = ( + "\ufeff---\r\ntitle: Café\r\ntags: [uno, due]\r\n---\r\n" + "Cafe\u0301\rBody\r\n" + ).encode() + + document = normalize(acquired(raw, media_type="text/markdown; charset=UTF-8"), "pipe:v1") + + assert document.content == "Café\nBody\n" + assert document.title == "Café" + assert document.metadata["frontmatter"] == {"tags": ("uno", "due"), "title": "Café"} + assert document.metadata["source"] == {"owner": "docs"} + assert document.metadata["acquisition"] == {"transport": "http"} + assert document.source_uri == "https://host/guide.md" + assert document.modified_at == datetime(2026, 7, 12, 12, 0, tzinfo=UTC) + assert document.content_hash == "sha256:" + hashlib.sha256(document.content.encode()).hexdigest() + + +def test_plain_text_that_only_resembles_frontmatter_is_not_dropped(): + document = normalize(acquired(b"---\nnot: closed\nbody"), "pipe:v1") + assert document.content == "---\nnot: closed\nbody" + assert "frontmatter" not in document.metadata + + +def test_frontmatter_can_end_at_eof_without_inventing_content(): + document = normalize(acquired(b"---\ntitle: Empty\n---"), "pipe:v1") + assert document.title == "Empty" + assert document.content == "" + + +@pytest.mark.parametrize( + "frontmatter", + [ + "title: first\ntitle: second", + "title: &shared value\ncopy: *shared", + "nested: " + "[" * 25 + "x" + "]" * 25, + "items: [" + ",".join("x" for _ in range(1100)) + "]", + "api_key: secret", + ], +) +def test_rejects_unsafe_frontmatter_as_typed_permanent_error(frontmatter): + raw = f"---\n{frontmatter}\n---\nbody".encode() + with pytest.raises(PermanentNormalizationError) as caught: + normalize(acquired(raw), "pipe:v1") + assert caught.value.reason == "invalid_frontmatter" + + +def test_pipeline_policy_errors_are_not_misclassified_as_bad_frontmatter(): + with pytest.raises(ValueError, match="pipeline_version") as caught: + normalize(acquired(b"---\ntitle: valid\n---\nbody"), "") + assert not isinstance(caught.value, PermanentNormalizationError) + + +@pytest.mark.parametrize( + ("content", "media_type", "reason"), + [ + (b"bad: \xff", "text/plain", "undecodable"), + (b"hello", "text/plain; charset=iso-8859-1", "unsupported_charset"), + (b"x" * (MAX_DOCUMENT_BYTES + 1), "text/plain", "oversized"), + ], +) +def test_rejects_invalid_input_as_typed_permanent_error(content, media_type, reason): + with pytest.raises(PermanentNormalizationError) as caught: + normalize(acquired(content, media_type=media_type), "pipe:v1") + assert caught.value.permanent is True + assert caught.value.reason == reason diff --git a/harness/tests/test_corpus_pipeline.py b/harness/tests/test_corpus_pipeline.py new file mode 100644 index 00000000..2e62499d --- /dev/null +++ b/harness/tests/test_corpus_pipeline.py @@ -0,0 +1,902 @@ +from datetime import UTC, datetime, timedelta + +import pytest + +from tht.corpus.chunk import ChunkPolicy +from tht.corpus.pipeline import CorpusPipeline, PipelineError, PipelineResult +from tht.corpus.store import CorpusStore +from tht.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest +from tht.ports.evidence import AcquiredDocument, SourceObject +from tht.ports.vector import VectorCapabilities, VectorHealth + + +class Source: + def __init__(self, documents): + self.documents = documents + self.acquire_calls = [] + + def discover(self): + return [item[0] for item in self.documents] + + def acquire(self, item): + self.acquire_calls.append(item.source_id) + payload = next(payload for source, payload in self.documents if source.source_id == item.source_id) + if isinstance(payload, Exception): + raise payload + return AcquiredDocument( + source=item, content=payload.encode(), media_type=item.metadata.get("media_type") + ) + + +class Embedder: + def __init__(self, dim=3, fail=False): + self.dim = dim + self.fail = fail + self.calls = [] + + def embed_documents(self, texts): + self.calls.extend(texts) + if self.fail: + raise RuntimeError("embed failed") + return [[float(i) for i in range(self.dim)] for _ in texts] + + +class Vectors: + capabilities = VectorCapabilities(search=True, existing_hashes=True, upsert=True) + + def __init__(self, fail=False): + self.fail = fail + self.records = [] + self.dimension = 3 + + def upsert(self, collection, records): + self.records.extend(records[:1] if self.fail else records) + if self.fail: + raise RuntimeError("partial write") + return len(records) + + def existing_hashes(self, collection, kinds): + return { + value.record.id: value.content_hash for value in self.records + } + + def health(self): + return VectorHealth( + ok=True, expected_dimension=3, observed_dimensions=(self.dimension,), + dimension_compatible=self.dimension == 3, + ) + + def delete_generation(self, collection, generation, workspace_id): + self.records = [ + value for value in self.records + if not (value.record.metadata["vector_generation"] == generation + and value.record.metadata.get("workspace_id") == workspace_id) + ] + return 0 + + def list_evidence_generations(self, collection, workspace_id): + return sorted({ + value.record.metadata["vector_generation"] for value in self.records + if value.record.kind == "evidence" + and value.record.metadata.get("workspace_id") == workspace_id + }) + + +class InterruptingVectors(Vectors): + def __init__(self): + super().__init__() + self.batches = [] + self.interrupt = True + + def upsert(self, collection, records): + self.batches.append([value.record.id for value in records]) + if self.interrupt: + self.interrupt = False + self.records.append(records[0]) + raise KeyboardInterrupt("process interruption after partial write") + self.records.extend(records) + return len(records) + + +def item(name, fingerprint): + return SourceObject( + source_id=f"fs:{name}", uri=f"file:///safe/{name}.md", fingerprint=f"sha256:{fingerprint}" + ) + + +def pipeline(tmp_path, source, *, embedder=None, vectors=None, model="model-a", policy=None, + retain=3): + return CorpusPipeline( + store=CorpusStore(tmp_path / "corpus"), sources=[source], + embedder=embedder or Embedder(), vector_store=vectors or Vectors(), + embedding_model=model, embedding_dimensions=3, + chunk_policy=policy or ChunkPolicy(version="chunk-v1", max_chars=100), + pipeline_version="evidence-v1", + retain_published_generations=retain, + ) + + +def test_retention_bounds_generations_and_purges_vectors_after_publish(tmp_path): + vectors = Vectors() + generations = [] + for index in range(4): + result = pipeline( + tmp_path, Source([(item("one", str(index)), f"version {index}")]), + vectors=vectors, retain=2, + ).run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + str(index) * 64, + ) + generations.append(result.generation) + store = CorpusStore(tmp_path / "corpus") + assert store.list_generations() == generations[-2:] + assert {r.record.metadata["vector_generation"] for r in vectors.records} == set(generations[-2:]) + assert store.active_generation() == generations[-1] + + +def test_retention_keeps_filesystem_when_vector_purge_fails_then_retries(tmp_path): + class FailingDelete(Vectors): + def __init__(self): + super().__init__() + self.fail_delete = True + + def delete_generation(self, collection, generation, workspace_id): + if self.fail_delete: + raise RuntimeError("credential secret") + return super().delete_generation(collection, generation, workspace_id) + + vectors = FailingDelete() + for index in range(2): + pipeline(tmp_path, Source([(item("one", str(index)), str(index))]), vectors=vectors, + retain=1).run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + str(index) * 64, + ) + assert len(CorpusStore(tmp_path / "corpus").list_generations()) == 2 + vectors.fail_delete = False + report = pipeline(tmp_path, Source([(item("one", "1"), "1")]), vectors=vectors, + retain=1).gc(workspace_root=tmp_path) + assert report["status"] == "succeeded" + assert len(CorpusStore(tmp_path / "corpus").list_generations()) == 1 + + +def test_gc_reconciles_vector_only_generation(tmp_path): + vectors = Vectors() + orphan = "gen:" + "f" * 32 + from tht.ports.vector import VectorWriteRecord + from tht.vectorstore.records import VectorRecord + vectors.records.append(VectorWriteRecord( + record=VectorRecord(id="orphan", kind="evidence", ref="doc:x", title="", content="x", + metadata={"vector_generation": orphan, "workspace_id": "default"}), + embedding=[0.0, 0.0, 0.0], content_hash="sha256:" + "0" * 64, + )) + candidate = pipeline(tmp_path, Source([]), vectors=vectors, retain=1) + report = candidate.gc(workspace_root=tmp_path) + assert report["evicted"] == [orphan] + assert vectors.list_evidence_generations("evidence", "default") == [] + assert candidate.gc(workspace_root=tmp_path)["evicted"] == [] + + +@pytest.mark.parametrize("status", ["running", "failed"]) +def test_gc_protects_generations_referenced_by_resumable_checkpoints(tmp_path, status): + generation = "gen:" + "e" * 32 + store = CorpusStore(tmp_path / "corpus") + store.stage(CorpusManifest(), {}, generation=generation) + run = tmp_path / ".tht-jobs" / "evidence" / "runs" / ("a" * 32) + (run / "artifacts").mkdir(parents=True) + (run / "checkpoint.json").write_text(__import__("json").dumps({"status": status})) + (run / "artifacts" / "plan.json").write_text( + __import__("json").dumps({"generation": generation}) + ) + candidate = pipeline(tmp_path, Source([]), vectors=Vectors(), retain=1) + report = candidate.gc(workspace_root=tmp_path) + assert generation in report["protected"] + assert store.generation_path(generation).exists() + + +def test_explicit_gc_blocks_while_job_holds_corpus_writer_lock(tmp_path): + import threading + + candidate = pipeline(tmp_path, Source([(item("one", "a"), "one")]), vectors=Vectors()) + entered = threading.Event() + release = threading.Event() + gc_finished = threading.Event() + + def pause(_context, stage): + if stage == "discover": + entered.set() + assert release.wait(5) + + job = threading.Thread(target=lambda: candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + after_stage_return=pause, + )) + job.start() + assert entered.wait(5) + + def collect(): + candidate.gc(workspace_root=tmp_path) + gc_finished.set() + + gc_thread = threading.Thread(target=collect) + gc_thread.start() + assert not gc_finished.wait(0.1) + release.set() + job.join(5) + gc_thread.join(5) + assert gc_finished.is_set() + assert candidate.store.active_generation() is not None + + +def test_gc_preserves_vector_dependencies_of_retained_manifests(tmp_path): + vectors = Vectors() + one = item("one", "a") + first = pipeline(tmp_path, Source([(one, "stable")]), vectors=vectors, retain=2).run().generation + second = pipeline( + tmp_path, Source([(one, "stable"), (item("two", "b"), "two")]), + vectors=vectors, retain=2, + ).run().generation + third = pipeline( + tmp_path, Source([(one, "stable"), (item("two", "c"), "changed")]), + vectors=vectors, retain=2, + ).run().generation + assert CorpusStore(tmp_path / "corpus").list_generations() == [second, third] + assert first in vectors.list_evidence_generations("evidence", "default") + + +def test_active_searcher_without_active_fails_closed_for_evidence(tmp_path): + from types import SimpleNamespace + from tht.search.evidence import active_searcher + + class Delegate: + def search(self, embedding, top_n=10, kinds=None, metadata_filter=None): + return ["legacy"] + + cfg = SimpleNamespace(paths=SimpleNamespace(artifacts=tmp_path / "artifacts")) + wrapped = active_searcher(cfg, Delegate()) + assert wrapped.search([1.0], kinds=["evidence"]) == [] + assert wrapped.search([1.0], kinds=["memory"]) == ["legacy"] + + +def test_active_searcher_splits_default_and_mixed_kinds_before_global_limit(tmp_path): + from types import SimpleNamespace + from tht.search.evidence import ActiveEvidenceSearcher + + store = CorpusStore(tmp_path / "corpus") + generation = store.stage( + CorpusManifest(metadata={"workspace_id": "default"}), {}, + generation="gen:" + "a" * 32, + ) + store.publish(generation) + calls = [] + + class Delegate: + def search(self, embedding, top_n=10, kinds=None, metadata_filter=None): + calls.append((kinds, metadata_filter)) + if kinds == ["evidence"]: + return [SimpleNamespace(id="active", similarity=0.8)] + return [SimpleNamespace(id="memory", similarity=0.9)] + + searcher = ActiveEvidenceSearcher(store, Delegate()) + hits = searcher.search([1.0], top_n=1, kinds=["evidence", "memory"]) + assert [hit.id for hit in hits] == ["memory"] + assert calls[0] == (["memory"], None) + # Empty manifest means no Evidence query, but the split remains explicit and safe. + assert all(call[0] != ["evidence"] for call in calls) + calls.clear() + searcher.search([1.0], top_n=1) + assert calls[0][0] == ["memory", "schema_column", "schema_table", "solved_question"] + assert all(call[0] is not None for call in calls) + + +def test_active_evidence_query_holds_lock_against_publish(tmp_path): + import threading + from types import SimpleNamespace + from tht.search.evidence import ActiveEvidenceSearcher + + first_pipeline = pipeline(tmp_path, Source([(item("one", "a"), "old")]), vectors=Vectors()) + first_pipeline.run() + store = first_pipeline.store + entered = threading.Event() + release = threading.Event() + published = threading.Event() + + class Delegate: + def search(self, embedding, top_n=10, kinds=None, metadata_filter=None): + entered.set() + assert release.wait(5) + return [SimpleNamespace(id="active", similarity=1.0)] + + search = threading.Thread( + target=lambda: ActiveEvidenceSearcher(store, Delegate()).search( + [1.0], kinds=["evidence"] + ) + ) + search.start() + assert entered.wait(5) + next_generation = store.stage(CorpusManifest(), {}) + + def publish(): + with store.writer_lock(): + store.publish(next_generation) + published.set() + + publisher = threading.Thread(target=publish) + publisher.start() + assert not published.wait(0.1) + release.set() + search.join(5) + publisher.join(5) + assert published.is_set() + + +def test_pipeline_result_dump_does_not_deepcopy_frozen_metadata(): + manifest = CorpusManifest(metadata={"nested": {"value": ["safe"]}}) + payload = PipelineResult( + "succeeded", None, False, (), (), (), manifest, + ).model_dump(mode="json") + assert payload["manifest_id"] is None + assert "manifest" not in payload + + +def test_pipeline_result_public_dump_is_bounded_and_excludes_evidence_content(tmp_path): + import json + + result = pipeline( + tmp_path, Source([(item("one", "a"), "SENSITIVE EVIDENCE CONTENT")]) + ).run() + payload = result.model_dump(mode="json") + encoded = json.dumps(payload) + assert "SENSITIVE EVIDENCE CONTENT" not in encoded + assert "documents" not in payload and "chunks" not in payload + large = PipelineResult( + "failed", None, False, + tuple(f"fs:item-{index}" for index in range(1000)), (), (), result.manifest, + ).model_dump(mode="json") + assert len(large["changed"]) == 100 + assert large["counts"]["changed"] == 1000 + assert len(json.dumps(large)) < 25_000 + + +def test_pipeline_result_repr_is_bounded_and_excludes_manifest_secrets(): + secret = "TOP_SECRET_CONTENT" + manifest = CorpusManifest.model_construct( + manifest_id="gen:" + "a" * 64, + documents=tuple(CanonicalDocument.model_construct(content=secret) for _ in range(1000)), + chunks=tuple(CanonicalChunk.model_construct(content=secret) for _ in range(1000)), + metadata={"password": secret, "credential": "Bearer " + secret}, + ) + result = PipelineResult( + "succeeded", "gen:" + "a" * 64, True, (), (), (), manifest, + run_id="b" * 32, + ) + + rendered = repr(result) + assert str(result) == rendered + assert len(rendered) < 1000 + assert secret not in rendered + assert "password" not in rendered + assert "credential" not in rendered + assert "manifest" not in rendered.lower() + assert "documents" not in rendered + assert "chunks" not in rendered + + +def test_reused_corpus_root_rejects_workspace_rename_before_any_mutation(tmp_path): + vectors = Vectors() + first = pipeline(tmp_path, Source([(item("one", "a"), "stable")]), vectors=vectors) + first.run_as_job( + workspace_id="workspace-a", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + active = first.store.active_generation() + records = list(vectors.records) + renamed_source = Source([(item("one", "a"), "stable")]) + renamed = pipeline(tmp_path, renamed_source, vectors=vectors) + with pytest.raises(PipelineError, match="different workspace"): + renamed.run_as_job( + workspace_id="workspace-b", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + assert renamed_source.acquire_calls == [] + assert renamed.store.active_generation() == active + assert vectors.records == records + + +def test_gc_rejects_workspace_mismatch_without_deleting(tmp_path): + vectors = Vectors() + owner = pipeline(tmp_path, Source([(item("one", "a"), "stable")]), vectors=vectors) + owner.run_as_job( + workspace_id="workspace-a", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + generations = owner.store.list_generations() + wrong = pipeline(tmp_path, Source([]), vectors=vectors) + wrong.workspace_id = "workspace-b" + with pytest.raises(PipelineError, match="different workspace"): + wrong.gc(workspace_root=tmp_path) + assert wrong.store.list_generations() == generations + + +@pytest.mark.parametrize("kinds", [None, ["evidence", "memory"], ["memory"]]) +def test_active_search_rejects_workspace_mismatch_before_delegate(tmp_path, kinds): + from tht.search.evidence import ActiveEvidenceSearcher, CorpusWorkspaceMismatchError + + vectors = Vectors() + owner = pipeline(tmp_path, Source([(item("one", "a"), "stable")]), vectors=vectors) + owner.run_as_job( + workspace_id="workspace-a", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + + class Delegate: + def search(self, *args, **kwargs): + raise AssertionError("workspace mismatch reached vector delegate") + + with pytest.raises(CorpusWorkspaceMismatchError, match="different workspace"): + ActiveEvidenceSearcher( + owner.store, Delegate(), expected_workspace_id="workspace-b", + ).search([1.0], kinds=kinds) + + +def test_unscoped_active_manifest_is_never_adopted_by_direct_run_or_gc(tmp_path): + store = CorpusStore(tmp_path / "corpus") + generation = store.stage(CorpusManifest(), {}) + store.publish(generation) + + class ForbiddenSource: + def discover(self): + raise AssertionError("invalid corpus reached source discovery") + + class ForbiddenVectors: + def __getattr__(self, name): + raise AssertionError(f"invalid corpus reached vector operation {name}") + + candidate = CorpusPipeline( + store=store, sources=[ForbiddenSource()], embedder=Embedder(), + vector_store=ForbiddenVectors(), embedding_model="model", embedding_dimensions=3, + chunk_policy=ChunkPolicy(version="chunk-v1", max_chars=100), + pipeline_version="evidence-v1", + ) + with pytest.raises(PipelineError, match="missing or invalid"): + candidate.run() + with pytest.raises(PipelineError, match="missing or invalid"): + candidate.gc(workspace_root=tmp_path) + assert store.active_generation() == generation + + +def test_unchanged_documents_skip_acquire_normalize_chunk_and_embed(tmp_path): + one = item("one", "a") + first_source = Source([(one, "hello")]) + first = pipeline(tmp_path, first_source) + first.run() + second_source = Source([(one, "ignored")]) + second_embedder = Embedder() + result = pipeline(tmp_path, second_source, embedder=second_embedder).run() + assert result.unchanged == ("fs:one",) + assert second_source.acquire_calls == [] + assert second_embedder.calls == [] + + +def test_unchanged_job_reuses_active_generation_without_new_directory(tmp_path): + source = Source([( + SourceObject( + source_id="fs:one", uri="file:///safe/one.md", fingerprint="sha256:a", + modified_at=datetime(2026, 1, 1, tzinfo=UTC), + metadata={"media_type": "text/markdown", "size": 5, "nested": {"b": 2, "a": 1}}, + ), + "hello", + )]) + candidate = pipeline(tmp_path, source) + args = dict(workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64) + first = candidate.run_as_job(**args) + snapshot = first.manifest.metadata["source_snapshot"]["fs:one"] + assert snapshot == { + "source_id": "fs:one", "uri": "file:///safe/one.md", "fingerprint": "sha256:a", + "modified_at": "2026-01-01T00:00:00Z", + "metadata": {"media_type": "text/markdown", "size": 5, + "nested": {"a": 1, "b": 2}}, + "media_type": "text/markdown", "size": 5, + } + count = len(candidate.store.list_generations()) + second = candidate.run_as_job(**args) + assert second.generation == first.generation + assert second.published is False + assert len(candidate.store.list_generations()) == count + + +@pytest.mark.parametrize("field", ["uri", "modified_at", "metadata"]) +def test_job_source_snapshot_change_forces_publish_with_same_fingerprint(tmp_path, field): + original = SourceObject( + source_id="fs:one", uri="file:///safe/one.md", fingerprint="sha256:a", + modified_at=datetime(2026, 1, 1, tzinfo=UTC), + metadata={"media_type": "text/markdown", "size": 5, "label": "original"}, + ) + vectors = Vectors() + args = dict(workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64) + first = pipeline(tmp_path, Source([(original, "hello")]), vectors=vectors).run_as_job(**args) + updates = { + "uri": "file:///safe/renamed.md", + "modified_at": original.modified_at + timedelta(seconds=1), + "metadata": {"media_type": "text/markdown", "size": 5, "label": "changed"}, + } + changed = original.model_copy(update={field: updates[field]}) + source = Source([(changed, "hello")]) + result = pipeline(tmp_path, source, vectors=vectors).run_as_job(**args) + assert result.published is True + assert result.generation != first.generation + assert source.acquire_calls == ["fs:one"] + + +@pytest.mark.parametrize("fingerprint_name", ["config_fingerprint", "input_fingerprint"]) +def test_job_binding_change_forces_publish(tmp_path, fingerprint_name): + vectors = Vectors() + source_object = item("one", "a") + args = dict(workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64) + first = pipeline(tmp_path, Source([(source_object, "hello")]), vectors=vectors).run_as_job(**args) + args[fingerprint_name] = "sha256:" + "3" * 64 + source = Source([(source_object, "hello")]) + result = pipeline(tmp_path, source, vectors=vectors).run_as_job(**args) + assert result.published is True + assert result.generation != first.generation + assert source.acquire_calls == [] + + +@pytest.mark.parametrize( + "damage", ["legacy_metadata", "corrupt_document_sources", "document", "vector"] +) +def test_job_incomplete_active_contract_never_noops(tmp_path, damage): + import json + + vectors = Vectors() + source_object = item("one", "a") + args = dict(workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64) + candidate = pipeline(tmp_path, Source([(source_object, "hello")]), vectors=vectors) + first = candidate.run_as_job(**args) + if damage == "legacy_metadata": + manifest_path = candidate.store.generation_path(first.generation) / "manifest.json" + payload = json.loads(manifest_path.read_text()) + payload["metadata"].pop("source_snapshot") + manifest_path.write_text(json.dumps(payload)) + elif damage == "corrupt_document_sources": + manifest_path = candidate.store.generation_path(first.generation) / "manifest.json" + payload = json.loads(manifest_path.read_text()) + payload["metadata"]["document_sources"] = {} + manifest_path.write_text(json.dumps(payload)) + elif damage == "document": + path = candidate.store.resolve_document(first.manifest.documents[0].document_id) + path.unlink() + else: + vectors.records.clear() + source = Source([(source_object, "hello")]) + result = pipeline(tmp_path, source, vectors=vectors).run_as_job(**args) + assert result.published is True + assert result.generation != first.generation + assert source.acquire_calls == ["fs:one"] + + +@pytest.mark.parametrize( + "damage", ["modified_at", "source_metadata", "media_type", "missing_chunk", + "altered_chunk", "extra_chunk", "vector_dimension"] +) +def test_job_corrupt_canonical_document_or_chunk_never_noops(tmp_path, damage): + import hashlib + import json + + vectors = Vectors() + source_object = SourceObject( + source_id="fs:one", uri="file:///safe/one.md", fingerprint="sha256:a", + modified_at=datetime(2026, 1, 1, tzinfo=UTC), + metadata={"media_type": "text/markdown", "size": 11, "owner": "docs"}, + ) + args = dict(workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64) + candidate = pipeline( + tmp_path, Source([(source_object, "hello world")]), vectors=vectors, + policy=ChunkPolicy(version="chunk-v1", max_chars=6), + ) + first = candidate.run_as_job(**args) + manifest_path = candidate.store.generation_path(first.generation) / "manifest.json" + payload = json.loads(manifest_path.read_text()) + document = payload["documents"][0] + chunks = payload["chunks"] + if damage == "modified_at": + document["modified_at"] = "2026-01-01T00:00:01Z" + elif damage == "source_metadata": + document["metadata"]["source"]["owner"] = "attacker" + elif damage == "media_type": + document["media_type"] = "text/plain" + elif damage == "missing_chunk": + payload["chunks"] = chunks[:-1] + elif damage == "altered_chunk": + chunks[0]["content"] = "HELLO " + chunks[0]["content_hash"] = "sha256:" + hashlib.sha256(b"HELLO ").hexdigest() + chunks[0]["chunk_id"] = "chunk:" + "a" * 64 + elif damage == "extra_chunk": + extra = CanonicalChunk( + chunk_id="chunk:" + "b" * 64, document_id=document["document_id"], + ordinal=len(chunks), content="", content_hash="sha256:" + hashlib.sha256(b"").hexdigest(), + source_uri=document["source_uri"], pipeline_version=document["pipeline_version"], + ) + chunks.append(extra.model_dump(mode="json")) + else: + vectors.dimension = 4 + manifest_path.write_text(json.dumps(payload)) + + source = Source([(source_object, "hello world")]) + result = pipeline( + tmp_path, source, vectors=vectors, + policy=ChunkPolicy(version="chunk-v1", max_chars=6), + ).run_as_job(**args) + assert result.published is True + assert result.generation != first.generation + assert source.acquire_calls == ["fs:one"] + + +def test_removed_documents_are_marked_and_absent_from_new_manifest(tmp_path): + one, two = item("one", "a"), item("two", "b") + pipeline(tmp_path, Source([(one, "one"), (two, "two")])).run() + result = pipeline(tmp_path, Source([(one, "one")])).run() + assert result.removed == ("fs:two",) + assert {doc.source_id for doc in result.manifest.documents} == {"fs:one"} + + +def test_model_or_chunk_policy_change_forces_full_rebuild(tmp_path): + one = item("one", "a") + pipeline(tmp_path, Source([(one, "hello")])).run() + source = Source([(one, "hello")]) + changed = pipeline(tmp_path, source, model="model-b").run() + assert changed.changed == ("fs:one",) + assert source.acquire_calls == ["fs:one"] + + +def test_partial_vector_failure_never_changes_active_or_exposes_generation(tmp_path): + one = item("one", "a") + good = pipeline(tmp_path, Source([(one, "old")])) + old = good.run().generation + changed = item("one", "b") + vectors = Vectors(fail=True) + broken = pipeline(tmp_path, Source([(changed, "new")]), vectors=vectors) + with pytest.raises(PipelineError): + broken.run() + assert broken.store.active_generation() == old + assert vectors.records[0].record.metadata["vector_generation"] != old + + +def test_dimension_mismatch_fails_before_vector_write_and_publish(tmp_path): + one = item("one", "a") + vectors = Vectors() + candidate = pipeline(tmp_path, Source([(one, "hello")]), embedder=Embedder(dim=2), vectors=vectors) + with pytest.raises(PipelineError, match="dimension"): + candidate.run() + assert vectors.records == [] + assert candidate.store.active_generation() is None + + +def test_dry_run_and_failed_acquire_never_change_active(tmp_path): + one = item("one", "a") + active = pipeline(tmp_path, Source([(one, "old")])).run().generation + changed = item("one", "b") + dry = pipeline(tmp_path, Source([(changed, "new")])).run(dry_run=True) + assert dry.published is False + assert dry.generation is None + assert dry.manifest.documents[0].content == "old" + with pytest.raises(PipelineError): + pipeline(tmp_path, Source([(changed, RuntimeError("boom"))])).run() + assert CorpusStore(tmp_path / "corpus").active_generation() == active + + +def test_job_pipeline_uses_ordered_plan_and_returns_run_id(tmp_path): + one = item("one", "a") + candidate = pipeline(tmp_path, Source([(one, "hello")])) + result = candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + assert result.status == "succeeded" + assert result.run_id and len(result.run_id) == 32 + checkpoint = tmp_path / ".tht-jobs" / "evidence" / "runs" / result.run_id / "checkpoint.json" + payload = __import__("json").loads(checkpoint.read_text()) + assert [stage["name"] for stage in payload["stages"]] == [ + "discover", "acquire_normalize_chunk", "embed", "vector_upsert", + "stage_validate", "publish", "retention_cleanup", + ] + + +def test_job_pipeline_dry_run_only_discovers_and_reports_changes(tmp_path): + one = item("one", "a") + source = Source([(one, "hello")]) + embedder = Embedder() + vectors = Vectors() + result = pipeline(tmp_path, source, embedder=embedder, vectors=vectors).run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + dry_run=True, + ) + assert result.changed == ("fs:one",) + assert source.acquire_calls == [] + assert embedder.calls == [] + assert vectors.records == [] + assert result.generation is None and result.published is False + + +@pytest.mark.parametrize("crash_stage", [ + "discover", "acquire_normalize_chunk", "embed", "vector_upsert", + "stage_validate", "publish", "retention_cleanup", +]) +def test_job_pipeline_crash_after_each_stage_resumes_without_duplicate_effects(tmp_path, crash_stage): + one = item("one", "a") + source = Source([(one, "hello")]) + embedder = Embedder() + vectors = Vectors() + candidate = pipeline(tmp_path, source, embedder=embedder, vectors=vectors) + + class Crash(BaseException): + pass + + def fault(_context, stage): + if stage == crash_stage: + raise Crash() + + with pytest.raises(Crash): + candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + after_stage_return=fault, + ) + runs = tmp_path / ".tht-jobs" / "evidence" / "runs" + crashed = next(runs.iterdir()).name + result = candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + resume_run_id=crashed, + ) + assert result.status == "succeeded" + assert source.acquire_calls == ["fs:one"] + assert len(embedder.calls) == 1 + assert len(vectors.records) == 1 + + +def test_job_pipeline_raw_upsert_failure_compensates_and_resumes_with_new_generation(tmp_path): + one = item("one", "a") + vectors = Vectors(fail=True) + candidate = pipeline(tmp_path, Source([(one, "hello")]), vectors=vectors) + first = candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + assert first.status == "failed" + assert vectors.records == [] + old_generation = first.generation + vectors.fail = False + resumed = candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + resume_run_id=first.run_id, + ) + assert resumed.status == "succeeded", resumed + assert resumed.generation != old_generation + assert candidate.store.active_generation() == resumed.generation + + +def test_job_pipeline_raw_stage_failure_compensates_vectors_and_resumes(tmp_path, monkeypatch): + one = item("one", "a") + vectors = Vectors() + candidate = pipeline(tmp_path, Source([(one, "hello")]), vectors=vectors) + real_stage = candidate.store.stage + calls = 0 + + def fail_once(*args, **kwargs): + nonlocal calls + calls += 1 + if calls == 1: + raise OSError("raw stage failure") + return real_stage(*args, **kwargs) + + monkeypatch.setattr(candidate.store, "stage", fail_once) + first = candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + assert first.status == "failed" and vectors.records == [] + resumed = candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + resume_run_id=first.run_id, + ) + assert resumed.status == "succeeded", resumed + + +@pytest.mark.parametrize("stage,filename", [ + ("discover", "plan.json"), + ("acquire_normalize_chunk", "manifest.json"), + ("embed", "embeddings.json"), +]) +@pytest.mark.parametrize("mutation", ["missing", "tampered"]) +def test_job_pipeline_rejects_corrupt_required_artifacts_before_resume( + tmp_path, stage, filename, mutation, +): + one = item("one", "a") + candidate = pipeline(tmp_path, Source([(one, "hello")])) + + class Crash(BaseException): + pass + + with pytest.raises(Crash): + candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + after_stage_return=lambda _context, name: ( + (_ for _ in ()).throw(Crash()) if name == stage else None + ), + ) + runs = tmp_path / ".tht-jobs" / "evidence" / "runs" + crashed = next(runs.iterdir()) + target = crashed / "artifacts" / filename + target.unlink() if mutation == "missing" else target.write_text("tampered") + from tht.jobs.runner import CorruptCheckpointError + with pytest.raises(CorruptCheckpointError, match="artifact"): + candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + resume_run_id=crashed.name, + ) + + +def test_vector_intent_is_reconciled_after_process_interruption_without_duplicate_upsert(tmp_path): + one = item("one", "a") + vectors = InterruptingVectors() + candidate = pipeline( + tmp_path, Source([(one, "a" * 250)]), vectors=vectors, + policy=ChunkPolicy(version="chunk-v1", max_chars=100), + ) + with pytest.raises(KeyboardInterrupt): + candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + runs = tmp_path / ".tht-jobs" / "evidence" / "runs" + interrupted = next(runs.iterdir()) + checkpoint = __import__("json").loads((interrupted / "checkpoint.json").read_text()) + vector_stage = checkpoint["stages"][3] + assert vector_stage["status"] == "running" + assert vector_stage["effect_state"] == "intent" + first_written = vectors.batches[0][0] + + result = candidate.run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + resume_run_id=interrupted.name, + ) + assert result.status == "succeeded" and result.published is True + assert first_written not in vectors.batches[1] + assert len(vectors.records) == 3 diff --git a/harness/tests/test_corpus_publish.py b/harness/tests/test_corpus_publish.py new file mode 100644 index 00000000..0596bd3a --- /dev/null +++ b/harness/tests/test_corpus_publish.py @@ -0,0 +1,166 @@ +import pytest +import os + +from tht.corpus.models import CorpusManifest +from tht.corpus.store import CorpusStore, UnsafeCorpusPath + + +def test_publish_switches_active_atomically_and_resolves_materialized_files(tmp_path): + store = CorpusStore(tmp_path / "corpus") + generation = store.stage(CorpusManifest(), {}) + seen = [] + store._replace = lambda source, target: (seen.append(source.read_text()), source.replace(target)) + published = store.publish(generation) + assert published == generation + assert store.active_generation() == generation + assert seen == [generation + "\n"] + + +def test_active_manifest_is_a_consistent_reader_snapshot(tmp_path): + store = CorpusStore(tmp_path / "corpus") + first = store.stage(CorpusManifest(metadata={"name": "first"}), {}) + second = store.stage(CorpusManifest(metadata={"name": "second"}), {}) + store.publish(first) + snapshot = store.active_manifest() + store.publish(second) + assert snapshot.metadata["name"] == "first" + assert store.active_manifest().metadata["name"] == "second" + + +def test_store_rejects_symlinked_generation_root(tmp_path): + outside = tmp_path / "outside" + outside.mkdir() + root = tmp_path / "corpus" + root.symlink_to(outside, target_is_directory=True) + with pytest.raises(UnsafeCorpusPath): + CorpusStore(root) + + +def test_active_pointer_cannot_escape_generation_root(tmp_path): + store = CorpusStore(tmp_path / "corpus") + store.root.mkdir(parents=True, exist_ok=True) + store.active_path.write_text("../outside\n") + with pytest.raises(UnsafeCorpusPath): + store.active_manifest() + + +def test_publish_restores_previous_active_when_directory_fsync_fails_after_replace(tmp_path, monkeypatch): + store = CorpusStore(tmp_path / "corpus") + first = store.stage(CorpusManifest(), {}) + second = store.stage(CorpusManifest(), {}) + store.publish(first) + def fail_once(): + store._fsync_directory = store._sync_root + raise OSError("post replace crash") + + store._fsync_directory = fail_once + with pytest.raises(OSError, match="post replace"): + store.publish(second) + assert store.active_generation() == first + + +def test_read_document_rejects_symlink_hardlink_and_hash_mismatch(tmp_path): + from tht.corpus.models import CanonicalDocument + + content = "trusted" + digest = "sha256:" + __import__("hashlib").sha256(content.encode()).hexdigest() + document = CanonicalDocument( + document_id="doc:" + "a" * 64, source_id="fs:one", source_uri="file:///one", + source_fingerprint="sha256:" + "b" * 64, content_hash=digest, content=content, + pipeline_version="evidence-v1", + ) + store = CorpusStore(tmp_path / "corpus") + generation = store.stage(CorpusManifest(documents=(document,)), {document.document_id: content}) + path = store.resolve_document(document.document_id, generation) + assert store.read_document(document.document_id, generation) == content + + path.unlink() + path.symlink_to(tmp_path / "outside") + (tmp_path / "outside").write_text(content) + with pytest.raises(UnsafeCorpusPath): + store.read_document(document.document_id, generation) + + path.unlink() + os.link(tmp_path / "outside", path) + with pytest.raises(UnsafeCorpusPath): + store.read_document(document.document_id, generation) + + path.unlink() + path.write_text("tampered") + with pytest.raises(UnsafeCorpusPath): + store.read_document(document.document_id, generation) + + +def test_generation_inventory_is_validated_and_sorted(tmp_path): + store = CorpusStore(tmp_path / "corpus") + first = store.stage(CorpusManifest(), {}, generation="gen:" + "1" * 32) + second = store.stage(CorpusManifest(), {}, generation="gen:" + "2" * 32) + (store.root / "unrelated").mkdir() + assert store.list_generations() == [first, second] + + +def test_published_inventory_excludes_staged_and_invalid_newer_directories(tmp_path): + store = CorpusStore(tmp_path / "corpus") + first = store.stage(CorpusManifest(), {}, generation="gen:" + "1" * 32) + store.publish(first) + store.stage(CorpusManifest(), {}, generation="gen:" + "2" * 32) + invalid = store.generation_path("gen:" + "3" * 32) + invalid.mkdir() + (invalid / "PUBLISHED").write_text("2026-01-01T00:00:00Z\n") + assert store.published_generations() == [first] + + +def test_owned_copy_uses_validated_descriptor_bytes_when_source_is_replaced(tmp_path, monkeypatch): + from tht.corpus.models import CanonicalDocument + import hashlib + + content = "active bytes" + document = CanonicalDocument( + document_id="doc:" + "c" * 64, source_id="fs:copy", source_uri="file:///copy", + source_fingerprint="sha256:" + "d" * 64, + content_hash="sha256:" + hashlib.sha256(content.encode()).hexdigest(), + content=content, pipeline_version="evidence-v1", + ) + store = CorpusStore(tmp_path / "corpus") + generation = store.stage(CorpusManifest(documents=(document,)), {document.document_id: content}) + store.publish(generation) + source = store.resolve_document(document.document_id) + real_read = os.read + + def replace_after_read(fd, size): + payload = real_read(fd, size) + source.unlink() + source.write_text("replacement") + return payload + + monkeypatch.setattr(os, "read", replace_after_read) + owned = store.materialize_document(document.document_id, tmp_path / "session" / "evidence.md") + assert owned.read_text() == content + assert hashlib.sha256(owned.read_bytes()).hexdigest() == document.content_hash.removeprefix("sha256:") + + +def test_materialized_snapshot_uses_identified_manifest_when_active_changes(tmp_path): + from tht.corpus.models import CanonicalDocument + import hashlib + + def doc(content, fingerprint): + return CanonicalDocument( + document_id="doc:" + hashlib.sha256(content.encode()).hexdigest(), + source_id="fs:item", source_uri="file:///item", + source_fingerprint="sha256:" + fingerprint * 64, + content_hash="sha256:" + hashlib.sha256(content.encode()).hexdigest(), + content=content, pipeline_version="evidence-v1", + ) + + store = CorpusStore(tmp_path / "corpus") + old = doc("old", "a") + old_generation = store.stage(CorpusManifest(documents=(old,)), {old.document_id: old.content}) + store.publish(old_generation) + snapshot = store.active_manifest() + new = doc("new", "b") + new_generation = store.stage(CorpusManifest(documents=(new,)), {new.document_id: new.content}) + store.publish(new_generation) + path = store.materialize_document( + snapshot.documents[0].document_id, tmp_path / "owned.md", generation=snapshot.manifest_id, + ) + assert path.read_text() == "old" diff --git a/harness/tests/test_doctor_cli.py b/harness/tests/test_doctor_cli.py new file mode 100644 index 00000000..58f195df --- /dev/null +++ b/harness/tests/test_doctor_cli.py @@ -0,0 +1,156 @@ +import json +from pathlib import Path + +from typer.testing import CliRunner + +from tht.cli import app + + +runner = CliRunner() + + +def _config(path: Path, *, absolute_sessions: Path | None = None) -> Path: + sessions = absolute_sessions or Path("sessions") + path.write_text( + "dwh:\n" + " type: postgres_direct\n" + " connection:\n" + " database: patient_db\n" + " schema: private_schema\n" + " user: pii_user\n" + " password: super-secret\n" + "roots:\n" + " artifacts: artifacts\n" + " indexes: indexes\n" + f" sessions: {sessions}\n" + ) + return path + + +def test_doctor_json_reports_portable_path_statuses_without_secrets(monkeypatch, tmp_path): + cfg = _config(tmp_path / "demo.yaml") + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + + result = runner.invoke(app, ["doctor", "--json", "--config", str(cfg)]) + + assert result.exit_code == 0 + payload = json.loads(result.stdout) + assert payload == { + "ok": True, + "components": { + "config": {"status": "ok"}, + "data_root": {"status": "ok"}, + "workspace_paths": {"status": "ok", "legacy_absolute": []}, + }, + } + assert "super-secret" not in result.stdout + assert "patient_db" not in result.stdout + assert "pii_user" not in result.stdout + assert result.stderr == "" + + +def test_doctor_json_flags_absolute_legacy_paths(monkeypatch, tmp_path): + cfg = _config(tmp_path / "demo.yaml", absolute_sessions=tmp_path / "old-sessions") + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + + result = runner.invoke(app, ["doctor", "--json", "--config", str(cfg)]) + + assert result.exit_code == 0 + payload = json.loads(result.stdout) + assert payload["components"]["workspace_paths"] == { + "status": "warning", + "legacy_absolute": ["sessions"], + } + assert str(tmp_path) not in result.stdout + + +def test_doctor_json_returns_structured_config_error(monkeypatch, tmp_path): + cfg = _config(tmp_path / "demo.yaml") + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + cfg.write_text(cfg.read_text().replace("sessions: sessions", "sessions: ../../private")) + + result = runner.invoke(app, ["doctor", "--json", "--config", str(cfg)]) + + assert result.exit_code == 1 + payload = json.loads(result.stdout) + assert payload["ok"] is False + assert payload["components"]["workspace_paths"]["status"] == "error" + assert "outside workspace root" in payload["components"]["workspace_paths"]["message"] + assert result.stderr == "" + + +def test_doctor_json_does_not_echo_invalid_config_values(monkeypatch, tmp_path): + cfg = _config(tmp_path / "demo.yaml") + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + cfg.write_text(cfg.read_text().replace("database: patient_db", "")) + + result = runner.invoke(app, ["doctor", "--json", "--config", str(cfg)]) + + assert result.exit_code == 1 + payload = json.loads(result.stdout) + assert payload["components"]["config"] == { + "status": "error", + "message": "configuration is invalid or unreadable", + } + assert "super-secret" not in result.stdout + assert "pii_user" not in result.stdout + + +def test_doctor_json_normalizes_malformed_yaml(monkeypatch, tmp_path): + cfg = tmp_path / "demo.yaml" + cfg.write_text("password: super-secret\nroots: [unterminated") + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + + result = runner.invoke(app, ["doctor", "--json", "--config", str(cfg)]) + + assert result.exit_code == 1 + assert json.loads(result.stdout)["components"]["config"] == { + "status": "error", + "message": "configuration is invalid or unreadable", + } + assert result.stderr == "" + assert "super-secret" not in result.stdout + assert "Traceback" not in result.stdout + + +def test_doctor_json_normalizes_unreadable_config(monkeypatch, tmp_path): + cfg = tmp_path / "demo.yaml" + cfg.mkdir() + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + + result = runner.invoke(app, ["doctor", "--json", "--config", str(cfg)]) + + assert result.exit_code == 1 + assert json.loads(result.stdout)["components"]["config"] == { + "status": "error", + "message": "configuration is invalid or unreadable", + } + assert result.stderr == "" + + +def test_doctor_human_output_is_actionable_and_redacted(monkeypatch, tmp_path): + cfg = _config(tmp_path / "demo.yaml", absolute_sessions=tmp_path / "patient-private") + monkeypatch.delenv("THT_DATA_ROOT", raising=False) + + result = runner.invoke(app, ["doctor", "--config", str(cfg)]) + + assert result.exit_code == 0 + assert "data_root: warning - set THT_DATA_ROOT to enable portable storage" in result.stdout + assert "workspace_paths: warning - absolute legacy roots: sessions" in result.stdout + assert str(tmp_path) not in result.stdout + assert "patient_db" not in result.stdout + assert "super-secret" not in result.stdout + + +def test_doctor_human_config_error_is_actionable_and_redacted(monkeypatch, tmp_path): + cfg = tmp_path / "patient-private.yaml" + cfg.write_text("password: super-secret\nroots: [unterminated") + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + + result = runner.invoke(app, ["doctor", "--config", str(cfg)]) + + assert result.exit_code == 1 + assert "config: error - configuration is invalid or unreadable" in result.stdout + assert str(tmp_path) not in result.stdout + assert "super-secret" not in result.stdout + assert result.stderr == "" diff --git a/harness/tests/test_dwh_adapters.py b/harness/tests/test_dwh_adapters.py new file mode 100644 index 00000000..8ffa8604 --- /dev/null +++ b/harness/tests/test_dwh_adapters.py @@ -0,0 +1,163 @@ +import pytest +from sqlalchemy.exc import OperationalError + +from tht.config import DatabaseConfig, RestConfig +from tht.db.sampling import distinct_values_rest, sample_column_rest +from tht.execute import ExecutionError +from tht.ports import DistinctValues, DwhAdapter +from tht.rest.client import RestError +from tht.adapters.dwh import PostgresDwhAdapter + + +def postgres_factory(): + from tht.adapters.dwh import PostgresDwhAdapter + + return PostgresDwhAdapter( + DatabaseConfig(database="analytics", schema="dw", user="reader", password="secret") + ) + + +def rest_factory(): + from tht.adapters.dwh import ThothRestDwhAdapter + + database = DatabaseConfig( + database="analytics", + schema="dw", + user="unused", + password="unused", + transport="rest", + ) + return ThothRestDwhAdapter( + database, + RestConfig(base_url="https://dwh.example.test", api_key="secret"), + ) + + +@pytest.mark.parametrize("factory", [postgres_factory, rest_factory]) +def test_adapter_rejects_write_sql_without_using_transport(factory): + with pytest.raises(ExecutionError, match="read-only enforcement"): + factory().run_query("delete from fact_sales", limit=10) + + +@pytest.mark.parametrize("factory", [postgres_factory, rest_factory]) +def test_adapter_satisfies_dwh_protocol(factory): + adapter = factory() + assert isinstance(adapter, DwhAdapter) + assert adapter.capabilities.introspection is True + + +@pytest.mark.parametrize("factory", [postgres_factory, rest_factory]) +@pytest.mark.parametrize("invalid_limit", [True, 1.5, 0, -1]) +def test_run_query_rejects_non_positive_integer_limit(factory, invalid_limit): + adapter = factory() + with pytest.raises(ValueError, match="positive integer"): + adapter.run_query("select 1", limit=invalid_limit) + + +@pytest.mark.parametrize("factory", [postgres_factory, rest_factory]) +def test_run_query_requires_explicit_limit(factory): + with pytest.raises(TypeError): + factory().run_query("select 1") + + +@pytest.mark.parametrize("invalid_limit", [True, 1.5, 0, -1]) +def test_rest_sampling_rejects_non_positive_integer_limit(invalid_limit): + class Client: + def top_values(self, *args): + raise AssertionError("transport must not be used") + + with pytest.raises(ValueError, match="positive integer"): + sample_column_rest(Client(), "dw", "sales", "region", limit=invalid_limit) + with pytest.raises(ValueError, match="positive integer"): + distinct_values_rest( + Client(), "dw", "sales", "region", max_values=invalid_limit + ) + + +def test_postgres_sampling_delegates_to_paired_sampling_functions(monkeypatch): + adapter = postgres_factory() + calls = [] + expected = DistinctValues(values=["A"], truncated=True) + + monkeypatch.setattr( + "tht.adapters.dwh.postgres.sampling.sample_column", + lambda engine, schema, table, column, *, limit: calls.append( + (engine, schema, table, column, limit) + ) + or ["A", "B"], + ) + distinct_calls = [] + monkeypatch.setattr( + "tht.adapters.dwh.postgres.sampling.distinct_values", + lambda engine, schema, table, column, *, max_values: distinct_calls.append(max_values) + or expected, + ) + + assert adapter.sample_column("sales", "region", limit=2) == ["A", "B"] + assert calls == [(adapter._engine, "dw", "sales", "region", 2)] + assert adapter.distinct_values("sales", "region", limit=17) is expected + assert distinct_calls == [17] + + +def test_rest_sampling_delegates_and_translates_transport_errors(monkeypatch): + adapter = rest_factory() + expected = DistinctValues(values=["A", "B"], truncated=False) + monkeypatch.setattr( + "tht.adapters.dwh.thoth_rest.sampling.sample_column_rest", + lambda client, schema, table, column, *, limit: ["A", "B"], + ) + monkeypatch.setattr( + "tht.adapters.dwh.thoth_rest.sampling.distinct_values_rest", + lambda client, schema, table, column, *, max_values: expected, + ) + assert adapter.sample_column("sales", "region", limit=2) == ["A", "B"] + assert adapter.distinct_values("sales", "region", limit=17) is expected + + def fail(*args, **kwargs): + raise RestError("transport failed") + + monkeypatch.setattr("tht.adapters.dwh.thoth_rest.sampling.sample_column_rest", fail) + monkeypatch.setattr("tht.adapters.dwh.thoth_rest.sampling.distinct_values_rest", fail) + with pytest.raises(ExecutionError, match="transport failed"): + adapter.sample_column("sales", "region", limit=2) + with pytest.raises(ExecutionError, match="transport failed"): + adapter.distinct_values("sales", "region", limit=17) + + +def test_rest_distinct_values_reports_transport_truncation(): + class Client: + def top_values(self, schema, table, column, limit): + assert (schema, table, column, limit) == ("dw", "sales", "region", 3) + return [{"value": "A"}, {"value": "B"}, {"value": "C"}] + + result = distinct_values_rest(Client(), "dw", "sales", "region", max_values=2) + assert result == DistinctValues(values=["A", "B"], truncated=True) + + +def test_postgres_health_only_normalizes_database_errors(monkeypatch): + adapter = postgres_factory() + database_error = OperationalError("select 1", {}, Exception("offline")) + monkeypatch.setattr("tht.adapters.dwh.postgres.ping", lambda engine: (_ for _ in ()).throw(database_error)) + assert adapter.health().ok is False + + monkeypatch.setattr( + "tht.adapters.dwh.postgres.ping", + lambda engine: (_ for _ in ()).throw(ValueError("programming bug")), + ) + with pytest.raises(ValueError, match="programming bug"): + adapter.health() + + +def test_non_default_timeout_reaches_query_and_explain(monkeypatch): + adapter = PostgresDwhAdapter( + DatabaseConfig(database="analytics", schema="dw", user="reader", password="secret"), + statement_timeout_ms=12_345, + ) + calls = [] + monkeypatch.setattr("tht.adapters.dwh.postgres.execute.run_query", + lambda engine, sql, *, limit, timeout_ms: calls.append(("run", timeout_ms))) + monkeypatch.setattr("tht.adapters.dwh.postgres.execute.explain", + lambda engine, sql, *, timeout_ms: calls.append(("explain", timeout_ms))) + adapter.run_query("select 1", limit=2) + adapter.explain("select 1") + assert calls == [("run", 12_345), ("explain", 12_345)] diff --git a/harness/tests/test_dwh_port_contract.py b/harness/tests/test_dwh_port_contract.py new file mode 100644 index 00000000..aad615c7 --- /dev/null +++ b/harness/tests/test_dwh_port_contract.py @@ -0,0 +1,70 @@ +from dataclasses import FrozenInstanceError + +import pytest + +from tht.execute import ExecResult, PlanSummary +from tht.mschema.models import PhysicalSchema +from tht.ports.dwh import ( + DwhAdapter, + DwhCapabilities, + DwhHealth, + DistinctValues, + UnsupportedCapability, +) + + +class FakeDwhAdapter: + capabilities = DwhCapabilities() + + def health(self) -> DwhHealth: + return DwhHealth(ok=True) + + def introspect(self) -> PhysicalSchema: + raise NotImplementedError + + def run_query(self, sql: str, *, limit: int) -> ExecResult: + raise NotImplementedError + + def explain(self, sql: str) -> PlanSummary: + raise NotImplementedError + + def sample_column(self, table: str, column: str, *, limit: int) -> list[object]: + raise NotImplementedError + + def distinct_values(self, table: str, column: str, *, limit: int) -> DistinctValues: + raise NotImplementedError + + +def test_fake_adapter_satisfies_runtime_protocol(): + adapter = FakeDwhAdapter() + + assert isinstance(adapter, DwhAdapter) + assert adapter.capabilities.explain is True + assert adapter.health().ok is True + + +def test_contract_types_are_public_and_capabilities_are_immutable(): + capabilities = DwhCapabilities() + + assert capabilities.introspection is True + assert capabilities.sampling is True + assert capabilities.distinct_values is True + assert issubclass(UnsupportedCapability, Exception) + with pytest.raises(FrozenInstanceError): + capabilities.explain = False + + +def test_all_contract_types_are_exported_from_public_package(): + from tht.ports import DwhAdapter as PublicDwhAdapter + from tht.ports import DwhCapabilities as PublicDwhCapabilities + from tht.ports import DwhHealth as PublicDwhHealth + from tht.ports import DistinctValues as PublicDistinctValues + from tht.ports import UnsupportedCapability as PublicUnsupportedCapability + + result = PublicDistinctValues(values=["a"], truncated=True) + assert result.values == ["a"] + assert result.truncated is True + assert PublicDwhAdapter is DwhAdapter + assert PublicDwhCapabilities is DwhCapabilities + assert PublicDwhHealth is DwhHealth + assert PublicUnsupportedCapability is UnsupportedCapability diff --git a/harness/tests/test_dwh_preprocess_job.py b/harness/tests/test_dwh_preprocess_job.py new file mode 100644 index 00000000..cdc0dbf1 --- /dev/null +++ b/harness/tests/test_dwh_preprocess_job.py @@ -0,0 +1,928 @@ +import hashlib +import json +from pathlib import Path + +from typer.testing import CliRunner + +from tht.cli import app +from tht.jobs.dwh_pipeline import DwhPreprocessPipeline +from tht.jobs.dwh_pipeline import active_generation_dir, config_dwh_binding +from tht.jobs.dwh_pipeline import resolve_dwh_snapshot +from tht.jobs.dwh_pipeline import lease_dwh_snapshot +from tht.jobs.locking import _lock_name + + +FP = "sha256:" + hashlib.sha256(b"test").hexdigest() + + +def snapshot_config(tmp_path, workspace_id="demo"): + from types import SimpleNamespace + + cfg = SimpleNamespace( + paths=SimpleNamespace(artifacts=tmp_path / "artifacts", indexes=tmp_path / "indexes"), + _workspace_id=workspace_id, + _config_source="test", + ) + cfg.model_dump_json = lambda: "test" + return cfg + + +def test_dwh_and_evidence_jobs_have_distinct_lock_names(): + assert _lock_name("demo", "dwh") != _lock_name("demo", "evidence") + + +def test_unowned_reads_fail_closed_without_creating_any_files(tmp_path): + import pytest + + cfg = snapshot_config(tmp_path) + with pytest.raises(Exception, match="not initialized"): + resolve_dwh_snapshot(cfg) + with pytest.raises(Exception, match="not initialized"): + with lease_dwh_snapshot(cfg): + pass + assert not (tmp_path / ".tht-dwh").exists() + + +def test_writer_claim_allows_only_lock_and_empty_generations(tmp_path): + import pytest + + for name, make_entry in ( + ("unexpected", lambda root: (root / "unexpected").write_text("x")), + ("stale-temp", lambda root: (root / ".OWNER.json.stale.tmp").write_text("x")), + ("unexpected-dir", lambda root: (root / "other").mkdir()), + ): + root = tmp_path / name / ".tht-dwh" + root.mkdir(parents=True, mode=0o700) + make_entry(root) + calls = [] + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=root.parent, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: calls.append("called"), + build_lsh=lambda physical, output: None, + ) + with pytest.raises(Exception, match="unbound"): + pipeline.run() + assert calls == [] + assert not (root / "OWNER.json").exists() + + allowed = tmp_path / "allowed" + (allowed / ".tht-dwh" / "generations").mkdir(parents=True, mode=0o700) + report = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=allowed, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ).run() + assert report.status == "succeeded" + + +def test_writer_rejects_unbound_legacy_artifacts_before_building(tmp_path): + import pytest + + legacy = tmp_path / "artifacts" / "mschema" / "physical.yaml" + legacy.parent.mkdir(parents=True) + legacy.write_text("legacy") + calls = [] + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: calls.append("called"), + build_lsh=lambda physical, output: None, + current_physical=legacy, + ) + with pytest.raises(Exception, match="legacy artifacts are unbound"): + pipeline.run() + assert calls == [] + assert not (tmp_path / ".tht-dwh" / "OWNER.json").exists() + + +def test_writer_rejects_dangling_legacy_symlinks_before_claim_or_callback(tmp_path): + import pytest + + legacy = tmp_path / "artifacts" / "mschema" / "physical.yaml" + legacy.parent.mkdir(parents=True) + legacy.symlink_to(tmp_path / "missing-catalog") + calls = [] + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: calls.append("called"), + build_lsh=lambda physical, output: None, + current_physical=legacy, + ) + with pytest.raises(Exception, match="legacy artifacts are unbound"): + pipeline.run() + assert calls == [] + assert not (tmp_path / ".tht-dwh" / "OWNER.json").exists() + + +def test_owner_publication_remains_on_locked_root_when_path_is_swapped( + monkeypatch, tmp_path, +): + import pytest + import tht.jobs.dwh_pipeline as module + + real_replace = module.os.replace + moved = tmp_path / "locked-root" + replacement = tmp_path / ".tht-dwh" + swapped = False + + def swapping_replace(source, destination, *args, **kwargs): + nonlocal swapped + if destination == "OWNER.json" and kwargs.get("dst_dir_fd") is not None: + swapped = True + replacement.rename(moved) + replacement.mkdir(mode=0o700) + (moved / "generation.lock").rename(replacement / "generation.lock") + return real_replace(source, destination, *args, **kwargs) + + monkeypatch.setattr(module.os, "replace", swapping_replace) + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: (_ for _ in ()).throw(AssertionError("callback called")), + build_lsh=lambda physical, output: None, + ) + with pytest.raises(Exception): + pipeline.run() + assert swapped + assert (moved / "OWNER.json").is_file() + assert not (replacement / "OWNER.json").exists() + assert (replacement / "generation.lock").is_file() + + +def test_owner_requires_exact_read_only_owner_mode_and_active_requires_binding(tmp_path): + import pytest + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + pipeline.run() + cfg = snapshot_config(tmp_path) + binding = config_dwh_binding(cfg) + assert active_generation_dir(tmp_path, binding) is not None + with pytest.raises(Exception, match="different workspace configuration"): + active_generation_dir(tmp_path, {**binding, "workspace_id": "other"}) + + marker = tmp_path / ".tht-dwh" / "OWNER.json" + marker.chmod(0o440) + with pytest.raises(Exception, match="ownership marker"): + resolve_dwh_snapshot(cfg) + + +def test_selected_dwh_stages_run_in_declared_order(tmp_path): + calls = [] + pipeline = DwhPreprocessPipeline( + workspace_id="demo", + workspace_root=tmp_path, + config_fingerprint=FP, + input_fingerprint=FP, + introspect=lambda output: (calls.append("introspect"), output.write_text("catalog")), + build_lsh=lambda physical, output: _write_lsh(calls, physical, output), + ) + + report = pipeline.run(("introspect", "lsh")) + + assert report.status == "succeeded" + assert calls == ["introspect", "lsh"] + assert [stage.name for stage in report.stages] == ["introspect", "lsh"] + active = (tmp_path / ".tht-dwh" / "ACTIVE").read_text().strip() + published = tmp_path / ".tht-dwh" / "generations" / active + assert (published / "physical.yaml").read_text() == "catalog" + assert sorted(path.name for path in published.iterdir()) == [ + "demo_lsh.pkl", "demo_meta.json", "demo_minhashes.pkl", + "generation-manifest.json", "physical.yaml", + ] + + +def test_shared_root_rejects_other_workspace_before_builder_or_read(tmp_path): + calls = [] + owner = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + published = owner.run() + contender = DwhPreprocessPipeline( + workspace_id="other", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: calls.append("introspect"), + build_lsh=lambda physical, output: calls.append("lsh"), + ) + + import pytest + with pytest.raises(Exception, match="different workspace configuration"): + contender.run() + with pytest.raises(Exception, match="different workspace configuration"): + resolve_dwh_snapshot(snapshot_config(tmp_path, "other")) + + assert calls == [] + assert resolve_dwh_snapshot(snapshot_config(tmp_path)).generation == published.run_id + + +def test_shared_root_mismatch_fails_without_deadlock_while_owner_reader_is_active(tmp_path): + import threading + + owner = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + owner.run() + contender = DwhPreprocessPipeline( + workspace_id="other", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: (_ for _ in ()).throw(AssertionError("builder called")), + build_lsh=lambda physical, output: None, + ) + finished = threading.Event() + errors = [] + with lease_dwh_snapshot(snapshot_config(tmp_path)): + thread = threading.Thread( + target=lambda: (errors.append(_capture_error(contender.run)), finished.set()) + ) + thread.start() + assert not finished.wait(0.1) + thread.join(2) + assert finished.is_set() + assert "different workspace configuration" in str(errors[0]) + + +def test_concurrent_brand_new_shared_root_has_one_atomic_owner_and_loser_never_builds(tmp_path): + import threading + + calls = {"alpha": 0, "beta": 0} + results = [] + barrier = threading.Barrier(2) + + def run(workspace): + def introspect(output): + calls[workspace] += 1 + output.write_text("catalog") + + def build(physical, output): + calls[workspace] += 1 + _write_lsh([], physical, output) + + candidate = DwhPreprocessPipeline( + workspace_id=workspace, workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=introspect, build_lsh=build, + ) + barrier.wait() + results.append((workspace, _capture_error(candidate.run))) + + threads = [threading.Thread(target=run, args=(name,)) for name in ("alpha", "beta")] + for thread in threads: + thread.start() + for thread in threads: + thread.join(5) + assert all(not thread.is_alive() for thread in threads) + winner = next(name for name, result in results if not isinstance(result, Exception)) + loser = next(name for name, result in results if isinstance(result, Exception)) + assert calls[winner] == 2 + assert calls[loser] == 0 + + +def test_missing_active_with_generations_and_symlink_owner_marker_fail_closed(tmp_path): + import pytest + + calls = [] + owner = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + owner.run() + (tmp_path / ".tht-dwh" / "ACTIVE").unlink() + owner.introspect = lambda output: calls.append("called") + with pytest.raises(Exception, match="without a consistent ACTIVE"): + owner.run() + assert calls == [] + + other_root = tmp_path / "other" + marker_root = other_root / ".tht-dwh" + marker_root.mkdir(parents=True) + external = tmp_path / "external-owner" + external.write_text("foreign") + (marker_root / "OWNER.json").symlink_to(external) + contender = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=other_root, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: calls.append("symlink-called"), + build_lsh=lambda physical, output: None, + ) + with pytest.raises(Exception, match="ownership marker"): + contender.run() + assert calls == [] + + +def _capture_error(operation): + try: + return operation() + except Exception as error: + return error + + +def _write_lsh(calls, physical: Path, output: Path): + calls.append("lsh") + assert physical.read_text() == "catalog" + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json"): + (output / name).write_text(name) + + +def test_preprocess_dwh_json_is_pristine(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + class Report: + status = "succeeded" + + def model_dump(self, mode=None): + return {"status": "succeeded", "run_id": "a" * 32, "stages": []} + + seen = {} + + def run(config, *, steps, resume): + seen.update(config=config, steps=steps, resume=resume) + return Report() + + monkeypatch.setattr(command, "run_dwh_from_config", run) + response = CliRunner().invoke( + app, + [ + "preprocess", "dwh", "--steps", "introspect,lsh", "--json", + "-c", str(tmp_path / "workspace.yaml"), + ], + ) + + assert response.exit_code == 0, response.output + assert json.loads(response.output)["run_id"] == "a" * 32 + assert seen["steps"] == ("introspect", "lsh") + + +def test_preprocess_dwh_rejects_unknown_or_duplicate_steps(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + called = False + + def forbidden(*args, **kwargs): + nonlocal called + called = True + + monkeypatch.setattr(command, "run_dwh_from_config", forbidden) + runner = CliRunner() + for value in ("introspect,unknown", "lsh,lsh", ""): + response = runner.invoke( + app, + ["preprocess", "dwh", "--steps", value, "--json", "-c", str(tmp_path / "w.yaml")], + ) + assert response.exit_code == 2 + assert json.loads(response.output)["status"] == "failed" + assert called is False + + +def test_failed_multi_file_build_never_replaces_active_generation(tmp_path): + def catalog(output): + output.write_text("old-catalog") + + first = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=catalog, + build_lsh=lambda physical, output: [ + (output / name).write_text(name) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + ).run(("introspect", "lsh")) + assert first.status == "succeeded" + old_active = (tmp_path / ".tht-dwh" / "ACTIVE").read_text() + + def partial_lsh(physical, output): + (output / "demo_lsh.pkl").write_text("new-but-partial") + raise RuntimeError("crash between LSH files") + + failed = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("new-catalog"), + build_lsh=partial_lsh, + ).run(("introspect", "lsh")) + + assert failed.status == "failed" + assert (tmp_path / ".tht-dwh" / "ACTIVE").read_text() == old_active + + resumed = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: (_ for _ in ()).throw( + AssertionError("completed introspection must not repeat") + ), + build_lsh=lambda physical, output: [ + (output / name).write_text("recovered") + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + ).run(("introspect", "lsh"), resume_run_id=failed.run_id) + assert resumed.status == "succeeded" + + +def test_unsafe_lsh_filename_is_rejected(tmp_path): + import pytest + + with pytest.raises(ValueError, match="flat safe"): + DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: None, build_lsh=lambda physical, output: None, + lsh_filenames=("../escape.pkl", "ok.pkl", "meta.json"), + ) + + +def test_active_fsync_failure_restores_previous_pointer(monkeypatch, tmp_path): + import os + import tht.jobs.dwh_pipeline as module + def build(physical, output): + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json"): + (output / name).write_text(name) + + first_pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("old"), build_lsh=build, + ) + first = first_pipeline.run() + root = tmp_path / ".tht-dwh" + root_identity = (root.stat().st_dev, root.stat().st_ino) + original_fsync = module.os.fsync + failed_once = False + + def fail_active_once(fd): + nonlocal failed_once + info = os.fstat(fd) + if ( + (info.st_dev, info.st_ino) == root_identity + and "ACTIVE" in os.listdir(fd) + and not failed_once + ): + failed_once = True + raise OSError("injected directory fsync failure") + original_fsync(fd) + + second = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("new"), build_lsh=build, + ) + monkeypatch.setattr(module.os, "fsync", fail_active_once) + failed = second.run() + assert failed.status == "failed" + assert (tmp_path / ".tht-dwh" / "ACTIVE").read_text().strip() == first.run_id + + +def test_snapshot_root_swap_after_lease_never_reads_replacement(monkeypatch, tmp_path): + import tht.jobs.dwh_pipeline as module + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("trusted"), + build_lsh=lambda physical, output: [ + (output / name).write_text("trusted") + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + ) + first = pipeline.run() + assert first.status == "succeeded" + root = tmp_path / ".tht-dwh" + moved = tmp_path / "moved-read-root" + replacement = root + real_read = module._read_owned_at + swapped = False + + def swapping_read(directory_fd, name, *, readonly): + nonlocal swapped + if name == "ACTIVE" and not swapped: + swapped = True + replacement.rename(moved) + replacement.mkdir(mode=0o700) + (replacement / "sentinel").write_text("replacement-secret") + return real_read(directory_fd, name, readonly=readonly) + + monkeypatch.setattr(module, "_read_owned_at", swapping_read) + try: + with lease_dwh_snapshot(snapshot_config(tmp_path)) as snapshot: + assert snapshot.physical.read_text() == "trusted" + except Exception as error: + assert "ACTIVE" in str(error) or "root" in str(error) + assert swapped + assert (replacement / "sentinel").read_text() == "replacement-secret" + + +def test_snapshot_copies_each_validated_artifact_once_without_reopen(monkeypatch, tmp_path): + import tht.jobs.dwh_pipeline as module + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("trusted"), + build_lsh=lambda physical, output: [ + (output / name).write_text("trusted") + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + ) + assert pipeline.run().status == "succeeded" + real_read = module._read_owned_at + reads = {} + + def mutate_on_reopen(directory_fd, name, *, readonly): + reads[name] = reads.get(name, 0) + 1 + if name.endswith(".pkl") and reads[name] > 1: + return b"MALICIOUS_PICKLE" + return real_read(directory_fd, name, readonly=readonly) + + monkeypatch.setattr(module, "_read_owned_at", mutate_on_reopen) + with lease_dwh_snapshot(snapshot_config(tmp_path)) as snapshot: + assert (snapshot.lsh_dir / "demo_lsh.pkl").read_text() == "trusted" + assert "MALICIOUS" not in (snapshot.lsh_dir / "demo_lsh.pkl").read_text() + assert all(count == 1 for count in reads.values()) + + +def test_reconcile_mismatch_closes_active_generation_fd(monkeypatch, tmp_path): + import os + from types import SimpleNamespace + import pytest + import tht.jobs.dwh_pipeline as module + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + report = pipeline.run() + run_dir = tmp_path / ".tht-jobs" / "dwh" / "runs" / report.run_id + real_active = module._active_generation_fd + + def mismatched_active(root_fd, binding): + generation, generation_fd = real_active(root_fd, binding) + return "f" * 32, generation_fd + + monkeypatch.setattr(module, "_active_generation_fd", mismatched_active) + source = SimpleNamespace( + run_id=report.run_id, + stages=(SimpleNamespace( + status="running", effect_state="intent", name="lsh", + artifact_files=("physical.yaml", "demo_lsh.pkl", "demo_minhashes.pkl", + "demo_meta.json"), + ),), + ) + before = len(os.listdir("/dev/fd")) + with pytest.raises(Exception, match="not ACTIVE"): + pipeline._reconcile_effects(source, run_dir) + assert len(os.listdir("/dev/fd")) == before + + +def test_pipeline_releases_materialized_snapshot_after_every_run(tmp_path): + import tht.jobs.dwh_pipeline as module + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + baseline = set(module._SNAPSHOT_DIRS) + for _ in range(3): + assert pipeline.run().status == "succeeded" + assert set(module._SNAPSHOT_DIRS) == baseline + assert pipeline._snapshot_holder is None + pipeline.introspect = lambda output: (_ for _ in ()).throw(RuntimeError("injected")) + assert pipeline.run().status == "failed" + assert set(module._SNAPSHOT_DIRS) == baseline + assert pipeline._snapshot_holder is None + + +def test_corrupt_resume_checkpoint_releases_materialized_snapshot(tmp_path): + import tht.jobs.dwh_pipeline as module + import pytest + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + assert pipeline.run().status == "succeeded" + baseline = set(module._SNAPSHOT_DIRS) + run_id = "e" * 32 + run_dir = tmp_path / ".tht-jobs" / "dwh" / "runs" / run_id + run_dir.mkdir(parents=True) + (run_dir / "checkpoint.json").write_text("not-json") + + with pytest.raises(Exception, match="checkpoint is invalid"): + pipeline.run(resume_run_id=run_id) + assert pipeline._snapshot_holder is None + assert set(module._SNAPSHOT_DIRS) == baseline + + +def test_job_spec_construction_failure_releases_materialized_snapshot(monkeypatch, tmp_path): + import tht.jobs.dwh_pipeline as module + import pytest + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _write_lsh([], physical, output), + ) + assert pipeline.run().status == "succeeded" + baseline = set(module._SNAPSHOT_DIRS) + captured = [] + real_materialize = module._materialize_generation_fd + + def capture(*args, **kwargs): + holder, root = real_materialize(*args, **kwargs) + captured.append(holder) + return holder, root + + monkeypatch.setattr(module, "_materialize_generation_fd", capture) + monkeypatch.setattr( + module, "JobSpec", + lambda **kwargs: (_ for _ in ()).throw(RuntimeError("job spec injected")), + ) + with pytest.raises(RuntimeError, match="job spec injected"): + pipeline.run() + assert pipeline._snapshot_holder is None + assert set(module._SNAPSHOT_DIRS) == baseline + assert captured and all(not path.exists() for path in captured) + + +def test_publish_root_swap_after_lease_never_writes_replacement(monkeypatch, tmp_path): + import tht.jobs.dwh_pipeline as module + + def make(content): + return DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text(content), + build_lsh=lambda physical, output: [ + (output / name).write_text(content) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + ) + + first = make("old").run() + assert first.status == "succeeded", first + root = tmp_path / ".tht-dwh" + moved = tmp_path / "moved-publish-root" + real_replace = module.os.replace + swapped = False + + def swapping_replace(source, destination, *args, **kwargs): + nonlocal swapped + if destination == "ACTIVE" and kwargs.get("dst_dir_fd") is not None and not swapped: + swapped = True + root.rename(moved) + root.mkdir(mode=0o700) + (root / "sentinel").write_text("replacement-safe") + return real_replace(source, destination, *args, **kwargs) + + monkeypatch.setattr(module.os, "replace", swapping_replace) + result = make("new").run() + assert result.status in {"succeeded", "failed"} + assert swapped + assert (root / "sentinel").read_text() == "replacement-safe" + moved_active = (moved / "ACTIVE").read_text().strip() + assert len(moved_active) == 32 + assert (moved / "generations" / moved_active).is_dir() + + +def test_cleanup_root_swap_after_lease_never_deletes_replacement(monkeypatch, tmp_path): + import tht.jobs.dwh_pipeline as module + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("trusted"), + build_lsh=lambda physical, output: [ + (output / name).write_text("trusted") + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + retain_generations=1, + ) + pipeline.run() + root = tmp_path / ".tht-dwh" + moved = tmp_path / "moved-cleanup-root" + real_open = module.os.open + swapped = False + + def swapping_open(path, flags, *args, **kwargs): + nonlocal swapped + if path == "generations" and kwargs.get("dir_fd") is not None and not swapped: + swapped = True + root.rename(moved) + root.mkdir(mode=0o700) + (root / "sentinel").write_text("replacement-safe") + return real_open(path, flags, *args, **kwargs) + + monkeypatch.setattr(module.os, "open", swapping_open) + pipeline._cleanup_generations() + assert swapped + assert (root / "sentinel").read_text() == "replacement-safe" + + +def test_snapshot_stays_on_one_generation_across_publish(tmp_path): + def pipeline(content): + return DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text(content), + build_lsh=lambda physical, output: [ + (output / name).write_text(content) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + ) + + first = pipeline("old").run() + cfg = snapshot_config(tmp_path) + snapshot = resolve_dwh_snapshot(cfg) + pipeline("new").run() + assert snapshot.generation == first.run_id + assert snapshot.physical.read_text() == "old" + assert (snapshot.lsh_dir / "demo_meta.json").read_text() == "old" + + +def test_generation_retention_keeps_active_and_one_rollback(tmp_path): + run_ids = [] + for index in range(5): + report = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output, i=index: output.write_text(str(i)), + build_lsh=lambda physical, output, i=index: [ + (output / name).write_text(str(i)) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + retain_generations=2, + ).run() + run_ids.append(report.run_id) + remaining = {path.name for path in (tmp_path / ".tht-dwh" / "generations").iterdir()} + assert remaining == set(run_ids[-2:]) + + +def test_corrupt_newer_directory_does_not_consume_rollback_slot(tmp_path): + run_ids = [] + pipeline = None + for index in range(3): + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output, i=index: output.write_text(str(i)), + build_lsh=lambda physical, output, i=index: [ + (output / name).write_text(str(i)) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], + retain_generations=3, + ) + run_ids.append(pipeline.run().run_id) + generations = tmp_path / ".tht-dwh" / "generations" + corrupt = generations / ("f" * 32) + corrupt.mkdir(mode=0o700) + (corrupt / "junk").write_text("not a published generation") + + pipeline.retain_generations = 2 + pipeline._cleanup_generations() + + assert (generations / run_ids[-1]).is_dir() + assert (generations / run_ids[-2]).is_dir() + assert not (generations / run_ids[0]).exists() + assert corrupt.is_dir() + + +def test_retention_n_counts_active_plus_n_minus_one_rollbacks_even_if_active_is_old(tmp_path): + import os + + run_ids = [] + pipeline = None + for index in range(3): + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output, i=index: output.write_text(str(i)), + build_lsh=lambda physical, output, i=index: [ + (output / name).write_text(str(i)) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], retain_generations=3, + ) + run_ids.append(pipeline.run().run_id) + generations = tmp_path / ".tht-dwh" / "generations" + os.utime(generations / run_ids[-1], ns=(1, 1)) + + pipeline.retain_generations = 2 + pipeline._cleanup_generations() + + remaining = {path.name for path in generations.iterdir() if path.is_dir()} + assert remaining == {run_ids[-1], run_ids[-2]} + + +def test_retention_candidate_swap_to_symlink_is_never_followed(monkeypatch, tmp_path): + import tht.jobs.dwh_pipeline as module + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("active"), + build_lsh=lambda physical, output: [ + (output / name).write_text("active") + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], retain_generations=1, + ) + pipeline.run() + generations = tmp_path / ".tht-dwh" / "generations" + candidate_name = "e" * 32 + candidate = generations / candidate_name + candidate.mkdir(mode=0o700) + external = tmp_path / "external-crafted" + external.mkdir() + sentinel = external / "sentinel" + sentinel.write_text("must-not-read-or-mutate") + real_open = module.os.open + swapped = False + + def swapping_open(path, flags, *args, **kwargs): + nonlocal swapped + if path == candidate_name and kwargs.get("dir_fd") is not None and not swapped: + swapped = True + candidate.rmdir() + candidate.symlink_to(external, target_is_directory=True) + return real_open(path, flags, *args, **kwargs) + + monkeypatch.setattr(module.os, "open", swapping_open) + pipeline._cleanup_generations() + assert swapped + assert sentinel.read_text() == "must-not-read-or-mutate" + assert candidate.is_symlink() + + +def test_reader_lease_blocks_retain_one_publisher_until_file_reads_finish(tmp_path): + import threading + import time + def make(content, retain=1): + return DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text(content), + build_lsh=lambda physical, output: [ + (output / name).write_text(content) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], retain_generations=retain, + ) + + first = make("old").run() + cfg = snapshot_config(tmp_path) + completed = threading.Event() + with lease_dwh_snapshot(cfg) as snapshot: + thread = threading.Thread(target=lambda: (make("new").run(), completed.set())) + thread.start() + time.sleep(0.05) + assert not completed.is_set() + assert snapshot.physical.read_text() == "old" + assert snapshot.generation == first.run_id + thread.join(timeout=2) + assert completed.is_set() + assert not (tmp_path / ".tht-dwh" / "generations" / first.run_id).exists() + + +def test_cleanup_never_follows_top_level_or_child_symlinks(tmp_path): + external = tmp_path / "external" + external.mkdir() + victim = external / "victim" + victim.write_text("safe") + + def make(content, retain=1): + return DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text(content), + build_lsh=lambda physical, output: [ + (output / name).write_text(content) + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json") + ], retain_generations=retain, + ) + + first = make("one", retain=2).run() + generations = tmp_path / ".tht-dwh" / "generations" + (generations / ("a" * 32)).symlink_to(external, target_is_directory=True) + make("two", retain=2).run() + old = generations / first.run_id + old.chmod(0o700) + (old / "hostile-link").symlink_to(victim) + make("three").run() + assert victim.read_text() == "safe" + assert victim.stat().st_mode & 0o200 diff --git a/harness/tests/test_evidence_port_contract.py b/harness/tests/test_evidence_port_contract.py new file mode 100644 index 00000000..eb28fc13 --- /dev/null +++ b/harness/tests/test_evidence_port_contract.py @@ -0,0 +1,239 @@ +from datetime import UTC, datetime, timedelta, timezone + +import pytest +from pydantic import ValidationError + +from tht.ports.evidence import ( + AcquiredDocument, + EvidenceSource, + EvidenceSourceError, + EvidenceSourceErrorCategory, + SourceObject, +) + + +class StubSource: + def discover(self): + return iter( + [ + SourceObject( + source_id="source:handbook", + uri="https://host/handbook.md", + fingerprint="sha256:abc", + ) + ] + ) + + def acquire(self, item: SourceObject) -> AcquiredDocument: + return AcquiredDocument( + source=item, + content=b"# Handbook", + acquired_at=datetime(2026, 7, 12, tzinfo=UTC), + media_type="text/markdown", + ) + + +def test_runtime_checkable_source_protocol(): + source = StubSource() + + assert isinstance(source, EvidenceSource) + assert source.acquire(next(source.discover())).content == b"# Handbook" + + +def test_source_objects_are_frozen_and_metadata_defaults_are_independent(): + first = SourceObject(source_id="source:a", uri="file:///a", fingerprint="sha256:a") + second = SourceObject(source_id="source:b", uri="file:///b", fingerprint="sha256:b") + + with pytest.raises(ValidationError): + first.uri = "file:///changed" # type: ignore[misc] + with pytest.raises(TypeError): + first.metadata["owner"] = "team-a" + assert second.metadata == {} + + +@pytest.mark.parametrize( + "key", + [ + "password", + "PassWd", + "api_key", + "x-api-key", + "accessToken", + "refresh.token", + "client secret", + "privateKey", + "session_cookie", + "Authorization", + ], +) +def test_source_metadata_rejects_credential_specific_keys(key): + with pytest.raises(ValidationError, match="credential-like"): + SourceObject( + source_id="source:a", + uri="https://host/a", + fingerprint="etag:abc", + metadata={"nested": [{key: "secret"}]}, + ) + + +def test_source_metadata_allows_benign_generic_token_and_secret_labels(): + source = SourceObject( + source_id="source:a", + uri="https://host/a", + fingerprint="etag:abc", + metadata={"token": "word count token", "secret": False}, + ) + + assert source.metadata["token"] == "word count token" + + +def test_nested_metadata_is_recursively_immutable_and_serializes_as_json(): + source = SourceObject( + source_id="source:a", + uri="https://host/a", + fingerprint="etag:abc", + metadata={"nested": {"items": [1, {"ok": True}]}}, + ) + + with pytest.raises(TypeError): + source.metadata["nested"]["items"][1]["ok"] = False + assert '"items":[1,{"ok":true}]' in source.model_dump_json() + + +@pytest.mark.parametrize( + "uri", + [ + "https://user:pass@host/a", + "https://host/a?api_key=secret", + "https://host/a?accessToken=secret", + ], +) +def test_source_uri_rejects_embedded_credentials(uri): + with pytest.raises(ValidationError, match="credentials"): + SourceObject(source_id="source:a", uri=uri, fingerprint="etag:abc") + + +def test_source_metadata_must_be_json_safe(): + with pytest.raises(ValidationError): + SourceObject( + source_id="source:a", + uri="file:///a", + fingerprint="sha256:a", + metadata={"path": object()}, + ) + + +def test_source_identity_and_fingerprint_must_be_namespaced(): + with pytest.raises(ValidationError, match="namespaced"): + SourceObject(source_id="plain", uri="file:///a", fingerprint="sha256:a") + with pytest.raises(ValidationError, match="namespaced"): + SourceObject(source_id="source:a", uri="file:///a", fingerprint="plain") + + +def test_acquired_document_does_not_accept_credentials_as_extra_fields(): + item = SourceObject(source_id="source:a", uri="https://host/a", fingerprint="etag:abc") + + with pytest.raises(ValidationError): + AcquiredDocument(source=item, content=b"a", api_key="secret") + + +def test_acquired_binary_content_has_explicit_json_round_trip(): + item = SourceObject(source_id="source:a", uri="https://host/a", fingerprint="etag:abc") + acquired = AcquiredDocument(source=item, content=b"\x00\xffbinary\x80") + + payload = acquired.model_dump_json() + restored = AcquiredDocument.model_validate_json(payload) + + assert restored.content == acquired.content + assert "binary" not in payload + + +def test_datetimes_must_be_aware_and_are_normalized_to_utc(): + with pytest.raises(ValidationError, match="timezone-aware"): + SourceObject( + source_id="source:a", + uri="https://host/a", + fingerprint="etag:abc", + modified_at=datetime(2026, 7, 12), + ) + + source = SourceObject( + source_id="source:a", + uri="https://host/a", + fingerprint="etag:abc", + modified_at=datetime(2026, 7, 12, 4, tzinfo=timezone(timedelta(hours=2))), + ) + assert source.modified_at.tzinfo is UTC + assert source.modified_at.hour == 2 + + acquired = AcquiredDocument( + source=source, + content=b"a", + acquired_at=datetime(2026, 7, 12, 2, tzinfo=UTC) + timedelta(hours=0), + ) + assert acquired.acquired_at.utcoffset() == timedelta(0) + + +def test_source_errors_are_typed_retryable_and_safe(): + transient = EvidenceSourceError( + "password=hunter2 at https://user:secret@host", + category=EvidenceSourceErrorCategory.TRANSIENT, + details={"status": 503}, + ) + permanent = EvidenceSourceError( + "unsupported media type", + category=EvidenceSourceErrorCategory.PERMANENT, + ) + + assert transient.retryable is True + assert permanent.retryable is False + assert transient.details["status"] == 503 + assert str(transient) == "evidence source operation failed" + assert transient.args == ("evidence source operation failed",) + assert "hunter2" not in repr(transient) + with pytest.raises(AttributeError): + transient.category = EvidenceSourceErrorCategory.PERMANENT + with pytest.raises(AttributeError): + transient.args = ("leak",) + with pytest.raises(AttributeError): + transient.details = {"unsafe": True} + assert "hunter2" not in repr(transient.__dict__) + with pytest.raises(TypeError): + transient.details["status"] = 200 + with pytest.raises(ValueError, match="credential-like"): + EvidenceSourceError( + "bad", + category=EvidenceSourceErrorCategory.PERMANENT, + details={"apiKey": "must-not-leak"}, + ) + with pytest.raises(ValidationError): + EvidenceSourceError( + "bad", + category=EvidenceSourceErrorCategory.PERMANENT, + details={"not_json": object()}, + ) + + +def test_source_error_preserves_original_only_through_exception_chaining(): + cause = RuntimeError("transport diagnostic with password=hunter2") + error = EvidenceSourceError( + "ignored unsafe diagnostic", + category=EvidenceSourceErrorCategory.TRANSIENT, + ) + + try: + raise error from cause + except EvidenceSourceError as caught: + assert caught.__cause__ is cause + assert "hunter2" not in str(caught) + assert "hunter2" not in caught.args + + +def test_model_copy_revalidates_source_and_acquired_records(): + source = SourceObject(source_id="source:a", uri="file:///a", fingerprint="sha256:a") + acquired = AcquiredDocument(source=source, content=b"a") + + with pytest.raises(ValidationError, match="namespaced"): + source.model_copy(update={"source_id": "invalid"}) + with pytest.raises(ValidationError, match="timezone-aware"): + acquired.model_copy(update={"acquired_at": datetime(2026, 7, 12)}) diff --git a/harness/tests/test_filesystem_evidence_source.py b/harness/tests/test_filesystem_evidence_source.py new file mode 100644 index 00000000..3f93789d --- /dev/null +++ b/harness/tests/test_filesystem_evidence_source.py @@ -0,0 +1,113 @@ +import os + +import pytest + +from tht.adapters.evidence import FilesystemEvidenceSource +from tht.ports.evidence import EvidenceSourceError + + +def test_filesystem_discovery_is_stable_and_acquisition_is_bounded(tmp_path): + (tmp_path / "z.md").write_text("z") + (tmp_path / "nested").mkdir() + (tmp_path / "nested" / "a.md").write_text("alpha") + source = FilesystemEvidenceSource(tmp_path, max_bytes=5) + + first = list(source.discover()) + assert [item.uri for item in first] == sorted(item.uri for item in first) + assert all(item.source_id.startswith("filesystem:") for item in first) + assert all(item.fingerprint.startswith("sha256:") for item in first) + assert source.acquire(first[0]).content in {b"alpha", b"z"} + + (tmp_path / "large.md").write_bytes(b"123456") + with pytest.raises(EvidenceSourceError) as caught: + list(source.discover()) + assert not caught.value.retryable + assert "large.md" not in str(caught.value) + + +def test_filesystem_rejects_symlink_escape(tmp_path): + root = tmp_path / "root" + root.mkdir() + outside = tmp_path / "secret.md" + outside.write_text("secret") + (root / "escape.md").symlink_to(outside) + + with pytest.raises(EvidenceSourceError) as caught: + list(FilesystemEvidenceSource(root).discover()) + assert not caught.value.retryable + assert str(outside) not in str(caught.value) + + +def test_filesystem_acquire_rejects_object_from_another_source(tmp_path): + left = tmp_path / "left" + right = tmp_path / "right" + left.mkdir() + right.mkdir() + (left / "doc.md").write_text("left") + (right / "doc.md").write_text("right") + item = next(iter(FilesystemEvidenceSource(left).discover())) + + with pytest.raises(EvidenceSourceError): + FilesystemEvidenceSource(right).acquire(item) + + +def test_filesystem_acquire_rejects_content_changed_since_discovery(tmp_path): + path = tmp_path / "doc.md" + path.write_text("first") + source = FilesystemEvidenceSource(tmp_path) + item = next(iter(source.discover())) + path.write_text("second") + + with pytest.raises(EvidenceSourceError) as caught: + source.acquire(item) + assert not caught.value.retryable + + +def test_filesystem_open_is_safe_when_file_is_swapped_for_symlink(tmp_path, monkeypatch): + root = tmp_path / "root" + root.mkdir() + path = root / "doc.md" + path.write_text("safe") + outside = tmp_path / "outside.md" + outside.write_text("secret") + source = FilesystemEvidenceSource(root) + real_open = os.open + swapped = False + + def racing_open(name, flags, *args, **kwargs): + nonlocal swapped + if name == "doc.md" and not swapped: + swapped = True + path.unlink() + path.symlink_to(outside) + return real_open(name, flags, *args, **kwargs) + + monkeypatch.setattr(os, "open", racing_open) + with pytest.raises(EvidenceSourceError): + list(source.discover()) + + +def test_filesystem_open_is_safe_when_ancestor_is_swapped_for_symlink(tmp_path, monkeypatch): + root = tmp_path / "root" + nested = root / "nested" + nested.mkdir(parents=True) + (nested / "doc.md").write_text("safe") + outside = tmp_path / "outside" + outside.mkdir() + (outside / "doc.md").write_text("secret") + source = FilesystemEvidenceSource(root) + real_open = os.open + swapped = False + + def racing_open(name, flags, *args, **kwargs): + nonlocal swapped + if name == "nested" and not swapped and kwargs.get("dir_fd") is not None: + swapped = True + (nested / "doc.md").unlink() + nested.rmdir() + nested.symlink_to(outside, target_is_directory=True) + return real_open(name, flags, *args, **kwargs) + + monkeypatch.setattr(os, "open", racing_open) + with pytest.raises(EvidenceSourceError): + list(source.discover()) diff --git a/harness/tests/test_http_evidence_source.py b/harness/tests/test_http_evidence_source.py new file mode 100644 index 00000000..38883dea --- /dev/null +++ b/harness/tests/test_http_evidence_source.py @@ -0,0 +1,331 @@ +import threading +import socket +from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer + +import pytest + +from tht.adapters.evidence import HttpManifestEvidenceSource +from tht.ports.evidence import EvidenceSourceError + + +class Handler(BaseHTTPRequestHandler): + etag_requests = 0 + etag_body_responses = 0 + redirect_target = "/redirected-v1" + redirect_request_validators = [] + final_request_validators = [] + + def do_GET(self): + if self.path.startswith("/etag"): + type(self).etag_requests += 1 + if self.headers.get("If-None-Match") == '"abc"': + self.send_response(304) + self.end_headers() + return + self.send_response(200) + self.send_header("ETag", '"abc"') + self.send_header("Content-Type", "text/markdown") + self.end_headers() + type(self).etag_body_responses += 1 + self.wfile.write(b"hello") + elif self.path == "/large": + self.send_response(200) + self.send_header("Content-Length", "20") + self.end_headers() + self.wfile.write(b"x" * 20) + elif self.path == "/busy": + self.send_response(503) + self.end_headers() + elif self.path == "/missing": + self.send_response(404) + self.end_headers() + elif self.path == "/redirect-private": + self.send_response(302) + self.send_header("Location", f"http://127.0.0.1:{self.server.server_port}/etag") + self.end_headers() + elif self.path == "/redirect-userinfo": + self.send_response(302) + self.send_header( + "Location", f"http://user:password@127.0.0.1:{self.server.server_port}/etag" + ) + self.end_headers() + elif self.path == "/stable-redirect": + type(self).redirect_request_validators.append(self.headers.get("If-None-Match")) + self.send_response(302) + self.send_header("Location", type(self).redirect_target) + self.end_headers() + elif self.path in {"/redirected-v1", "/redirected-v2"}: + type(self).final_request_validators.append( + (self.path, self.headers.get("If-None-Match")) + ) + etag = '"v1"' if self.path.endswith("v1") else '"v2"' + if self.headers.get("If-None-Match") == etag: + self.send_response(304) + self.end_headers() + return + self.send_response(200) + self.send_header("ETag", etag) + self.end_headers() + self.wfile.write(self.path.encode()) + else: + self.send_response(200) + self.send_header("Last-Modified", "Wed, 21 Oct 2015 07:28:00 GMT") + self.end_headers() + self.wfile.write(b"fallback") + + def log_message(self, format, *args): + pass + + +@pytest.fixture +def server_url(): + Handler.etag_requests = 0 + Handler.etag_body_responses = 0 + Handler.redirect_target = "/redirected-v1" + Handler.redirect_request_validators = [] + Handler.final_request_validators = [] + server = ThreadingHTTPServer(("127.0.0.1", 0), Handler) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + yield f"http://127.0.0.1:{server.server_port}" + finally: + server.shutdown() + thread.join() + + +def test_http_uses_etag_and_strips_query_from_provenance(server_url): + source = HttpManifestEvidenceSource( + [f"{server_url}/etag?token=secret"], allow_private_hosts=True + ) + item = next(iter(source.discover())) + + assert item.fingerprint.startswith("etag:") + assert item.fingerprint != "etag:abc" + assert item.uri == f"{server_url}/etag" + assert "secret" not in item.model_dump_json() + assert source.acquire(item).content == b"hello" + + +def test_http_uses_last_modified_then_content_hash(server_url): + modified = next(iter(HttpManifestEvidenceSource( + [f"{server_url}/modified"], allow_private_hosts=True + ).discover())) + assert modified.fingerprint.startswith("last-modified:") + + class NoValidators(Handler): + def do_GET(self): + self.send_response(200) + self.end_headers() + self.wfile.write(b"content") + + server = ThreadingHTTPServer(("127.0.0.1", 0), NoValidators) + thread = threading.Thread(target=server.serve_forever, daemon=True) + thread.start() + try: + item = next(iter(HttpManifestEvidenceSource( + [f"http://127.0.0.1:{server.server_port}/doc"], allow_private_hosts=True + ).discover())) + assert item.fingerprint.startswith("sha256:") + finally: + server.shutdown() + thread.join() + + +@pytest.mark.parametrize("path,retryable", [("/busy", True), ("/missing", False)]) +def test_http_classifies_status_errors(server_url, path, retryable): + with pytest.raises(EvidenceSourceError) as caught: + list(HttpManifestEvidenceSource( + [server_url + path], allow_private_hosts=True + ).discover()) + assert caught.value.retryable is retryable + assert server_url not in str(caught.value) + + +def test_http_rejects_oversize_and_private_redirect(server_url): + with pytest.raises(EvidenceSourceError) as large: + list(HttpManifestEvidenceSource( + [server_url + "/large"], max_bytes=10, allow_private_hosts=True + ).discover()) + assert not large.value.retryable + + with pytest.raises(EvidenceSourceError) as redirect: + list(HttpManifestEvidenceSource([server_url + "/redirect-private"]).discover()) + assert not redirect.value.retryable + + +def test_http_rejects_unsupported_manifest_scheme(): + with pytest.raises(ValueError, match="http"): + HttpManifestEvidenceSource(["file:///tmp/secret"]) + + +def test_http_conditional_discovery_reuses_cached_verified_bytes(server_url): + source = HttpManifestEvidenceSource([server_url + "/etag"], allow_private_hosts=True) + first = next(iter(source.discover())) + second = next(iter(source.discover())) + + assert second == first + assert source.acquire(second).content == b"hello" + assert Handler.etag_requests == 3 + assert Handler.etag_body_responses == 1 + + +def test_http_rejects_mixed_public_private_dns_answers(monkeypatch): + monkeypatch.setattr(socket, "getaddrinfo", lambda *args, **kwargs: [ + (socket.AF_INET, socket.SOCK_STREAM, 6, "", ("93.184.216.34", 80)), + (socket.AF_INET6, socket.SOCK_STREAM, 6, "", ("::1", 80, 0, 0)), + ]) + with pytest.raises(EvidenceSourceError) as caught: + list(HttpManifestEvidenceSource(["http://example.test/doc"]).discover()) + assert not caught.value.retryable + + +def test_http_rejects_userinfo_redirect(server_url): + with pytest.raises(EvidenceSourceError) as caught: + list(HttpManifestEvidenceSource( + [server_url + "/redirect-userinfo"], allow_private_hosts=True + ).discover()) + assert not caught.value.retryable + + +class FakeSocket: + def __init__(self, address): + self.address = address + + def getpeername(self): + return (self.address, 443) + + +class FakeResponse: + status_code = 200 + headers = {} + is_redirect = False + + def __init__(self, *, peer="127.0.0.1", stream_error=None, location=None): + connection = type("Connection", (), {"sock": FakeSocket(peer)})() + self.raw = type("Raw", (), {"_connection": connection})() + self.stream_error = stream_error + self.closed = False + if location: + self.is_redirect = True + self.status_code = 302 + self.headers = {"Location": location} + else: + self.is_redirect = False + self.status_code = 200 + self.headers = {} + + def iter_content(self, chunk_size): + if self.stream_error: + raise self.stream_error + yield b"ok" + + def close(self): + self.closed = True + + +class FakeSession: + def __init__(self, response): + self.response = response + + def get(self, *args, **kwargs): + return self.response + + +def test_http_rejects_public_to_private_rebind(monkeypatch): + monkeypatch.setattr(socket, "getaddrinfo", lambda *args, **kwargs: [ + (socket.AF_INET, socket.SOCK_STREAM, 6, "", ("93.184.216.34", 443)), + ]) + source = HttpManifestEvidenceSource(["https://example.test/doc"]) + response = FakeResponse(peer="127.0.0.1") + source._session = FakeSession(response) + with pytest.raises(EvidenceSourceError): + list(source.discover()) + assert response.closed + + +def test_http_rejects_public_redirect_to_private_destination(monkeypatch): + monkeypatch.setattr(socket, "getaddrinfo", lambda host, *args, **kwargs: [ + (socket.AF_INET, socket.SOCK_STREAM, 6, "", ( + "93.184.216.34" if host == "example.test" else "127.0.0.1", 443 + )), + ]) + source = HttpManifestEvidenceSource(["https://example.test/doc"]) + response = FakeResponse( + peer="93.184.216.34", location="https://private.test/secret" + ) + source._session = FakeSession(response) + with pytest.raises(EvidenceSourceError) as caught: + list(source.discover()) + assert not caught.value.retryable + assert response.closed + + +def test_http_closes_response_when_streaming_fails(): + source = HttpManifestEvidenceSource( + ["https://example.test/doc"], allow_private_hosts=True + ) + response = FakeResponse(stream_error=socket.timeout("read timed out")) + source._session = FakeSession(response) + with pytest.raises(EvidenceSourceError): + list(source.discover()) + assert response.closed + + +def test_http_binds_validators_to_exact_final_redirect_url(server_url): + source = HttpManifestEvidenceSource( + [server_url + "/stable-redirect"], allow_private_hosts=True + ) + first = next(iter(source.discover())) + second = next(iter(source.discover())) + + assert first == second + assert Handler.redirect_request_validators == [None, None] + assert Handler.final_request_validators == [ + ("/redirected-v1", None), + ("/redirected-v1", '"v1"'), + ] + + +def test_http_redirect_path_change_fetches_and_replaces_body(server_url): + source = HttpManifestEvidenceSource( + [server_url + "/stable-redirect"], allow_private_hosts=True + ) + first = next(iter(source.discover())) + assert source.acquire(first).content == b"/redirected-v1" + + Handler.redirect_target = "/redirected-v2" + with pytest.raises(EvidenceSourceError): + source.acquire(first) + current = next(iter(source.discover())) + + assert source.acquire(current).content == b"/redirected-v2" + assert ("/redirected-v2", None) in Handler.final_request_validators + + +def test_http_rejects_unsolicited_304_without_bound_validator(monkeypatch): + monkeypatch.setattr(socket, "getaddrinfo", lambda *args, **kwargs: [ + (socket.AF_INET, socket.SOCK_STREAM, 6, "", ("93.184.216.34", 443)), + ]) + source = HttpManifestEvidenceSource(["https://example.test/doc"]) + response = FakeResponse(peer="93.184.216.34") + response.status_code = 304 + source._session = FakeSession(response) + + with pytest.raises(EvidenceSourceError) as caught: + list(source.discover()) + assert not caught.value.retryable + assert response.closed + + +def test_http_rejects_cross_origin_304_for_cached_provenance(server_url): + source = HttpManifestEvidenceSource([server_url + "/etag"], allow_private_hosts=True) + item = next(iter(source.discover())) + response = FakeResponse() + response.status_code = 304 + source._session = FakeSession(response) + + with pytest.raises(EvidenceSourceError) as caught: + source._download("https://other.example/doc", item.uri) + assert not caught.value.retryable + assert response.closed diff --git a/harness/tests/test_job_locking.py b/harness/tests/test_job_locking.py new file mode 100644 index 00000000..0d952583 --- /dev/null +++ b/harness/tests/test_job_locking.py @@ -0,0 +1,92 @@ +import multiprocessing +import os +from pathlib import Path + +import pytest + +from tht.jobs.locking import JobAlreadyRunningError, WorkspaceJobLock + + +def _hold_lock(root: str, ready, release): + with WorkspaceJobLock(Path(root), "demo", "evidence"): + ready.set() + release.wait(10) + + +def _crash_with_lock(root: str, ready): + lock = WorkspaceJobLock(Path(root), "demo", "evidence") + lock.acquire() + ready.set() + raise SystemExit(7) + + +def test_same_workspace_and_job_are_exclusive_across_processes(tmp_path): + context = multiprocessing.get_context("spawn") + ready = context.Event() + release = context.Event() + process = context.Process(target=_hold_lock, args=(str(tmp_path), ready, release)) + process.start() + assert ready.wait(10) + try: + with pytest.raises(JobAlreadyRunningError): + WorkspaceJobLock(tmp_path, "demo", "evidence").acquire() + finally: + release.set() + process.join(10) + assert process.exitcode == 0 + + +def test_evidence_and_dwh_jobs_have_distinct_locks(tmp_path): + with WorkspaceJobLock(tmp_path, "demo", "evidence"): + with WorkspaceJobLock(tmp_path, "demo", "dwh"): + pass + + +def test_lock_keys_cannot_escape_lock_directory(tmp_path): + with pytest.raises(ValueError, match="filesystem-safe"): + WorkspaceJobLock(tmp_path, "demo", "../evidence") + + +def test_preexisting_lock_symlink_is_rejected(tmp_path): + lock = WorkspaceJobLock(tmp_path, "demo", "evidence") + lock.path.parent.mkdir(parents=True) + target = tmp_path / "target" + target.write_text("do not modify") + lock.path.symlink_to(target) + with pytest.raises(OSError): + lock.acquire() + assert target.read_text() == "do not modify" + + +def test_preexisting_locks_directory_symlink_is_rejected(tmp_path): + jobs = tmp_path / ".tht-jobs" + jobs.mkdir() + outside = tmp_path / "outside" + outside.mkdir() + (jobs / ".locks").symlink_to(outside, target_is_directory=True) + with pytest.raises(OSError): + WorkspaceJobLock(tmp_path, "demo", "evidence").acquire() + assert list(outside.iterdir()) == [] + + +def test_lock_file_is_owner_only_regular_single_link(tmp_path): + with WorkspaceJobLock(tmp_path, "demo", "evidence") as lock: + stat = os.stat(lock.path, follow_symlinks=False) + assert stat.st_uid == os.getuid() + assert stat.st_nlink == 1 + assert stat.st_mode & 0o777 == 0o600 + + +def test_lock_is_recoverable_after_process_crash_without_stale_deletion(tmp_path): + context = multiprocessing.get_context("spawn") + ready = context.Event() + process = context.Process(target=_crash_with_lock, args=(str(tmp_path), ready)) + process.start() + assert ready.wait(10) + process.join(10) + assert process.exitcode == 7 + + lock_path = WorkspaceJobLock(tmp_path, "demo", "evidence").path + assert lock_path.exists() + with WorkspaceJobLock(tmp_path, "demo", "evidence"): + assert lock_path.exists() diff --git a/harness/tests/test_job_runner.py b/harness/tests/test_job_runner.py new file mode 100644 index 00000000..0874b047 --- /dev/null +++ b/harness/tests/test_job_runner.py @@ -0,0 +1,400 @@ +import json +import os +import pytest +from pydantic import ValidationError + +from tht.jobs.models import JobSpec +from tht.jobs.runner import CorruptCheckpointError, StageArtifacts, run_job +import tht.jobs.runner as runner_module + + +def _spec(tmp_path, **updates): + values = { + "workspace_id": "demo", + "job_type": "evidence", + "workspace_root": tmp_path, + "spec_version": "jobs-v1", + "pipeline_version": "evidence-v1", + "config_fingerprint": "sha256:" + "1" * 64, + "input_fingerprint": "sha256:" + "2" * 64, + "stage_ids": ("stage",), + } + values.update(updates) + return JobSpec(**values) + + +def test_job_models_are_immutable(tmp_path): + spec = _spec(tmp_path) + with pytest.raises(ValidationError, match="Instance is frozen"): + spec.job_type = "dwh" + + resumed = spec.with_resume("a" * 32) + assert resumed.workspace_root == tmp_path + assert resumed.resume_run_id == "a" * 32 + + +def test_failed_stage_is_resumable_and_skips_completed_stage(tmp_path): + calls = [] + + def discover(context): + calls.append(("discover", context.dry_run)) + + def acquire(_context): + calls.append(("acquire", False)) + raise RuntimeError("source /customer/alice token=secret unavailable") + + first = run_job(_spec(tmp_path, stage_ids=("discover", "acquire")), [discover, acquire]) + assert first.status == "failed" + assert [stage.status for stage in first.stages] == ["succeeded", "failed"] + assert first.stages[1].error.model_dump() == { + "category": "internal", + "code": "stage_exception", + "message": "stage execution failed", + } + + def acquire(_context): + calls.append(("recovered", False)) + + second = run_job( + _spec(tmp_path, resume_run_id=first.run_id, stage_ids=("discover", "acquire")), + [discover, acquire], + ) + assert second.resumed_from == first.run_id + assert second.status == "succeeded" + assert calls == [("discover", False), ("acquire", False), ("recovered", False)] + + +def test_resume_carries_successful_stage_artifacts_into_new_run(tmp_path): + def discover(context): + artifacts = context.run_dir / "artifacts" + artifacts.mkdir() + (artifacts / "discovery.json").write_text('{"source":"one"}') + return StageArtifacts(("discovery.json",)) + + first = run_job( + _spec(tmp_path, stage_ids=("discover", "acquire")), + [discover, lambda _context: (_ for _ in ()).throw(RuntimeError("crash"))], + ) + + def acquire(context): + assert (context.run_dir / "artifacts" / "discovery.json").read_text() == '{"source":"one"}' + + resumed = run_job( + _spec(tmp_path, resume_run_id=first.run_id, stage_ids=("discover", "acquire")), + [discover, acquire], + ) + assert resumed.status == "succeeded" + + +def test_crash_after_stage_effect_resumes_without_repeating_stage(tmp_path): + calls = [] + + def stage(context): + calls.append("stage") + artifacts = context.run_dir / "artifacts" + artifacts.mkdir() + (artifacts / "effect.json").write_text("ok") + return StageArtifacts(("effect.json",)) + + class Crash(BaseException): + pass + + with pytest.raises(Crash): + run_job( + _spec(tmp_path), [stage], + after_stage_return=lambda *_: (_ for _ in ()).throw(Crash()), + ) + runs = tmp_path / ".tht-jobs" / "evidence" / "runs" + crashed_run = next(runs.iterdir()).name + resumed = run_job(_spec(tmp_path).with_resume(crashed_run), [stage]) + assert resumed.status == "succeeded" + assert calls == ["stage"] + + +def test_resume_rejects_tampered_successful_stage_artifact(tmp_path): + def stage(context): + artifacts = context.run_dir / "artifacts" + artifacts.mkdir() + (artifacts / "effect.json").write_text("ok") + return StageArtifacts(("effect.json",)) + + report = run_job(_spec(tmp_path), [stage]) + path = tmp_path / ".tht-jobs" / "evidence" / "runs" / report.run_id / "artifacts" / "effect.json" + path.write_text("tampered") + with pytest.raises(CorruptCheckpointError, match="artifact"): + run_job(_spec(tmp_path).with_resume(report.run_id), [stage]) + + +def test_resume_rejects_extra_symlink_before_any_stage(tmp_path): + report = run_job(_spec(tmp_path), [lambda _context: StageArtifacts()]) + artifacts = tmp_path / ".tht-jobs" / "evidence" / "runs" / report.run_id / "artifacts" + (artifacts / "unsafe").symlink_to(tmp_path) + called = False + + def forbidden(_context): + nonlocal called + called = True + + with pytest.raises(CorruptCheckpointError, match="artifact"): + run_job(_spec(tmp_path).with_resume(report.run_id), [forbidden]) + assert called is False + + +def test_nonexistent_well_formed_resume_run_id_is_rejected(tmp_path): + with pytest.raises(CorruptCheckpointError, match="checkpoint"): + run_job(_spec(tmp_path).with_resume("a" * 32), [lambda _context: None]) + + +@pytest.mark.parametrize("tamper", ["artifact_and_manifest", "spec", "producer"]) +def test_resume_rejects_manifest_root_or_binding_tamper(tmp_path, tamper): + def stage(context): + artifacts = context.run_dir / "artifacts" + artifacts.mkdir() + (artifacts / "effect.json").write_text("ok") + return StageArtifacts(("effect.json",)) + + report = run_job(_spec(tmp_path), [stage]) + artifacts = tmp_path / ".tht-jobs" / "evidence" / "runs" / report.run_id / "artifacts" + manifest_path = artifacts / "artifact-manifest.json" + manifest = json.loads(manifest_path.read_text()) + if tamper == "artifact_and_manifest": + (artifacts / "effect.json").write_text("evil") + digest = __import__("hashlib").sha256(b"evil").hexdigest() + manifest["stages"]["stage"]["files"]["effect.json"] = { + "sha256": digest, "size": 4, + } + elif tamper == "spec": + manifest["spec_fingerprint"] = "sha256:" + "0" * 64 + else: + manifest["stages"]["other"] = manifest["stages"].pop("stage") + manifest_path.write_text(json.dumps(manifest, sort_keys=True, separators=(",", ":")) + "\n") + + with pytest.raises(CorruptCheckpointError, match="artifact"): + run_job(_spec(tmp_path).with_resume(report.run_id), [stage]) + + +def test_successful_job_is_idempotently_resumable(tmp_path): + calls = [] + + def normalize(_context): + calls.append("normalize") + + first = run_job(_spec(tmp_path), [normalize]) + second = run_job(_spec(tmp_path, resume_run_id=first.run_id), [normalize]) + assert first.status == second.status == "succeeded" + assert second.resumed_from == first.run_id + assert calls == ["normalize"] + + +def test_checkpoints_and_report_are_json_safe_and_do_not_disclose_workspace_path(tmp_path): + def publish(_context): + return {"ignored": "/customer/alice", "password": "secret"} + + report = run_job(_spec(tmp_path), [publish]) + run_dir = tmp_path / ".tht-jobs" / "evidence" / "runs" / report.run_id + checkpoint = json.loads((run_dir / "checkpoint.json").read_text()) + payload = (run_dir / "report.json").read_text() + parsed = json.loads(payload) + + assert checkpoint["status"] == "succeeded" + assert parsed["schema_version"] == 1 + assert parsed["run_id"] == report.run_id + assert str(tmp_path) not in payload + assert "alice" not in payload + assert "secret" not in payload + assert parsed["started_at"].endswith("Z") + assert parsed["finished_at"].endswith("Z") + + +def test_dry_run_is_exposed_to_stages_and_report(tmp_path): + observed = [] + + def plan(context): + observed.append(context.dry_run) + + report = run_job(_spec(tmp_path, dry_run=True), [plan]) + assert observed == [True] + assert report.dry_run is True + assert report.status == "succeeded" + + +def test_corrupt_checkpoint_is_rejected_without_running_stages(tmp_path): + first = run_job(_spec(tmp_path), [lambda _context: None]) + checkpoint = ( + tmp_path / ".tht-jobs" / "evidence" / "runs" / first.run_id / "checkpoint.json" + ) + checkpoint.write_text("{not-json") + called = False + + def stage(_context): + nonlocal called + called = True + + with pytest.raises(CorruptCheckpointError, match="checkpoint is invalid"): + run_job(_spec(tmp_path, resume_run_id=first.run_id), [stage]) + assert called is False + + +def test_stage_timestamps_are_aware_and_ordered(tmp_path): + report = run_job(_spec(tmp_path), [lambda _context: None]) + stage = report.stages[0] + assert stage.started_at.tzinfo is not None + assert stage.finished_at.tzinfo is not None + assert stage.started_at <= stage.finished_at + assert report.started_at <= stage.started_at <= report.finished_at + + +@pytest.mark.parametrize( + ("update", "replacement"), + [ + ("dry_run", True), + ("spec_version", "jobs-v2"), + ("pipeline_version", "evidence-v2"), + ("config_fingerprint", "sha256:" + "3" * 64), + ("input_fingerprint", "sha256:" + "4" * 64), + ("workspace_id", "other"), + ("job_type", "dwh"), + ], +) +def test_resume_rejects_changed_identity_or_inputs_before_stage_execution( + tmp_path, update, replacement +): + first = run_job(_spec(tmp_path), [lambda _context: None]) + called = False + + def stage(_context): + nonlocal called + called = True + + values = {update: replacement, "resume_run_id": first.run_id} + with pytest.raises(CorruptCheckpointError, match="incompatible"): + run_job(_spec(tmp_path, **values), [stage]) + assert called is False + + +@pytest.mark.parametrize("stages", [[], [lambda _context: None, lambda _context: None]]) +def test_resume_rejects_removed_or_inserted_stages(tmp_path, stages): + def first_stage(_context): + pass + + first = run_job(_spec(tmp_path), [first_stage]) + with pytest.raises(CorruptCheckpointError, match="incompatible"): + run_job(_spec(tmp_path, resume_run_id=first.run_id), stages) + + +def test_resume_rejects_reordered_stages(tmp_path): + def one(_context): + pass + + def two(_context): + pass + + first = run_job(_spec(tmp_path, stage_ids=("one", "two")), [one, two]) + with pytest.raises(CorruptCheckpointError, match="incompatible"): + run_job( + _spec(tmp_path, resume_run_id=first.run_id, stage_ids=("two", "one")), + [two, one], + ) + + +def test_hostile_exception_identity_never_enters_terminal_report(tmp_path): + Hostile = type("ApiKey_secret_/customer/alice", (Exception,), {}) + + def fail(_context): + raise Hostile("password=hunter2") + + report = run_job(_spec(tmp_path), [fail]) + payload = report.model_dump_json() + assert report.status == "failed" + assert report.stages[0].error.model_dump() == { + "category": "internal", + "code": "stage_exception", + "message": "stage execution failed", + } + assert "secret" not in payload + assert "alice" not in payload + assert "hunter2" not in payload + + +def test_new_run_without_resume_allows_intentional_spec_change(tmp_path): + first = run_job(_spec(tmp_path), [lambda _context: None]) + second = run_job(_spec(tmp_path, input_fingerprint="sha256:" + "9" * 64), [lambda _context: None]) + assert second.status == "succeeded" + assert second.run_id != first.run_id + assert second.resumed_from is None + + +def test_run_directories_are_private_and_fsynced_before_atomic_replace(tmp_path, monkeypatch): + events = [] + real_replace = os.replace + + monkeypatch.setattr(runner_module.os, "fsync", lambda _fd: events.append("fsync")) + + def tracked_replace(source, destination): + events.append("replace") + real_replace(source, destination) + + monkeypatch.setattr(runner_module.os, "replace", tracked_replace) + report = run_job(_spec(tmp_path), [lambda _context: None]) + run_dir = tmp_path / ".tht-jobs" / "evidence" / "runs" / report.run_id + + assert run_dir.stat().st_mode & 0o777 == 0o700 + first_replace = events.index("replace") + assert "fsync" in events[:first_replace] + assert "fsync" in events[first_replace + 1 :] + + +def _tamper_checkpoint(tmp_path, report, transform): + path = tmp_path / ".tht-jobs" / report.job_type / "runs" / report.run_id / "checkpoint.json" + payload = json.loads(path.read_text()) + transform(payload) + path.write_text(json.dumps(payload)) + + +@pytest.mark.parametrize( + "transform", + [ + lambda payload: payload["stages"].pop(), + lambda payload: payload["stages"].reverse(), + lambda payload: payload["stages"].append(dict(payload["stages"][0])), + lambda payload: payload["stages"][0].update(name="substitute"), + lambda payload: payload.update(input_fingerprint="sha256:" + "f" * 64), + lambda payload: payload["stages"][0].update(status="pending"), + ], +) +def test_semantically_tampered_checkpoint_fails_without_orphan_run(tmp_path, transform): + def one(_context): + pass + + def two(_context): + pass + + spec = _spec(tmp_path, stage_ids=("one", "two")) + first = run_job(spec, [one, two]) + runs = tmp_path / ".tht-jobs" / "evidence" / "runs" + before = {path.name for path in runs.iterdir()} + _tamper_checkpoint(tmp_path, first, transform) + called = False + + def forbidden(_context): + nonlocal called + called = True + + with pytest.raises(CorruptCheckpointError, match="invalid|incompatible"): + run_job(spec.with_resume(first.run_id), [forbidden, forbidden]) + assert called is False + assert {path.name for path in runs.iterdir()} == before + + +def test_stored_compatibility_fingerprint_tamper_fails_without_orphan(tmp_path): + first = run_job(_spec(tmp_path), [lambda _context: None]) + runs = tmp_path / ".tht-jobs" / "evidence" / "runs" + before = {path.name for path in runs.iterdir()} + _tamper_checkpoint( + tmp_path, + first, + lambda payload: payload.update(compatibility_fingerprint="sha256:" + "0" * 64), + ) + with pytest.raises(CorruptCheckpointError, match="invalid"): + run_job(_spec(tmp_path, resume_run_id=first.run_id), [lambda _context: None]) + assert {path.name for path in runs.iterdir()} == before diff --git a/harness/tests/test_lsh_job_resume.py b/harness/tests/test_lsh_job_resume.py new file mode 100644 index 00000000..81d9c4a0 --- /dev/null +++ b/harness/tests/test_lsh_job_resume.py @@ -0,0 +1,151 @@ +import hashlib + +import pytest + +from tht.jobs.dwh_pipeline import DwhPreprocessPipeline + + +FP = "sha256:" + hashlib.sha256(b"test").hexdigest() + + +def test_lsh_failure_resumes_exact_run_without_repeating_introspection(tmp_path): + calls = [] + + def introspect(output): + calls.append("introspect") + output.write_text("catalog") + + def fail_lsh(physical, output): + calls.append("lsh-failed") + raise RuntimeError("database detail that must not leak") + + failed = DwhPreprocessPipeline( + workspace_id="demo", + workspace_root=tmp_path, + config_fingerprint=FP, + input_fingerprint=FP, + introspect=introspect, + build_lsh=fail_lsh, + ).run(("introspect", "lsh")) + assert failed.status == "failed" + + resumed = DwhPreprocessPipeline( + workspace_id="demo", + workspace_root=tmp_path, + config_fingerprint=FP, + input_fingerprint=FP, + introspect=introspect, + build_lsh=lambda physical, output: _recover_lsh(calls, output), + ).run(("introspect", "lsh"), resume_run_id=failed.run_id) + + assert resumed.status == "succeeded" + assert resumed.resumed_from == failed.run_id + assert calls == ["introspect", "lsh-failed", "lsh-recovered"] + + +def _recover_lsh(calls, output): + calls.append("lsh-recovered") + for name in ("demo_lsh.pkl", "demo_minhashes.pkl", "demo_meta.json"): + (output / name).write_text(name) + + +def test_resume_rejects_a_different_stage_selection(tmp_path): + failed = DwhPreprocessPipeline( + workspace_id="demo", + workspace_root=tmp_path, + config_fingerprint=FP, + input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: (_ for _ in ()).throw(RuntimeError()), + ).run(("introspect", "lsh")) + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", + workspace_root=tmp_path, + config_fingerprint=FP, + input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: None, + ) + try: + pipeline.run(("lsh",), resume_run_id=failed.run_id) + except Exception as error: + assert "incompatible" in str(error) + else: + raise AssertionError("resume with different stages must fail") + + +def test_resume_rejects_tampered_succeeded_stage_artifact(tmp_path): + failed = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: (_ for _ in ()).throw(RuntimeError()), + ).run(("introspect", "lsh")) + artifact = ( + tmp_path / ".tht-jobs" / "dwh" / "runs" / failed.run_id + / "artifacts" / "physical.yaml" + ) + artifact.write_text("tampered") + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _recover_lsh([], output), + ) + with pytest.raises(Exception, match="artifact manifest is invalid"): + pipeline.run(("introspect", "lsh"), resume_run_id=failed.run_id) + + +def test_post_publish_crash_reconciles_same_generation_on_resume(tmp_path): + crashed = False + builder_calls = 0 + + def crash_once(_generation): + nonlocal crashed + if not crashed: + crashed = True + raise KeyboardInterrupt("simulated process death") + + def build(physical, output): + nonlocal builder_calls + builder_calls += 1 + _recover_lsh([], output) + + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=build, + after_publish=crash_once, + ) + with pytest.raises(KeyboardInterrupt): + pipeline.run(("introspect", "lsh")) + active = (tmp_path / ".tht-dwh" / "ACTIVE").read_text().strip() + checkpoint = next((tmp_path / ".tht-jobs" / "dwh" / "runs").glob("*/checkpoint.json")) + source_run_id = checkpoint.parent.name + assert active == source_run_id + + resumed = pipeline.run(("introspect", "lsh"), resume_run_id=source_run_id) + assert resumed.status == "succeeded" + assert builder_calls == 1 + assert (tmp_path / ".tht-dwh" / "ACTIVE").read_text().strip() == source_run_id + + +def test_resume_of_succeeded_run_detects_tampered_published_file(tmp_path): + pipeline = DwhPreprocessPipeline( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint=FP, input_fingerprint=FP, + introspect=lambda output: output.write_text("catalog"), + build_lsh=lambda physical, output: _recover_lsh([], output), + ) + succeeded = pipeline.run(("introspect", "lsh")) + published = ( + tmp_path / ".tht-dwh" / "generations" / succeeded.run_id / "demo_meta.json" + ) + published.chmod(0o600) + published.write_text("tampered") + + with pytest.raises(Exception, match="published DWH"): + pipeline.run(("introspect", "lsh"), resume_run_id=succeeded.run_id) diff --git a/harness/tests/test_memory_metadata.py b/harness/tests/test_memory_metadata.py index 673bb488..4e5ed79d 100644 --- a/harness/tests/test_memory_metadata.py +++ b/harness/tests/test_memory_metadata.py @@ -56,12 +56,12 @@ def test_save_one_memory_preserves_subject_through_upsert_row(): from tht.memory import save_one_memory writer = MagicMock() - writer.upsert_records.return_value = 1 + writer.upsert.return_value = 1 embedder = MagicMock() embedder.embed_documents.return_value = [[0.0] * 8] - save_one_memory([_record(decision_seq=1)], decision_seq=1, writer=writer, embedder=embedder) - row = writer.upsert_records.call_args[0][1][0] - md = row["metadata"] + save_one_memory([_record(decision_seq=1)], decision_seq=1, store=writer, embedder=embedder) + row = writer.upsert.call_args[0][1][0] + md = row.record.metadata assert md["subject"] == "dim_pazienti" assert md["detail"] == "promossa" assert md["rationale"] == "perche' serve" diff --git a/harness/tests/test_memory_save_one.py b/harness/tests/test_memory_save_one.py index e686cc6a..7a679668 100644 --- a/harness/tests/test_memory_save_one.py +++ b/harness/tests/test_memory_save_one.py @@ -8,11 +8,8 @@ pgvector as a one-row upsert. This test pins the pure core of that behavior: - writer.sync is NEVER called (that is the full-resync path) """ from datetime import datetime -from pathlib import Path from unittest.mock import MagicMock -import pytest - from tht.memory import MemoryRecord, memory_vector_record_for_decision, save_one_memory @@ -47,31 +44,31 @@ def test_returns_none_for_unknown_decision_seq(): def test_save_one_calls_upsert_with_single_row_never_sync(): records = [_record(seq=7)] writer = MagicMock() - writer.upsert_records.return_value = 1 + writer.upsert.return_value = 1 embedder = MagicMock() embedder.embed_documents.return_value = [[0.1] * 8] - upserted = save_one_memory(records, decision_seq=7, writer=writer, embedder=embedder) + upserted = save_one_memory(records, decision_seq=7, store=writer, embedder=embedder) assert upserted == 1 writer.sync.assert_not_called() # the whole point of D11: no full resync - writer.upsert_records.assert_called_once() - args = writer.upsert_records.call_args + writer.upsert.assert_called_once() + args = writer.upsert.call_args # table is memory, exactly one row assert args[0][0] == "memory" rows = args[0][1] assert len(rows) == 1 - assert rows[0]["record_key"] == "memory:mem-0007" - assert "embedding" in rows[0] + assert rows[0].record.id == "memory:mem-0007" + assert rows[0].embedding def test_save_one_no_record_for_seq_is_noop(): records = [_record(seq=7)] writer = MagicMock() embedder = MagicMock() - upserted = save_one_memory(records, decision_seq=42, writer=writer, embedder=embedder) + upserted = save_one_memory(records, decision_seq=42, store=writer, embedder=embedder) assert upserted == 0 - writer.upsert_records.assert_not_called() + writer.upsert.assert_not_called() writer.sync.assert_not_called() embedder.embed_documents.assert_not_called() @@ -81,9 +78,9 @@ def test_save_one_uses_writer_key_for_upsert(): Verified indirectly: save_one_memory takes the writer as its client argument.""" records = [_record(seq=7)] writer = MagicMock() - writer.upsert_records.return_value = 1 + writer.upsert.return_value = 1 embedder = MagicMock() embedder.embed_documents.return_value = [[0.0] * 4] - save_one_memory(records, decision_seq=7, writer=writer, embedder=embedder) + save_one_memory(records, decision_seq=7, store=writer, embedder=embedder) # one upsert call, single row, table=memory - assert writer.upsert_records.call_count == 1 + assert writer.upsert.call_count == 1 diff --git a/harness/tests/test_portable_paths.py b/harness/tests/test_portable_paths.py new file mode 100644 index 00000000..2f60763c --- /dev/null +++ b/harness/tests/test_portable_paths.py @@ -0,0 +1,110 @@ +from pathlib import Path + +import pytest + +from tht.config import ConfigError, load_config +from tht.paths import resolve_workspace_paths + + +def _write_config(path: Path, *, sessions: str = "sessions", absolute: Path | None = None) -> Path: + root = absolute or Path("artifacts") + path.write_text( + "dwh:\n" + " type: postgres_direct\n" + " connection:\n" + " database: db\n" + " schema: public\n" + " user: user\n" + " password: secret\n" + "roots:\n" + f" artifacts: {root}\n" + f" indexes: {root if absolute else 'indexes'}\n" + f" sessions: {absolute if absolute else sessions}\n" + ) + return path + + +def test_relative_paths_resolve_under_workspace_root(tmp_path): + cfg_path = _write_config(tmp_path / "demo.yaml") + cfg = load_config(cfg_path) + + resolved = resolve_workspace_paths(cfg_path, cfg, tmp_path / "data") + + assert resolved.workspace == tmp_path / "data/workspaces/demo" + assert resolved.sessions == tmp_path / "data/workspaces/demo/sessions" + assert resolved.artifacts == tmp_path / "data/workspaces/demo/artifacts" + assert resolved.indexes == tmp_path / "data/workspaces/demo/indexes" + assert resolved.corpus == tmp_path / "data/workspaces/demo/corpus" + + +def test_absolute_legacy_paths_are_preserved(tmp_path): + legacy = tmp_path / "existing-workspace" + cfg_path = _write_config(tmp_path / "demo.yaml", absolute=legacy) + cfg = load_config(cfg_path) + + resolved = resolve_workspace_paths(cfg_path, cfg, tmp_path / "data") + + assert resolved.sessions == legacy + assert resolved.artifacts == legacy + assert resolved.indexes == legacy + + +def test_path_escape_is_rejected(tmp_path): + cfg_path = _write_config(tmp_path / "demo.yaml", sessions="../../private") + cfg = load_config(cfg_path) + + with pytest.raises(ConfigError, match="outside workspace root"): + resolve_workspace_paths(cfg_path, cfg, tmp_path / "data") + + +def test_workspace_symlink_escape_is_rejected(tmp_path): + cfg_path = _write_config(tmp_path / "demo.yaml") + cfg = load_config(cfg_path) + data_root = tmp_path / "data" + (data_root / "workspaces").mkdir(parents=True) + (data_root / "workspaces" / "demo").symlink_to(tmp_path / "private", target_is_directory=True) + + with pytest.raises(ConfigError, match="outside workspaces root"): + resolve_workspace_paths(cfg_path, cfg, data_root) + + +def test_nested_root_symlink_escape_is_rejected(tmp_path): + cfg_path = _write_config(tmp_path / "demo.yaml") + cfg = load_config(cfg_path) + workspace = tmp_path / "data/workspaces/demo" + workspace.mkdir(parents=True) + (workspace / "sessions").symlink_to(tmp_path / "private", target_is_directory=True) + + with pytest.raises(ConfigError, match="outside workspace root"): + resolve_workspace_paths(cfg_path, cfg, tmp_path / "data") + + +def test_corpus_symlink_escape_is_rejected(tmp_path): + cfg_path = _write_config(tmp_path / "demo.yaml") + cfg = load_config(cfg_path) + workspace = tmp_path / "data/workspaces/demo" + workspace.mkdir(parents=True) + (workspace / "corpus").symlink_to(tmp_path / "private", target_is_directory=True) + + with pytest.raises(ConfigError, match="outside workspace root"): + resolve_workspace_paths(cfg_path, cfg, tmp_path / "data") + + +def test_data_root_environment_activates_portable_paths(monkeypatch, tmp_path): + cfg_path = _write_config(tmp_path / "demo.yaml") + monkeypatch.setenv("THT_DATA_ROOT", str(tmp_path / "data")) + + cfg = load_config(cfg_path) + + assert cfg.paths.sessions == tmp_path / "data/workspaces/demo/sessions" + assert cfg.paths.artifacts == tmp_path / "data/workspaces/demo/artifacts" + + +def test_no_data_root_preserves_legacy_relative_paths(monkeypatch, tmp_path): + cfg_path = _write_config(tmp_path / "demo.yaml") + monkeypatch.delenv("THT_DATA_ROOT", raising=False) + + cfg = load_config(cfg_path) + + assert cfg.paths.sessions == Path("sessions") + assert cfg.paths.artifacts == Path("artifacts") diff --git a/harness/tests/test_preprocess_cli.py b/harness/tests/test_preprocess_cli.py new file mode 100644 index 00000000..8040865e --- /dev/null +++ b/harness/tests/test_preprocess_cli.py @@ -0,0 +1,131 @@ +import json +from types import SimpleNamespace + +from typer.testing import CliRunner + +from tht.cli import app + + +def test_preprocess_evidence_json_is_pristine(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + result = SimpleNamespace(model_dump=lambda mode=None: { + "status": "succeeded", "generation": "gen:abc", "published": True + }) + monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) + response = CliRunner().invoke( + app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")] + ) + assert response.exit_code == 0, response.output + assert json.loads(response.output)["generation"] == "gen:abc" + + +def test_preprocess_failure_is_structured_and_nonzero(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + monkeypatch.setattr(command, "run_from_config", lambda *a, **k: (_ for _ in ()).throw(RuntimeError("secret detail"))) + response = CliRunner().invoke( + app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")] + ) + assert response.exit_code != 0 + assert json.loads(response.output) == {"status": "failed", "error": "preprocessing failed"} + assert "secret detail" not in response.output + + +def test_preprocess_failed_job_report_is_sanitized_json_and_nonzero(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + result = SimpleNamespace(model_dump=lambda mode=None: { + "status": "failed", "run_id": "a" * 32, "published": False, + "generation": "gen:" + "b" * 32, "changed": ["fs:one"], + }) + monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) + response = CliRunner().invoke( + app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")] + ) + assert response.exit_code == 1 + payload = json.loads(response.output) + assert payload["status"] == "failed" + assert payload["error"] == "preprocessing job failed" + assert "traceback" not in response.output.lower() + + +def test_preprocess_real_failed_stage_result_exits_nonzero(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + from test_corpus_pipeline import Source, item, pipeline + + result = pipeline( + tmp_path, Source([(item("one", "a"), RuntimeError("SENSITIVE EVIDENCE secret"))]) + ).run_as_job( + workspace_id="demo", workspace_root=tmp_path, + config_fingerprint="sha256:" + "1" * 64, + input_fingerprint="sha256:" + "2" * 64, + ) + assert result.status == "failed" + monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) + response = CliRunner().invoke( + app, ["preprocess", "evidence", "--json", "-c", str(tmp_path / "workspace.yaml")] + ) + assert response.exit_code == 1 + assert json.loads(response.output)["status"] == "failed" + assert "SENSITIVE EVIDENCE" not in response.output + assert "secret" not in response.output + + +def test_preprocess_evidence_text_uses_uncapped_result_counts(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + result = SimpleNamespace(model_dump=lambda mode=None: { + "status": "succeeded", "run_id": "a" * 32, + "generation": "gen:" + "b" * 64, "published": True, + "changed": ["fs:item"] * 100, + "unchanged": ["fs:item"] * 100, + "removed": ["fs:item"] * 100, + "counts": {"changed": 1001, "unchanged": 902, "removed": 803}, + }) + monkeypatch.setattr(command, "run_from_config", lambda *args, **kwargs: result) + + response = CliRunner().invoke( + app, ["preprocess", "evidence", "-c", str(tmp_path / "workspace.yaml")] + ) + + assert response.exit_code == 0, response.output + assert "changed=1001 unchanged=902 removed=803" in response.output + + +def test_preprocess_resume_rejects_generation_id_before_configuration(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + called = False + + def forbidden(*args, **kwargs): + nonlocal called + called = True + + monkeypatch.setattr(command, "run_from_config", forbidden) + response = CliRunner().invoke( + app, + [ + "preprocess", "evidence", "--resume", "gen:" + "a" * 32, + "--json", "-c", str(tmp_path / "workspace.yaml"), + ], + ) + assert response.exit_code != 0 + assert json.loads(response.output) == { + "status": "failed", "error": "resume requires a preprocessing run id" + } + assert called is False + + +def test_preprocess_evidence_gc_json_is_pristine(monkeypatch, tmp_path): + import tht.cli.preprocess_cmd as command + + monkeypatch.setattr(command, "gc_from_config", lambda *a, **k: { + "status": "succeeded", "dry_run": True, "evicted": [], "failures": [], + }) + response = CliRunner().invoke( + app, ["preprocess", "evidence", "gc", "--dry-run", "--json", "-c", + str(tmp_path / "workspace.yaml")] + ) + assert response.exit_code == 0, response.output + assert json.loads(response.output)["dry_run"] is True diff --git a/harness/tests/test_s3_evidence_source.py b/harness/tests/test_s3_evidence_source.py new file mode 100644 index 00000000..710b44fc --- /dev/null +++ b/harness/tests/test_s3_evidence_source.py @@ -0,0 +1,199 @@ +from datetime import UTC, datetime + +import pytest + +from tht.ports.evidence import EvidenceSourceError + + +class Body: + def __init__(self, data): self.data, self.closed = data, False + def read(self, amount): return self.data[:amount] + def close(self): self.closed = True + + +class Client: + def __init__(self): self.body, self.list_calls = Body(b"hello"), 0 + def get_paginator(self, name): return self + def list_objects_v2(self, **kwargs): + self.list_calls += 1 + return {"Contents": [{"Key": "clinical/a.md", "ETag": '"abc"', + "Size": 5, "LastModified": datetime(2026, 1, 1, tzinfo=UTC)}]} + def paginate(self, **kwargs): + yield {"Contents": [{"Key": "clinical/a.md", "ETag": '"abc"', + "Size": 5, "LastModified": datetime(2026, 1, 1, tzinfo=UTC)}]} + def get_object(self, **kwargs): + assert kwargs == {"Bucket": "evidence", "Key": "clinical/a.md"} + return {"Body": self.body, "ContentLength": 5, "ContentType": "text/markdown", + "ETag": '"abc"'} + + +def test_s3_canonical_uri_version_fingerprint_and_closed_body(): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + source = S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client) + item = next(iter(source.discover())) + assert item.uri == "s3://evidence/clinical/a.md" + assert item.fingerprint.startswith("etag:") + assert source.acquire(item).content == b"hello" + assert client.body.closed + + +def test_s3_etag_fallback_and_bounds(): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + with pytest.raises(ValueError): + S3EvidenceSource(bucket="evidence", client=client, max_objects=0) + + +def test_s3_rejects_private_or_insecure_endpoint_without_explicit_opt_in(): + from tht.adapters.evidence.s3 import S3EvidenceSource + with pytest.raises(ValueError, match="trusted"): + S3EvidenceSource(bucket="evidence", endpoint_url="https://127.0.0.1:9000", client=Client()) + with pytest.raises(ValueError, match="HTTPS"): + S3EvidenceSource(bucket="evidence", endpoint_url="http://s3.example.test", client=Client()) + source = S3EvidenceSource(bucket="evidence", endpoint_url="http://127.0.0.1:9000", + trusted_endpoint=True, allow_private_endpoint=True, + allow_insecure_endpoint=True, client=Client()) + assert source is not None + + +@pytest.mark.parametrize("bucket", ["UPPER", "bad_bucket", "-start", "end-", "a..b"]) +def test_s3_rejects_invalid_bucket_names(bucket): + from tht.adapters.evidence.s3 import S3EvidenceSource + with pytest.raises(ValueError, match="bucket"): + S3EvidenceSource(bucket=bucket, client=Client()) + + +@pytest.mark.parametrize("bucket", ["127.0.0.1", "192.168.1.1"]) +def test_s3_rejects_ip_shaped_bucket(bucket): + from tht.adapters.evidence.s3 import S3EvidenceSource + with pytest.raises(ValueError, match="bucket"): + S3EvidenceSource(bucket=bucket, client=Client()) + + +def test_s3_rejects_endpoint_query_path_fragment_and_untrusted_custom_host(): + from tht.adapters.evidence.s3 import S3EvidenceSource + for endpoint in ("https://s3.example.test/path", "https://s3.example.test/?x=1", + "https://s3.example.test/#x"): + with pytest.raises(ValueError, match="root"): + S3EvidenceSource(bucket="evidence", endpoint_url=endpoint, + trusted_endpoint=True, client=Client()) + with pytest.raises(ValueError, match="trusted"): + S3EvidenceSource(bucket="evidence", endpoint_url="https://s3.example.test", client=Client()) + + +def test_s3_rejects_out_of_prefix_key_and_missing_validator(): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + client.list_objects_v2 = lambda **kwargs: {"Contents": [{"Key": "other/a.md", "ETag": '"x"'}]} + with pytest.raises(EvidenceSourceError): + list(S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client).discover()) + client.list_objects_v2 = lambda **kwargs: {"Contents": [{"Key": "clinical/a.md"}]} + with pytest.raises(EvidenceSourceError): + list(S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client).discover()) + + +def test_s3_rejects_leading_slash_prefix_empty_and_control_keys(): + from tht.adapters.evidence.s3 import S3EvidenceSource + with pytest.raises(ValueError, match="prefix"): + S3EvidenceSource(bucket="evidence", prefix="/clinical", client=Client()) + for key in ("", "clinical/a\x00.md", "clinical/a\x7f.md"): + client = Client() + client.list_objects_v2 = lambda **kwargs: {"Contents": [{"Key": key, "ETag": '"x"'}]} + with pytest.raises(EvidenceSourceError): + list(S3EvidenceSource(bucket="evidence", prefix="clinical/", client=client).discover()) + + +@pytest.mark.parametrize("prefix", ["/bad", "x" * 1025, "bad\x00prefix", "bad\x7fprefix"]) +def test_s3_rejects_invalid_prefix_before_client_request(prefix): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + with pytest.raises(ValueError, match="prefix"): + S3EvidenceSource(bucket="evidence", prefix=prefix, client=client) + assert client.list_calls == 0 + + +def test_s3_hard_page_limit_never_requests_page_max_plus_one(): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + def listing(**kwargs): + client.list_calls += 1 + return {"Contents": [{"Key": f"clinical/{client.list_calls}.md", "ETag": '"x"'}], + "IsTruncated": True, "NextContinuationToken": str(client.list_calls)} + client.list_objects_v2 = listing + with pytest.raises(EvidenceSourceError): + list(S3EvidenceSource(bucket="evidence", prefix="clinical/", max_pages=2, + client=client).discover()) + assert client.list_calls == 2 + + +def test_s3_acquire_rejects_exact_etag_drift_and_closes_body(): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + source = S3EvidenceSource(bucket="evidence", client=client) + item = next(iter(source.discover())) + client.get_object = lambda **kwargs: {"Body": client.body, "ContentLength": 5, + "ETag": '"changed"'} + with pytest.raises(EvidenceSourceError): + source.acquire(item) + assert client.body.closed + + +def test_s3_acquire_rejects_forged_reconstructed_item_before_get(): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + source = S3EvidenceSource(bucket="evidence", client=client) + item = next(iter(source.discover())) + forged = item.model_copy(update={"fingerprint": "etag:" + "0" * 64}) + client.get_object = lambda **kwargs: (_ for _ in ()).throw(AssertionError("called")) + with pytest.raises(EvidenceSourceError): + source.acquire(forged) + + +@pytest.mark.parametrize("host", ["127.0.0.1", "10.0.0.1", "169.254.1.1", "0.0.0.0", + "[::1]", "[fe80::1]", "[::]"]) +def test_s3_literal_non_global_endpoint_requires_private_opt_in(host): + from tht.adapters.evidence.s3 import S3EvidenceSource + with pytest.raises(ValueError, match="private"): + S3EvidenceSource(bucket="evidence", endpoint_url=f"https://{host}:9000", + trusted_endpoint=True, client=Client()) + + +def test_s3_size_limit_closes_body(): + from tht.adapters.evidence.s3 import S3EvidenceSource + client = Client() + source = S3EvidenceSource(bucket="evidence", client=client, max_bytes=4) + item = next(iter(source.discover())) + with pytest.raises(EvidenceSourceError): + source.acquire(item) + assert client.body.closed + + +def test_s3_config_serialization_masks_credentials(): + from tht.config import S3EvidenceSourceConfig + config = S3EvidenceSourceConfig(type="s3", bucket="evidence", + access_key="access-secret", secret_key="write-secret") + assert "access-secret" not in repr(config) + assert "write-secret" not in repr(config) + + +def test_s3_config_loads_credentials_from_secret_files(tmp_path): + from tht.config import load_config + access, secret = tmp_path / "access", tmp_path / "secret" + access.write_text("access-value") + secret.write_text("secret-value") + workspace = tmp_path / "workspace.yaml" + workspace.write_text(f""" +dwh: + type: postgres_direct + connection: {{database: d, schema: public, user: u, password: p}} +evidence: + sources: + - type: s3 + bucket: evidence + access_key_file: {access} + secret_key_file: {secret} +""") + source = load_config(workspace).evidence.sources[0] + assert source.access_key.get_secret_value() == "access-value" + assert source.secret_key.get_secret_value() == "secret-value" diff --git a/harness/tests/test_schema_introspect_guard.py b/harness/tests/test_schema_introspect_guard.py index 056e361f..5a93bf43 100644 --- a/harness/tests/test_schema_introspect_guard.py +++ b/harness/tests/test_schema_introspect_guard.py @@ -4,6 +4,8 @@ from typer.testing import CliRunner from tht.cli import app from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical +from tht.config import ExamplesConfig +from tht.cli.schema_cmd import _add_examples def _write_catalog(tmp_path): @@ -32,17 +34,62 @@ def _write_config(tmp_path): return cfg -def test_introspect_cache_hit_skips_dwh(tmp_path): - # Le credenziali sono fasulle: se la guardia non scattasse PRIMA del branch - # transport, il comando tenterebbe la connessione e fallirebbe. +def test_introspect_rejects_unbound_legacy_cache(tmp_path): catalog = _write_catalog(tmp_path) - before = catalog.read_bytes() cfg = _write_config(tmp_path) res = CliRunner().invoke(app, ["schema", "introspect", "-c", str(cfg)]) + assert res.exit_code == 1 + assert "legacy artifacts are unbound" in res.output + assert catalog.exists() + + +def test_introspect_fresh_root_initializes_through_writer_job(tmp_path, monkeypatch): + import tht.cli.schema_cmd as module + + cfg = _write_config(tmp_path) + physical = PhysicalSchema( + database="d", schema="s", introspected_at=datetime(2026, 1, 1), + tables={"dim_patient": TablePhysical(columns={"id": ColumnPhysical(type="bigint")})}, + ) + + def refresh(_cfg, *, output_path=None, **_kwargs): + physical.to_yaml(output_path) + return physical + + monkeypatch.setattr(module, "refresh_catalog", refresh) + res = CliRunner().invoke(app, ["schema", "introspect", "-c", str(cfg)]) assert res.exit_code == 0, res.output - assert "OK (cache)" in res.output + assert (tmp_path / ".tht-dwh" / "OWNER.json").is_file() assert "1 tabelle" in res.output - assert catalog.read_bytes() == before + + +def test_lsh_build_fresh_root_initializes_introspection_and_lsh(tmp_path, monkeypatch): + import tht.cli.lsh_cmd as lsh_module + import tht.cli.schema_cmd as schema_module + import tht.lshindex as lshindex_module + + cfg = _write_config(tmp_path) + physical = PhysicalSchema( + database="d", schema="s", introspected_at=datetime(2026, 1, 1), + tables={"dim_patient": TablePhysical(columns={"id": ColumnPhysical(type="bigint")})}, + ) + + def refresh(_cfg, *, output_path=None, **_kwargs): + physical.to_yaml(output_path) + return physical + + def build(_cfg, *, physical_file, output_dir, **_kwargs): + assert physical_file.is_file() + for name in ("s_lsh.pkl", "s_minhashes.pkl", "s_meta.json"): + (output_dir / name).write_text("index") + return {}, [], [], {} + + monkeypatch.setattr(schema_module, "refresh_catalog", refresh) + monkeypatch.setattr(lsh_module, "build_lsh_artifacts", build) + monkeypatch.setattr(lshindex_module, "load_index", lambda *_args, **_kwargs: (None, {}, None)) + res = CliRunner().invoke(app, ["lsh", "build", "-c", str(cfg)]) + assert res.exit_code == 0, res.output + assert (tmp_path / ".tht-dwh" / "OWNER.json").is_file() def test_introspect_refresh_bypasses_cache(tmp_path): @@ -68,3 +115,23 @@ def test_render_without_catalog_guides_fallback(tmp_path): res = CliRunner().invoke(app, ["schema", "render", "-c", str(cfg)]) assert res.exit_code == 1 assert "Esegui prima" in res.output + + +def test_examples_skip_one_unreadable_column_and_continue(caplog): + physical = PhysicalSchema( + database="d", schema="s", introspected_at=datetime(2026, 1, 1), + tables={"t": TablePhysical(columns={ + "bad": ColumnPhysical(type="text"), "good": ColumnPhysical(type="text") + })}, + ) + + class Dwh: + def sample_column(self, table, column, *, limit): + if column == "bad": + raise RuntimeError("denied") + return ["kept"] + + _add_examples(Dwh(), physical, ExamplesConfig(max_per_column=3)) + assert physical.tables["t"].columns["bad"].examples == [] + assert physical.tables["t"].columns["good"].examples == ["kept"] + assert "Campionamento saltato" in caplog.text diff --git a/harness/tests/test_search_pack.py b/harness/tests/test_search_pack.py index 5f246219..edcfc199 100644 --- a/harness/tests/test_search_pack.py +++ b/harness/tests/test_search_pack.py @@ -5,6 +5,8 @@ from types import SimpleNamespace from typer.testing import CliRunner from tht.cli import app +from tht.config import load_config +from tht.jobs.dwh_pipeline import DwhPreprocessPipeline, config_dwh_binding from tht.mschema.models import ColumnPhysical, PhysicalSchema, TablePhysical from tht.vectorstore.embeddings import EmbeddingsError @@ -43,11 +45,11 @@ class _FakeSearcher: def _workspace(tmp_path, with_session=None): - PhysicalSchema( + physical = PhysicalSchema( database="d", schema="s", introspected_at=datetime(2026, 1, 1), tables={"fact_ablazione": TablePhysical( comment="Ablazioni", columns={"cod_paz": ColumnPhysical(type="bigint")})}, - ).to_yaml(tmp_path / "artifacts" / "mschema" / "physical.yaml") + ) cfg = tmp_path / "workspace.yaml" cfg.write_text( "database: {database: d, schema: s, user: u, password: p, transport: direct}\n" @@ -56,6 +58,18 @@ def _workspace(tmp_path, with_session=None): f"paths: {{artifacts: {tmp_path/'artifacts'}, indexes: {tmp_path/'i'}, " f"sessions: {tmp_path/'sessions'}}}\n" ) + binding = config_dwh_binding(load_config(cfg)) + DwhPreprocessPipeline( + workspace_id=binding["workspace_id"], workspace_root=tmp_path, + config_fingerprint=binding["config_fingerprint"], + input_fingerprint=binding["input_fingerprint"], + introspect=lambda output: physical.to_yaml(output), + build_lsh=lambda _physical, output: [ + (output / name).write_text("index") + for name in ("s_lsh.pkl", "s_minhashes.pkl", "s_meta.json") + ], + lsh_filenames=("s_lsh.pkl", "s_minhashes.pkl", "s_meta.json"), + ).run() if with_session: sdir = tmp_path / "sessions" / with_session sdir.mkdir(parents=True) @@ -81,7 +95,9 @@ def test_pack_single_embed_and_sections(tmp_path, monkeypatch): assert res.exit_code == 0, res.output assert emb.calls == 1 # UN solo embedding per le tre ricerche assert "fact_ablazione" in res.output and "Ablazioni" in res.output - assert "Dominio ablazione" in res.output + # Evidence is fail-closed until an ACTIVE corpus exists; legacy vector rows + # must not leak into a new search pack. + assert "Dominio ablazione" not in res.output assert "SELECT 1" in res.output diff --git a/harness/tests/test_search_similar_kinds.py b/harness/tests/test_search_similar_kinds.py index 5a16be0a..049d9519 100644 --- a/harness/tests/test_search_similar_kinds.py +++ b/harness/tests/test_search_similar_kinds.py @@ -90,6 +90,25 @@ def test_search_similar_reraises_non_404_with_kinds(monkeypatch): _client().search_similar("memory", [0.1] * 4, 5, kinds=["memory"]) +def test_generation_filter_is_sent_exactly_and_legacy_404_fails_closed(monkeypatch): + calls = [] + + def fake_call(self, function, payload): + calls.append(payload) + raise VectorRestError("HTTP 404 missing filtered RPC") + + monkeypatch.setattr(VectorRestClient, "_call", fake_call) + metadata_filter = {"vector_generation": "gen:abc", "document_ids": ["doc:1"]} + with pytest.raises(VectorRestError, match="404"): + _client().search_similar( + "evidence", [0.1] * 4, 5, kinds=["evidence"], metadata_filter=metadata_filter + ) + assert calls == [{ + "query_embedding": [0.1] * 4, "limit_count": 5, "table_name": "evidence", + "kinds": ["evidence"], "metadata_filter": metadata_filter, + }] + + def test_rest_searcher_forwards_kinds_to_client(): calls = [] diff --git a/harness/tests/test_solved_question.py b/harness/tests/test_solved_question.py index 54c4fb62..91c9e5d0 100644 --- a/harness/tests/test_solved_question.py +++ b/harness/tests/test_solved_question.py @@ -43,18 +43,18 @@ def test_solved_kind_maps_to_memory_table(): def test_save_upserts_single_row_into_memory_table(): writer = MagicMock() writer.existing_hashes.return_value = {} - writer.upsert_records.return_value = 1 + writer.upsert.return_value = 1 embedder = MagicMock() embedder.embed_documents.return_value = [[0.1] * 8] - assert save_solved_question(_rec(), writer=writer, embedder=embedder) == 1 + assert save_solved_question(_rec(), store=writer, embedder=embedder) == 1 writer.sync.assert_not_called() - table, rows = writer.upsert_records.call_args[0] + table, rows = writer.upsert.call_args[0] assert table == "memory" assert len(rows) == 1 - assert rows[0]["record_key"] == "solved:s1" - assert rows[0]["metadata"]["kind"] == SOLVED_KIND - assert rows[0]["metadata"]["sql"].startswith("SELECT") + assert rows[0].record.id == "solved:s1" + assert rows[0].record.kind == SOLVED_KIND + assert rows[0].record.metadata["sql"].startswith("SELECT") def test_save_skips_when_question_and_sql_unchanged(): @@ -62,9 +62,9 @@ def test_save_skips_when_question_and_sql_unchanged(): writer = MagicMock() writer.existing_hashes.return_value = {r.id: _solved_hash(r)} embedder = MagicMock() - assert save_solved_question(r, writer=writer, embedder=embedder) == 0 + assert save_solved_question(r, store=writer, embedder=embedder) == 0 embedder.embed_documents.assert_not_called() - writer.upsert_records.assert_not_called() + writer.upsert.assert_not_called() def test_sql_change_alone_triggers_reupsert(): @@ -72,7 +72,7 @@ def test_sql_change_alone_triggers_reupsert(): new = _rec(sql="SELECT 1") # stessa domanda, SQL diverso writer = MagicMock() writer.existing_hashes.return_value = {old.id: _solved_hash(old)} - writer.upsert_records.return_value = 1 + writer.upsert.return_value = 1 embedder = MagicMock() embedder.embed_documents.return_value = [[0.0] * 4] - assert save_solved_question(new, writer=writer, embedder=embedder) == 1 + assert save_solved_question(new, store=writer, embedder=embedder) == 1 diff --git a/harness/tests/test_vector_migration_packaging.py b/harness/tests/test_vector_migration_packaging.py new file mode 100644 index 00000000..24544c97 --- /dev/null +++ b/harness/tests/test_vector_migration_packaging.py @@ -0,0 +1,59 @@ +import os +import shutil +import subprocess +import sys +import zipfile +from pathlib import Path + + +def test_built_wheel_installs_vector_migrations_and_discovers_cli(tmp_path): + harness = Path(__file__).parents[1] + wheelhouse = tmp_path / "wheelhouse" + target = tmp_path / "site" + wheelhouse.mkdir() + uv = shutil.which("uv") + assert uv is not None, "uv is required to verify the production wheel" + build_env = {**os.environ, "UV_CACHE_DIR": str(tmp_path / "uv-cache")} + subprocess.run( + [ + uv, + "build", + "--wheel", + "--out-dir", + str(wheelhouse), + str(harness), + ], + check=True, + capture_output=True, + text=True, + env=build_env, + ) + wheel = next(wheelhouse.glob("tht-*.whl")) + with zipfile.ZipFile(wheel) as archive: + names = set(archive.namelist()) + assert "tht/migrations/vector/001_extensions.sql" in names + assert "tht/migrations/vector/003_roles.sql" in names + + subprocess.run( + [sys.executable, "-m", "pip", "install", "--no-deps", "--target", str(target), wheel], + check=True, + capture_output=True, + text=True, + ) + env = {**os.environ, "PYTHONPATH": str(target)} + probe = subprocess.run( + [ + sys.executable, + "-c", + "from typer.testing import CliRunner; from tht.cli import app; " + "r=CliRunner().invoke(app, ['vector','migrate','--help']); " + "print(r.output); raise SystemExit(r.exit_code)", + ], + env=env, + check=False, + capture_output=True, + text=True, + cwd=tmp_path, + ) + assert probe.returncode == 0, probe.stderr + probe.stdout + assert "--status" in probe.stdout diff --git a/harness/tests/test_vector_port_contract.py b/harness/tests/test_vector_port_contract.py new file mode 100644 index 00000000..db97831d --- /dev/null +++ b/harness/tests/test_vector_port_contract.py @@ -0,0 +1,250 @@ +from dataclasses import FrozenInstanceError +from unittest.mock import MagicMock + +import pytest + +from tht.adapters.vector.thoth_http import ThothHttpVectorStore +from tht.adapters.vector.legacy_direct import LegacyDirectVectorStore +from tht.evidence.model import EvidenceDoc +from tht.ports.vector import ( + VectorHit, + VectorRecord, + VectorStore, + VectorReadUnavailable, + VectorWriteRecord, + VectorWriteUnavailable, +) +from tht.vectorstore.records import evidence_records + + +def test_http_store_reports_reader_without_writer(): + reader = MagicMock() + store = ThothHttpVectorStore(reader=reader, writer=None) + + assert store.capabilities.search is True + assert store.capabilities.upsert is False + with pytest.raises(VectorWriteUnavailable): + store.upsert("memory", []) + + +def test_http_store_supports_writer_without_reader(): + writer = MagicMock() + store = ThothHttpVectorStore(reader=None, writer=writer, expected_dimension=768) + + assert store.capabilities.search is False + assert store.capabilities.existing_hashes is True + assert store.capabilities.upsert is True + with pytest.raises(VectorReadUnavailable): + store.search(["memory"], [0.1], limit=1) + + +@pytest.mark.parametrize("limit", [True, False, 1.0, 0, -1]) +def test_http_search_requires_a_strict_positive_integer_limit(limit): + store = ThothHttpVectorStore(reader=MagicMock(), writer=None) + + with pytest.raises(ValueError, match="positive integer"): + store.search(["memory"], [0.1], limit=limit) + + +def test_http_store_keeps_reader_and_writer_operations_separate(): + reader = MagicMock() + reader.search_similar.return_value = [ + { + "similarity": 0.75, + "metadata": { + "record_key": "m1", + "kind": "memory", + "ref": "session:s1", + "title": "Choice", + "content": "Use the curated table", + }, + } + ] + writer = MagicMock() + writer.existing_hashes.return_value = {"m1": "abc"} + writer.upsert_records.return_value = 1 + store = ThothHttpVectorStore(reader=reader, writer=writer) + + hits = store.search(["memory"], [0.1, 0.2], limit=3, kinds=["memory"]) + assert hits == [ + VectorHit( + id="m1", + kind="memory", + ref="session:s1", + title="Choice", + content="Use the curated table", + metadata={ + "record_key": "m1", + "kind": "memory", + "ref": "session:s1", + "title": "Choice", + "content": "Use the curated table", + }, + similarity=0.75, + ) + ] + reader.search_similar.assert_called_once_with( + "memory", [0.1, 0.2], 3, kinds=["memory"] + ) + writer.search_similar.assert_not_called() + + assert store.existing_hashes("memory", ["memory"]) == {"m1": "abc"} + writer.existing_hashes.assert_called_once_with("memory", ["memory"]) + + records = [ + VectorWriteRecord( + record=VectorRecord( + id="m1", + kind="memory", + ref="session:s1", + title="Choice", + content="Use the curated table", + ), + embedding=[0.1, 0.2], + content_hash="abc", + ) + ] + assert store.upsert("memory", records) == 1 + writer.upsert_records.assert_called_once() + reader.upsert_records.assert_not_called() + + +def test_http_upsert_serializes_a_canonical_builder_record(): + record = evidence_records( + [EvidenceDoc(id="joins", title="Join guidance", body="Use the curated join")], + max_chunk_chars=1000, + )[0] + writer = MagicMock() + writer.upsert_records.return_value = 1 + store = ThothHttpVectorStore(reader=MagicMock(), writer=writer) + + assert store.upsert( + "evidence", + [VectorWriteRecord(record=record, embedding=[0.2, 0.3], content_hash="digest")], + ) == 1 + row = writer.upsert_records.call_args.args[1][0] + assert row["record_key"] == "evidence:joins:0" + assert row["metadata"]["status"] == "reviewed" + assert row["embedding"] == [0.2, 0.3] + assert row["content_hash"] == "digest" + + +def test_http_upsert_preserves_metadata_named_like_transport_fields(): + record = VectorRecord( + id="collision", + kind="memory", + ref="session:s1", + title="Collision", + content="Semantic metadata must survive", + metadata={"embedding": "semantic embedding", "content_hash": "semantic hash"}, + ) + writer = MagicMock() + store = ThothHttpVectorStore(reader=MagicMock(), writer=writer) + + store.upsert( + "memory", + [VectorWriteRecord(record=record, embedding=[0.4], content_hash="transport hash")], + ) + row = writer.upsert_records.call_args.args[1][0] + assert row["embedding"] == [0.4] + assert row["content_hash"] == "transport hash" + assert row["metadata"]["embedding"] == "semantic embedding" + assert row["metadata"]["content_hash"] == "semantic hash" + + +def test_http_store_is_runtime_vector_store(): + store = ThothHttpVectorStore(reader=MagicMock(), writer=None) + assert isinstance(store, VectorStore) + + +def test_vector_contract_is_exported_from_public_packages(): + from tht.adapters.vector import ThothHttpVectorStore as PublicHttpStore + from tht.ports import VectorStore as PublicVectorStore + from tht.ports import VectorWriteRecord as PublicVectorWriteRecord + from tht.ports import VectorReadUnavailable as PublicVectorReadUnavailable + + assert PublicHttpStore is ThothHttpVectorStore + assert PublicVectorStore is VectorStore + assert PublicVectorWriteRecord is VectorWriteRecord + assert PublicVectorReadUnavailable is VectorReadUnavailable + + capabilities = store_capabilities = ThothHttpVectorStore( + reader=MagicMock(), writer=None + ).capabilities + assert capabilities.search is True + with pytest.raises(FrozenInstanceError): + store_capabilities.search = False + + +def test_http_health_uses_reader_list_tables_and_reports_failure(): + reader = MagicMock() + store = ThothHttpVectorStore(reader=reader, writer=None) + assert store.health().ok is True + + reader.list_tables.side_effect = RuntimeError("offline") + health = store.health() + assert health.ok is False + assert health.detail == "offline" + + +def test_http_health_reports_read_write_and_dimension_status_independently(): + reader = MagicMock() + reader.list_tables.return_value = [ + {"table_name": "memory", "vector_dimensions": 768} + ] + writer = MagicMock() + writer.list_tables.return_value = [ + {"table_name": "memory", "vector_dimensions": 768} + ] + store = ThothHttpVectorStore(reader, writer, expected_dimension=768) + + health = store.health() + assert health.ok is True + assert health.read_configured is True + assert health.read_reachable is True + assert health.write_configured is True + assert health.write_reachable is True + assert health.expected_dimension == 768 + assert health.observed_dimensions == (768,) + assert health.dimension_compatible is True + + +def test_http_health_does_not_hide_writer_failure_behind_reader_success(): + reader = MagicMock() + reader.list_tables.return_value = [] + writer = MagicMock() + writer.list_tables.side_effect = RuntimeError("writer offline") + store = ThothHttpVectorStore(reader, writer, expected_dimension=768) + + health = store.health() + assert health.ok is False + assert health.read_reachable is True + assert health.write_reachable is False + assert health.write_detail == "writer offline" + assert health.dimension_compatible is None + + +def test_http_health_covers_read_only_and_write_only_configuration(): + reader = MagicMock() + reader.list_tables.return_value = [{"vector_dimensions": 384}] + read_health = ThothHttpVectorStore(reader, None, expected_dimension=768).health() + assert read_health.ok is False + assert read_health.write_configured is False + assert read_health.write_reachable is None + assert read_health.dimension_compatible is False + + writer = MagicMock() + writer.list_tables.return_value = [{"vector_dimensions": 768}] + write_health = ThothHttpVectorStore(None, writer, expected_dimension=768).health() + assert write_health.ok is True + assert write_health.read_configured is False + assert write_health.read_reachable is None + assert write_health.dimension_compatible is True + + +@pytest.mark.parametrize("limit", [True, False, 1.0, 0, -1]) +def test_legacy_direct_search_requires_a_strict_positive_integer_limit(limit): + store = LegacyDirectVectorStore(engine=MagicMock()) + + with pytest.raises(ValueError, match="positive integer"): + store.search(["memory"], [0.1], limit=limit) diff --git a/harness/tht/adapters/dwh/__init__.py b/harness/tht/adapters/dwh/__init__.py new file mode 100644 index 00000000..87e3eacc --- /dev/null +++ b/harness/tht/adapters/dwh/__init__.py @@ -0,0 +1,6 @@ +"""Data-warehouse adapter implementations.""" + +from tht.adapters.dwh.postgres import PostgresDwhAdapter +from tht.adapters.dwh.thoth_rest import ThothRestDwhAdapter + +__all__ = ["PostgresDwhAdapter", "ThothRestDwhAdapter"] diff --git a/harness/tht/adapters/dwh/postgres.py b/harness/tht/adapters/dwh/postgres.py new file mode 100644 index 00000000..a926005e --- /dev/null +++ b/harness/tht/adapters/dwh/postgres.py @@ -0,0 +1,54 @@ +"""Direct PostgreSQL implementation of the DWH port.""" + +from tht.config import DatabaseConfig +from sqlalchemy.exc import OperationalError, SQLAlchemyError + +from tht.db import execute, sampling +from tht.db.connection import can_create_in_schema, make_engine, ping, writable_tables +from tht.db.introspect import introspect +from tht.execute import ExecResult, PlanSummary +from tht.mschema.models import PhysicalSchema +from tht.ports.dwh import DistinctValues, DwhCapabilities, DwhHealth + + +class PostgresDwhAdapter: + capabilities = DwhCapabilities() + + def __init__(self, config: DatabaseConfig, *, statement_timeout_ms: int = 30_000): + self._config = config + self._engine = make_engine(config) + self._statement_timeout_ms = statement_timeout_ms + + def health(self) -> DwhHealth: + try: + ping(self._engine) + except OperationalError as exc: + return DwhHealth(ok=False, detail=str(exc.orig), error_kind="connection") + except SQLAlchemyError as exc: + return DwhHealth(ok=False, detail=str(exc), error_kind="connection") + writable = tuple(writable_tables(self._engine, self._config.db_schema)) + can_create = can_create_in_schema(self._engine, self._config.db_schema) + return DwhHealth(ok=True, database=self._config.database, schema=self._config.db_schema, + read_only=not writable and not can_create, + writable_tables=writable, can_create=can_create) + + def introspect(self) -> PhysicalSchema: + return introspect(self._engine, self._config.database, self._config.db_schema) + + def run_query(self, sql: str, *, limit: int) -> ExecResult: + return execute.run_query( + self._engine, sql, limit=limit, timeout_ms=self._statement_timeout_ms + ) + + def explain(self, sql: str) -> PlanSummary: + return execute.explain(self._engine, sql, timeout_ms=self._statement_timeout_ms) + + def sample_column(self, table: str, column: str, *, limit: int) -> list[object]: + return sampling.sample_column( + self._engine, self._config.db_schema, table, column, limit=limit + ) + + def distinct_values(self, table: str, column: str, *, limit: int) -> DistinctValues: + return sampling.distinct_values( + self._engine, self._config.db_schema, table, column, max_values=limit + ) diff --git a/harness/tht/adapters/dwh/thoth_rest.py b/harness/tht/adapters/dwh/thoth_rest.py new file mode 100644 index 00000000..db738dbd --- /dev/null +++ b/harness/tht/adapters/dwh/thoth_rest.py @@ -0,0 +1,56 @@ +"""Thoth/PostgREST implementation of the DWH port.""" + +from tht.config import DatabaseIdentityConfig, RestConfig +from tht.db.introspect import introspect_rest +from tht.db import sampling +from tht.execute import ExecResult, ExecutionError, PlanSummary +from tht.mschema.models import PhysicalSchema +from tht.ports.dwh import DistinctValues, DwhCapabilities, DwhHealth +from tht.rest.client import RestClient, RestError +from tht.rest.execute import explain_rest, run_controlled_rest + + +class ThothRestDwhAdapter: + capabilities = DwhCapabilities() + + def __init__(self, database: DatabaseIdentityConfig, rest: RestConfig): + self._database = database + self._client = RestClient(rest) + + def health(self) -> DwhHealth: + try: + result = self._client.ping() + except RestError as exc: + return DwhHealth(ok=False, detail=str(exc), error_kind="connection") + ok = bool(result.get("db_connected") and result.get("schema_accessible")) + return DwhHealth(ok=ok, detail=None if ok else str(result), + database=self._database.database, schema=self._database.db_schema, + endpoint=self._client.cfg.base_url, read_only=True, + error_kind=None if ok else "inaccessible") + + def introspect(self) -> PhysicalSchema: + return introspect_rest( + self._client, self._database.database, self._database.db_schema + ) + + def run_query(self, sql: str, *, limit: int) -> ExecResult: + return run_controlled_rest(self._client, sql, limit=limit) + + def explain(self, sql: str) -> PlanSummary: + return explain_rest(self._client, sql) + + def sample_column(self, table: str, column: str, *, limit: int) -> list[object]: + try: + return sampling.sample_column_rest( + self._client, self._database.db_schema, table, column, limit=limit + ) + except RestError as exc: + raise ExecutionError(str(exc)) from exc + + def distinct_values(self, table: str, column: str, *, limit: int) -> DistinctValues: + try: + return sampling.distinct_values_rest( + self._client, self._database.db_schema, table, column, max_values=limit + ) + except RestError as exc: + raise ExecutionError(str(exc)) from exc diff --git a/harness/tht/adapters/evidence/__init__.py b/harness/tht/adapters/evidence/__init__.py new file mode 100644 index 00000000..805c886b --- /dev/null +++ b/harness/tht/adapters/evidence/__init__.py @@ -0,0 +1,7 @@ +"""Evidence source adapter implementations.""" + +from tht.adapters.evidence.filesystem import FilesystemEvidenceSource +from tht.adapters.evidence.http import HttpManifestEvidenceSource +from tht.adapters.evidence.s3 import S3EvidenceSource + +__all__ = ["FilesystemEvidenceSource", "HttpManifestEvidenceSource", "S3EvidenceSource"] diff --git a/harness/tht/adapters/evidence/filesystem.py b/harness/tht/adapters/evidence/filesystem.py new file mode 100644 index 00000000..8eca8d1a --- /dev/null +++ b/harness/tht/adapters/evidence/filesystem.py @@ -0,0 +1,148 @@ +"""Contained, race-safe filesystem Evidence source.""" + +import hashlib +import os +import stat +from datetime import UTC, datetime +from pathlib import Path, PurePosixPath +from urllib.parse import unquote, urlsplit + +from tht.ports.evidence import ( + AcquiredDocument, + EvidenceSourceError, + EvidenceSourceErrorCategory, + SourceObject, +) + + +class FilesystemEvidenceSource: + def __init__( + self, + root: Path | str, + *, + patterns: tuple[str, ...] | list[str] = ("**/*.md",), + max_bytes: int = 10 * 1024 * 1024, + ) -> None: + if max_bytes < 1: + raise ValueError("max_bytes must be positive") + if not patterns or any( + not pattern or Path(pattern).is_absolute() or ".." in Path(pattern).parts + for pattern in patterns + ): + raise ValueError("at least one non-empty discovery pattern is required") + try: + self.root = Path(root).expanduser().resolve(strict=True) + self._root_fd = os.open( + self.root, + os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW | os.O_CLOEXEC, + ) + except OSError as error: + raise ValueError("filesystem evidence root is unavailable") from error + self.patterns = tuple(patterns) + self.max_bytes = max_bytes + + def __del__(self): + root_fd = getattr(self, "_root_fd", None) + if root_fd is not None: + try: + os.close(root_fd) + except OSError: + pass + + @staticmethod + def _safe_error(operation: str, *, transient: bool = False, **details): + return EvidenceSourceError( + "filesystem source operation failed", + category=( + EvidenceSourceErrorCategory.TRANSIENT + if transient + else EvidenceSourceErrorCategory.PERMANENT + ), + details={"operation": operation, **details}, + ) + + def _open_read(self, relative: PurePosixPath) -> tuple[bytes, os.stat_result]: + parts = relative.parts + if not parts or any(part in {"", ".", ".."} for part in parts): + raise self._safe_error("path_validation") + directory_fd = os.dup(self._root_fd) + file_fd = None + try: + for component in parts[:-1]: + next_fd = os.open( + component, + os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW | os.O_CLOEXEC, + dir_fd=directory_fd, + ) + os.close(directory_fd) + directory_fd = next_fd + file_fd = os.open( + parts[-1], + os.O_RDONLY | os.O_NOFOLLOW | os.O_CLOEXEC, + dir_fd=directory_fd, + ) + file_stat = os.fstat(file_fd) + if not stat.S_ISREG(file_stat.st_mode): + raise self._safe_error("path_validation") + if file_stat.st_size > self.max_bytes: + raise self._safe_error("read", limit_bytes=self.max_bytes) + content = bytearray() + while len(content) <= self.max_bytes: + chunk = os.read(file_fd, min(64 * 1024, self.max_bytes + 1 - len(content))) + if not chunk: + break + content.extend(chunk) + if len(content) > self.max_bytes: + raise self._safe_error("read", limit_bytes=self.max_bytes) + return bytes(content), file_stat + except EvidenceSourceError: + raise + except OSError as error: + raise self._safe_error("open") from error + finally: + if file_fd is not None: + os.close(file_fd) + os.close(directory_fd) + + def _item( + self, relative: PurePosixPath, content: bytes, file_stat: os.stat_result + ) -> SourceObject: + relative_text = relative.as_posix() + return SourceObject( + source_id=f"filesystem:{hashlib.sha256(relative_text.encode()).hexdigest()}", + uri=(self.root / relative_text).as_uri(), + fingerprint=f"sha256:{hashlib.sha256(content).hexdigest()}", + modified_at=datetime.fromtimestamp(file_stat.st_mtime, tz=UTC), + metadata={"relative_path": relative_text}, + ) + + def discover(self): + candidates = { + path.relative_to(self.root).as_posix() + for pattern in self.patterns + for path in self.root.glob(pattern) + } + for relative_text in sorted(candidates): + relative = PurePosixPath(relative_text) + content, file_stat = self._open_read(relative) + yield self._item(relative, content, file_stat) + + def acquire(self, item: SourceObject) -> AcquiredDocument: + parsed = urlsplit(item.uri) + if parsed.scheme != "file" or parsed.netloc or parsed.query or parsed.fragment: + raise self._safe_error("acquire") + try: + relative = Path(unquote(parsed.path)).relative_to(self.root) + except ValueError as error: + raise self._safe_error("acquire") from error + pure_relative = PurePosixPath(relative.as_posix()) + content, file_stat = self._open_read(pure_relative) + expected = self._item(pure_relative, content, file_stat) + if item.source_id != expected.source_id or item.fingerprint != expected.fingerprint: + raise self._safe_error("acquire") + return AcquiredDocument( + source=expected, + content=content, + media_type="text/markdown" if relative.suffix.lower() == ".md" else None, + acquired_at=datetime.now(UTC), + ) diff --git a/harness/tht/adapters/evidence/http.py b/harness/tht/adapters/evidence/http.py new file mode 100644 index 00000000..9c895693 --- /dev/null +++ b/harness/tht/adapters/evidence/http.py @@ -0,0 +1,287 @@ +"""Explicit-manifest HTTP Evidence source with SSRF-safe bounded acquisition.""" + +import hashlib +import ipaddress +import socket +from collections import OrderedDict +from datetime import UTC, datetime +from email.utils import parsedate_to_datetime +from urllib.parse import urljoin, urlsplit + +import requests + +from tht.ports.evidence import ( + AcquiredDocument, + EvidenceSourceError, + EvidenceSourceErrorCategory, + SourceObject, + canonical_provenance_uri, +) + + +class HttpManifestEvidenceSource: + def __init__( + self, + urls: list[str] | tuple[str, ...], + *, + connect_timeout: float = 5, + read_timeout: float = 30, + max_bytes: int = 10 * 1024 * 1024, + max_redirects: int = 5, + allow_private_hosts: bool = False, + max_cache_bytes: int = 64 * 1024 * 1024, + ) -> None: + if not urls: + raise ValueError("HTTP evidence manifest must contain at least one URL") + if ( + connect_timeout <= 0 + or read_timeout <= 0 + or max_bytes < 1 + or max_redirects < 0 + or max_cache_bytes < 1 + ): + raise ValueError("HTTP evidence limits must be positive") + self._transport_by_uri: dict[str, str] = {} + for url in urls: + self._validate_url_shape(url) + provenance = canonical_provenance_uri(url) + if provenance in self._transport_by_uri: + raise ValueError("HTTP evidence manifest contains duplicate canonical provenance") + self._transport_by_uri[provenance] = url + self.connect_timeout = connect_timeout + self.read_timeout = read_timeout + self.max_bytes = max_bytes + self.max_redirects = max_redirects + self.allow_private_hosts = allow_private_hosts + self.max_cache_bytes = max_cache_bytes + self._session = requests.Session() + self._session.trust_env = False + self._cache: OrderedDict[str, AcquiredDocument] = OrderedDict() + # provenance -> (exact final effective URL, ETag, Last-Modified) + self._validators: dict[str, tuple[str, str | None, str | None]] = {} + self._cache_bytes = 0 + + def __repr__(self) -> str: + return f"HttpManifestEvidenceSource(objects={len(self._transport_by_uri)})" + + @staticmethod + def _safe_error(operation: str, *, transient: bool = False, **details): + return EvidenceSourceError( + "HTTP source operation failed", + category=( + EvidenceSourceErrorCategory.TRANSIENT + if transient + else EvidenceSourceErrorCategory.PERMANENT + ), + details={"operation": operation, **details}, + ) + + @staticmethod + def _validate_url_shape(url: str) -> None: + parsed = urlsplit(url) + if parsed.scheme not in {"http", "https"} or not parsed.hostname: + raise ValueError("HTTP evidence URLs must use http or https") + if parsed.username is not None or parsed.password is not None: + raise ValueError("HTTP evidence URLs must not contain userinfo credentials") + + @staticmethod + def _source_id(uri: str) -> str: + return f"http:{hashlib.sha256(uri.encode()).hexdigest()}" + + @staticmethod + def _normalized_ip(value: str) -> ipaddress.IPv4Address | ipaddress.IPv6Address: + address = ipaddress.ip_address(value.split("%", 1)[0]) + if isinstance(address, ipaddress.IPv6Address) and address.ipv4_mapped: + return address.ipv4_mapped + return address + + def _resolve_allowed(self, url: str) -> set[ipaddress.IPv4Address | ipaddress.IPv6Address]: + try: + self._validate_url_shape(url) + except ValueError as error: + raise self._safe_error("url_validation") from error + if self.allow_private_hosts: + return set() + parsed = urlsplit(url) + port = parsed.port or (443 if parsed.scheme == "https" else 80) + try: + rows = socket.getaddrinfo(parsed.hostname, port, type=socket.SOCK_STREAM) + addresses = {self._normalized_ip(row[4][0]) for row in rows} + except (OSError, ValueError) as error: + raise self._safe_error("resolution", transient=True) from error + if not addresses: + raise self._safe_error("resolution", transient=True) + # Reject the entire answer set if any address is private/reserved. Choosing only a public + # member would leave DNS ordering as a policy bypass. + if any(not address.is_global for address in addresses): + raise self._safe_error("network_policy") + return addresses + + def _verify_peer( + self, + response, + allowed: set[ipaddress.IPv4Address | ipaddress.IPv6Address], + ) -> None: + if self.allow_private_hosts: + return + try: + connection = response.raw._connection + peer = self._normalized_ip(connection.sock.getpeername()[0]) + except (AttributeError, OSError, TypeError, ValueError) as error: + raise self._safe_error("peer_validation", transient=True) from error + if not peer.is_global or peer not in allowed: + raise self._safe_error("network_policy") + + @staticmethod + def _status_category(status: int) -> EvidenceSourceErrorCategory: + if status in {408, 425, 429} or 500 <= status <= 599: + return EvidenceSourceErrorCategory.TRANSIENT + return EvidenceSourceErrorCategory.PERMANENT + + def _conditional_headers(self, provenance: str, request_url: str) -> dict[str, str]: + cached = self._cache.get(self._source_id(provenance)) + validators = self._validators.get(provenance) + if cached is None or validators is None: + return {} + final_url, etag, last_modified = validators + if request_url != final_url: + return {} + headers = {} + if etag: + headers["If-None-Match"] = etag + if last_modified: + headers["If-Modified-Since"] = last_modified + return headers + + def _remember( + self, + provenance: str, + final_url: str, + document: AcquiredDocument, + validators: tuple[str | None, str | None], + ) -> None: + source_id = document.source.source_id + old = self._cache.pop(source_id, None) + if old is not None: + self._cache_bytes -= len(old.content) + self._cache[source_id] = document + self._cache_bytes += len(document.content) + self._validators[provenance] = (final_url, *validators) + while self._cache and self._cache_bytes > self.max_cache_bytes: + evicted_id, evicted = self._cache.popitem(last=False) + self._cache_bytes -= len(evicted.content) + for uri in tuple(self._validators): + if self._source_id(uri) == evicted_id: + del self._validators[uri] + + def _download(self, transport_url: str, provenance: str) -> AcquiredDocument: + current = transport_url + try: + for redirect_count in range(self.max_redirects + 1): + headers = self._conditional_headers(provenance, current) + allowed = self._resolve_allowed(current) + response = None + try: + response = self._session.get( + current, + headers=headers, + stream=True, + allow_redirects=False, + timeout=(self.connect_timeout, self.read_timeout), + ) + self._verify_peer(response, allowed) + if response.is_redirect: + if redirect_count == self.max_redirects: + raise self._safe_error("redirect") + destination = urljoin(current, response.headers.get("Location", "")) + try: + self._validate_url_shape(destination) + except ValueError as error: + raise self._safe_error("redirect") from error + # The next iteration binds validators to the exact destination URL. + current = destination + continue + if response.status_code == 304: + cached = self._cache.get(self._source_id(provenance)) + binding = self._validators.get(provenance) + if ( + cached is None + or not headers + or binding is None + or binding[0] != current + ): + raise self._safe_error("conditional_response") + self._cache.move_to_end(cached.source.source_id) + return cached + if not 200 <= response.status_code <= 299: + raise EvidenceSourceError( + "HTTP status failure", + category=self._status_category(response.status_code), + details={"operation": "download", "status": response.status_code}, + ) + length = response.headers.get("Content-Length") + if length is not None and int(length) > self.max_bytes: + raise self._safe_error("download", limit_bytes=self.max_bytes) + content = bytearray() + for chunk in response.iter_content( + chunk_size=min(64 * 1024, self.max_bytes + 1) + ): + content.extend(chunk) + if len(content) > self.max_bytes: + raise self._safe_error("download", limit_bytes=self.max_bytes) + etag = response.headers.get("ETag") + last_modified = response.headers.get("Last-Modified") + media_type = ( + response.headers.get("Content-Type", "").split(";", 1)[0] or None + ) + break + finally: + if response is not None: + response.close() + except EvidenceSourceError: + raise + except (requests.Timeout, requests.ConnectionError, TimeoutError) as error: + raise self._safe_error("download", transient=True) from error + except requests.RequestException as error: + raise self._safe_error("download", transient=True) from error + except (OSError, ValueError) as error: + raise self._safe_error("download") from error + + modified_at = None + if etag: + fingerprint = f"etag:{hashlib.sha256(etag.encode()).hexdigest()}" + elif last_modified: + try: + modified_at = parsedate_to_datetime(last_modified).astimezone(UTC) + fingerprint = f"last-modified:{int(modified_at.timestamp())}" + except (TypeError, ValueError, OverflowError): + fingerprint = f"sha256:{hashlib.sha256(content).hexdigest()}" + else: + fingerprint = f"sha256:{hashlib.sha256(content).hexdigest()}" + item = SourceObject( + source_id=self._source_id(provenance), + uri=provenance, + fingerprint=fingerprint, + modified_at=modified_at, + ) + document = AcquiredDocument( + source=item, + content=bytes(content), + media_type=media_type, + acquired_at=datetime.now(UTC), + ) + self._remember(provenance, current, document, (etag, last_modified)) + return document + + def discover(self): + for provenance in sorted(self._transport_by_uri): + yield self._download(self._transport_by_uri[provenance], provenance).source + + def acquire(self, item: SourceObject) -> AcquiredDocument: + transport = self._transport_by_uri.get(item.uri) + if transport is None or item.source_id != self._source_id(item.uri): + raise self._safe_error("acquire") + document = self._download(transport, item.uri) + if document.source.fingerprint != item.fingerprint: + raise self._safe_error("acquire") + return document diff --git a/harness/tht/adapters/evidence/s3.py b/harness/tht/adapters/evidence/s3.py new file mode 100644 index 00000000..e0a45fb7 --- /dev/null +++ b/harness/tht/adapters/evidence/s3.py @@ -0,0 +1,152 @@ +"""Bounded S3-compatible Evidence source using the supported boto3 client.""" + +import hashlib +import ipaddress +import re +from datetime import UTC, datetime +from urllib.parse import quote, urlsplit + +from tht.ports.evidence import ( + AcquiredDocument, EvidenceSourceError, EvidenceSourceErrorCategory, SourceObject, +) + + +class S3EvidenceSource: + def __init__(self, *, bucket: str, prefix: str = "", endpoint_url: str | None = None, + region: str | None = None, access_key: str | None = None, + secret_key: str | None = None, session_token: str | None = None, + trusted_endpoint: bool = False, + allow_private_endpoint: bool = False, allow_insecure_endpoint: bool = False, + max_bytes: int = 10 * 1024 * 1024, max_objects: int = 10_000, + max_pages: int = 100, page_size: int = 1000, client=None) -> None: + bucket_valid = re.fullmatch(r"(?=.{3,63}$)(?!-)(?!.*\.\.)(?!.*\.-)(?!.*-\.)" + r"[a-z0-9](?:[a-z0-9.-]*[a-z0-9])?", bucket) + try: + ipaddress.ip_address(bucket) + bucket_is_ip = True + except ValueError: + bucket_is_ip = False + if (not bucket_valid or bucket_is_ip + or any(value < 1 for value in (max_bytes, max_objects, max_pages, page_size))): + raise ValueError("S3 evidence limits and bucket must be non-empty and positive") + if (prefix.startswith("/") or len(prefix.encode()) > 1024 + or any(ord(char) < 32 or ord(char) == 127 for char in prefix)): + raise ValueError("S3 prefix is invalid") + if endpoint_url: + parsed = urlsplit(endpoint_url) + if parsed.username or parsed.password: + raise ValueError("S3 endpoint must not contain credentials") + if parsed.scheme not in {"http", "https"}: + raise ValueError("S3 endpoint scheme must be exactly https or explicitly allowed http") + if parsed.scheme == "http" and not allow_insecure_endpoint: + raise ValueError("S3 endpoint must use HTTPS unless explicitly allowed") + if not parsed.hostname: + raise ValueError("S3 endpoint must include a hostname") + if parsed.path not in {"", "/"} or parsed.query or parsed.fragment: + raise ValueError("S3 custom endpoint must be an origin root without query/fragment") + if not trusted_endpoint: + raise ValueError("S3 custom endpoint requires explicit trusted_endpoint opt-in") + try: + literal = ipaddress.ip_address(parsed.hostname) + except ValueError: + literal = None + if literal is not None and not literal.is_global and not allow_private_endpoint: + raise ValueError("S3 private endpoint requires explicit opt-in") + self.bucket, self.prefix = bucket, prefix + self.max_bytes, self.max_objects = max_bytes, max_objects + self.max_pages, self.page_size = max_pages, min(page_size, 1000) + if client is None: + try: + import boto3 + from botocore.config import Config as BotoConfig + except ImportError as exc: # pragma: no cover - deployment optional dependency + raise RuntimeError("Install tht[s3] to use S3 Evidence") from exc + client = boto3.client("s3", endpoint_url=endpoint_url, region_name=region, + aws_access_key_id=access_key, + aws_secret_access_key=secret_key, + aws_session_token=session_token, verify=True, + config=BotoConfig(s3={"addressing_style": "path"})) + self._client = client + self._items: dict[str, tuple[SourceObject, str]] = {} + + @staticmethod + def _error(operation: str, transient: bool = False): + return EvidenceSourceError("S3 source operation failed", + category=(EvidenceSourceErrorCategory.TRANSIENT if transient + else EvidenceSourceErrorCategory.PERMANENT), + details={"operation": operation}) + + def discover(self): + count = pages = 0 + try: + token = None + for _ in range(self.max_pages): + params = {"Bucket": self.bucket, "Prefix": self.prefix, + "MaxKeys": self.page_size} + if token is not None: + params["ContinuationToken"] = token + page = self._client.list_objects_v2(**params) + pages += 1 + for row in page.get("Contents", []): + count += 1 + if count > self.max_objects: + raise self._error("object_limit") + key, etag = row.get("Key"), row.get("ETag") + if (not isinstance(key, str) or not key or not key.startswith(self.prefix) + or len(key.encode()) > 1024 + or any(ord(char) < 32 or ord(char) == 127 for char in key)): + raise self._error("invalid_key") + if not isinstance(etag, str) or not etag or len(etag) > 1024: + raise self._error("missing_validator") + uri = f"s3://{self.bucket}/{quote(key, safe='/')}" + fingerprint = f"etag:{hashlib.sha256(etag.encode()).hexdigest()}" + source_id = "s3:" + hashlib.sha256(uri.encode()).hexdigest() + modified = row.get("LastModified") + if modified is not None: + modified = modified.astimezone(UTC) + item = SourceObject(source_id=source_id, uri=uri, fingerprint=fingerprint, + modified_at=modified, + metadata={"size": int(row.get("Size", 0))}) + self._items[source_id] = (item, etag) + yield item + if not page.get("IsTruncated"): + return + token = page.get("NextContinuationToken") + if not isinstance(token, str) or not token: + raise self._error("list_continuation") + raise self._error("list_limit") + except EvidenceSourceError: + raise + except Exception as exc: + raise self._error("list", transient=True) from exc + + def acquire(self, item: SourceObject) -> AcquiredDocument: + binding = self._items.get(item.source_id) + if binding is None or item != binding[0]: + raise self._error("acquire") + discovered, etag = binding + key = discovered.uri.split(f"s3://{self.bucket}/", 1)[1] + from urllib.parse import unquote + key = unquote(key) + kwargs = {"Bucket": self.bucket, "Key": key} + body = None + try: + response = self._client.get_object(**kwargs) + body = response["Body"] + if response.get("ETag") != etag: + raise self._error("etag_changed") + if int(response.get("ContentLength", 0)) > self.max_bytes: + raise self._error("download_limit") + content = body.read(self.max_bytes + 1) + if len(content) > self.max_bytes: + raise self._error("download_limit") + return AcquiredDocument(source=item, content=content, + media_type=response.get("ContentType"), + acquired_at=datetime.now(UTC)) + except EvidenceSourceError: + raise + except Exception as exc: + raise self._error("download", transient=True) from exc + finally: + if body is not None: + body.close() diff --git a/harness/tht/adapters/factory.py b/harness/tht/adapters/factory.py new file mode 100644 index 00000000..0177a4db --- /dev/null +++ b/harness/tht/adapters/factory.py @@ -0,0 +1,141 @@ +"""Central construction of deployment-specific adapters.""" + +from tht.adapters.dwh import PostgresDwhAdapter, ThothRestDwhAdapter +from tht.adapters.evidence import FilesystemEvidenceSource, HttpManifestEvidenceSource +from tht.adapters.evidence.s3 import S3EvidenceSource +from tht.adapters.vector import PgVectorStore, ThothHttpVectorStore +from tht.config import Config, ConfigError +from tht.db.connection import make_engine +from tht.ports.dwh import DwhAdapter +from tht.ports.vector import VectorStore +from tht.vectorstore.rest_client import VectorRestClient + + +def build_dwh(cfg: Config) -> DwhAdapter: + """Build the DWH adapter selected by the validated workspace resource.""" + resource = cfg.dwh + match resource.type: + case "postgres_direct": + return PostgresDwhAdapter( + resource.connection, + statement_timeout_ms=cfg.execution.statement_timeout_ms, + ) + case "thoth_rest": + return ThothRestDwhAdapter(resource.database, resource.endpoint) + case other: # pragma: no cover - Pydantic's discriminator rejects this first. + raise ConfigError(f"Adapter DWH non supportato: {other}") + + +def build_vector_store(cfg: Config, *, require_write: bool = False) -> VectorStore: + """Build the vector adapter, optionally requiring an HTTP writer credential.""" + resource = cfg.vectors + if resource is None: + raise ConfigError("Risorsa vectors non configurata") + + match resource.type: + case "pgvector_direct": + reader = resource.reader or resource.connection + if require_write and resource.writer is None: + raise ConfigError("Vector writer non configurato per pgvector_direct") + return PgVectorStore( + reader, + resource.writer, + expected_dimension=cfg.embeddings.dim if cfg.embeddings is not None else None, + ) + case "thoth_vector_http": + if require_write and resource.writer is None: + raise ConfigError("Vector writer non configurato") + return ThothHttpVectorStore( + VectorRestClient(resource.reader) if resource.reader is not None else None, + VectorRestClient(resource.writer) if resource.writer is not None else None, + expected_dimension=cfg.embeddings.dim if cfg.embeddings is not None else None, + ) + case other: # pragma: no cover - Pydantic's discriminator rejects this first. + raise ConfigError(f"Adapter vector non supportato: {other}") + + +def build_vector_loader(cfg: Config, collection: str): + """Compatibility construction for legacy collection sync commands.""" + resource = cfg.vectors + if resource is None: + raise ConfigError("Risorsa vectors non configurata") + if cfg.embeddings is None: + raise ConfigError("Embeddings non configurati") + + if ( + resource.type == "thoth_vector_http" + and resource.writer is not None + and (cfg.profile == "workstation" or resource.direct is None) + ): + from tht.vectorstore.rest_writer import RestVectorWriter + + return RestVectorWriter(VectorRestClient(resource.writer), table=collection) + + connection = ( + resource.writer or resource.connection + if resource.type == "pgvector_direct" + else resource.direct + ) + if connection is None: + raise ConfigError("Vector writer non configurato") + from tht.vectorstore.store import VectorStore as TableVectorStore + + return TableVectorStore( + make_engine(connection), + schema=connection.db_schema, + table=collection, + dim=cfg.embeddings.dim, + ) + + +def build_evidence_sources(cfg: Config): + """Build configured Evidence sources, including the legacy curated filesystem tree.""" + evidence = cfg.evidence + if evidence is None: + return [] + sources = [] + if evidence.source_root is not None: + sources.append(FilesystemEvidenceSource(evidence.source_root / evidence.evidence_dir)) + for resource in evidence.sources: + match resource.type: + case "filesystem": + sources.append( + FilesystemEvidenceSource( + resource.root, + patterns=resource.patterns, + max_bytes=resource.max_bytes, + ) + ) + case "http": + sources.append( + HttpManifestEvidenceSource( + [url.get_secret_value() for url in resource.urls], + connect_timeout=resource.connect_timeout, + read_timeout=resource.read_timeout, + max_bytes=resource.max_bytes, + max_redirects=resource.max_redirects, + allow_private_hosts=resource.allow_private_hosts, + max_cache_bytes=resource.max_cache_bytes, + ) + ) + case "s3": + def secret(value): + return value.get_secret_value() if value is not None else None + + sources.append(S3EvidenceSource( + bucket=resource.bucket, prefix=resource.prefix, + endpoint_url=resource.endpoint_url, region=resource.region, + access_key=secret(resource.access_key), secret_key=secret(resource.secret_key), + session_token=secret(resource.session_token), + trusted_endpoint=resource.trusted_endpoint, + allow_private_endpoint=resource.allow_private_endpoint, + allow_insecure_endpoint=resource.allow_insecure_endpoint, + max_bytes=resource.max_bytes, max_objects=resource.max_objects, + max_pages=resource.max_pages, page_size=resource.page_size, + )) + case other: # pragma: no cover - Pydantic rejects unsupported discriminators. + raise ConfigError(f"Adapter evidence non supportato: {other}") + return sources + + +__all__ = ["build_dwh", "build_evidence_sources", "build_vector_loader", "build_vector_store"] diff --git a/harness/tht/adapters/vector/__init__.py b/harness/tht/adapters/vector/__init__.py new file mode 100644 index 00000000..4c6fa146 --- /dev/null +++ b/harness/tht/adapters/vector/__init__.py @@ -0,0 +1,7 @@ +"""Vector-store adapter implementations.""" + +from tht.adapters.vector.legacy_direct import LegacyDirectVectorStore +from tht.adapters.vector.pgvector import PgVectorStore +from tht.adapters.vector.thoth_http import ThothHttpVectorStore + +__all__ = ["LegacyDirectVectorStore", "PgVectorStore", "ThothHttpVectorStore"] diff --git a/harness/tht/adapters/vector/legacy_direct.py b/harness/tht/adapters/vector/legacy_direct.py new file mode 100644 index 00000000..c14202f3 --- /dev/null +++ b/harness/tht/adapters/vector/legacy_direct.py @@ -0,0 +1,70 @@ +"""Compatibility adapter for the existing direct PostgreSQL vector reader.""" + +from sqlalchemy import Engine + +from tht.ports.vector import ( + VectorCapabilities, + VectorHealth, + VectorStoreError, + VectorWriteRecord, + VectorWriteUnavailable, + require_positive_limit, +) +from tht.vectorstore.store import VectorHit, VectorStore as TableVectorStore + + +class LegacyDirectVectorStore: + """Read-only port wrapper around the legacy table-scoped pgvector store.""" + + capabilities = VectorCapabilities(search=True, existing_hashes=False, upsert=False) + + def __init__(self, engine: Engine, schema: str = "vectors", dim: int = 768): + self._engine = engine + self._schema = schema + self._dim = dim + + def health(self) -> VectorHealth: + try: + with self._engine.connect() as connection: + connection.exec_driver_sql("SELECT 1") + except Exception as exc: + return VectorHealth( + ok=False, + detail=str(exc), + read_configured=True, + read_reachable=False, + read_detail=str(exc), + expected_dimension=self._dim, + ) + return VectorHealth( + ok=True, + read_configured=True, + read_reachable=True, + expected_dimension=self._dim, + ) + + def search( + self, + collections: list[str], + embedding: list[float], + *, + limit: int, + kinds: list[str] | None = None, + metadata_filter: dict[str, object] | None = None, + ) -> list[VectorHit]: + require_positive_limit(limit) + if metadata_filter is not None: + raise VectorStoreError("Legacy vector store cannot enforce metadata filtering") + hits: list[VectorHit] = [] + for collection in collections: + table = TableVectorStore( + self._engine, schema=self._schema, table=collection, dim=self._dim + ) + hits.extend(table.search(embedding, top_n=limit, kinds=kinds)) + return sorted(hits, key=lambda hit: hit.similarity, reverse=True)[:limit] + + def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]: + raise VectorWriteUnavailable("Legacy direct reader has no writer interface") + + def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int: + raise VectorWriteUnavailable("Legacy direct reader has no writer interface") diff --git a/harness/tht/adapters/vector/pgvector.py b/harness/tht/adapters/vector/pgvector.py new file mode 100644 index 00000000..d0db7495 --- /dev/null +++ b/harness/tht/adapters/vector/pgvector.py @@ -0,0 +1,462 @@ +"""Direct PostgreSQL/pgvector implementation of the vector port.""" + +import json +import re + +from psycopg2 import sql +from sqlalchemy import Engine + +from tht.config import DatabaseConfig +from tht.db.connection import make_engine +from tht.ports.vector import ( + VectorCapabilities, + VectorHealth, + VectorReadUnavailable, + VectorStoreError, + VectorWriteRecord, + VectorWriteUnavailable, + require_positive_limit, +) +from tht.vectorstore.store import VectorHit, hit_from_metadata + + +COLLECTION_KINDS = { + "schema_records": {"schema_table", "schema_column"}, + "evidence": {"evidence"}, + "memory": {"memory", "solved_question"}, +} +ALLOWED_COLLECTIONS = frozenset(COLLECTION_KINDS) +ALLOWED_KINDS = frozenset().union(*COLLECTION_KINDS.values()) +_VECTOR_DIMENSION = re.compile(r"^(?:[a-z_][a-z0-9_]*\.)?vector\((\d+)\)$") + + +def _collection(schema: str, name: str) -> sql.Identifier: + if name not in ALLOWED_COLLECTIONS: + raise VectorStoreError(f"Collection not allowed: {name}") + return sql.Identifier(schema, name) + + +def _vector_literal(values: list[float]) -> str: + return "[" + ",".join(str(float(value)) for value in values) + "]" + + +def _vector_type(schema: str) -> sql.Identifier: + return sql.Identifier(schema, "vector") + + +def _cosine_operator(schema: str) -> sql.Composed: + return sql.SQL("OPERATOR({}.<=>)").format(sql.Identifier(schema)) + + +def _validate_collection_kinds(collection: str, kinds: list[str]) -> None: + invalid = set(kinds) - COLLECTION_KINDS[collection] + if invalid: + raise VectorStoreError(f"Kind not allowed for {collection}: {', '.join(sorted(invalid))}") + + +def _validate_known_kinds(kinds: list[str]) -> None: + invalid = set(kinds) - ALLOWED_KINDS + if invalid: + raise VectorStoreError(f"Kind not allowed: {', '.join(sorted(invalid))}") + + +class PgVectorStore: + """Direct store with independent reader and writer database credentials.""" + + def __init__( + self, + read_config: DatabaseConfig | None, + write_config: DatabaseConfig | None = None, + *, + expected_dimension: int | None = None, + ): + self._reader = make_engine(read_config) if read_config is not None else None + self._writer = make_engine(write_config) if write_config is not None else None + config = read_config or write_config + self._schema = config.db_schema if config is not None else "vectors" + if read_config and write_config and read_config.db_schema != write_config.db_schema: + raise VectorStoreError("Reader and writer vector schemas must match") + self._expected_dimension = expected_dimension + + @property + def capabilities(self) -> VectorCapabilities: + writable = self._writer is not None + return VectorCapabilities( + search=self._reader is not None, + existing_hashes=writable, + upsert=writable, + metadata_filter=self._reader is not None, + delete_generation=writable, + list_evidence_generations=writable, + ) + + def _probe( + self, engine: Engine | None, *, writable: bool + ) -> tuple[bool | None, str | None, set[int]]: + if engine is None: + return None, None, set() + try: + raw = engine.raw_connection() + try: + with raw.cursor() as cursor: + cursor.execute("SELECT 1") + cursor.execute( + "SELECT has_schema_privilege(current_user, %s, 'USAGE')", + (self._schema,), + ) + schema_usage = bool(cursor.fetchone()[0]) + if not schema_usage: + return False, "vector schema incomplete: missing schema usage", set() + cursor.execute( + """SELECT c.relname, format_type(a.atttypid, a.atttypmod), + has_table_privilege(current_user, c.oid, 'SELECT'), + has_table_privilege(current_user, c.oid, 'INSERT'), + has_table_privilege(current_user, c.oid, 'UPDATE'), + has_column_privilege(current_user, c.oid, 'record_key', 'SELECT') + AND has_column_privilege( + current_user, c.oid, 'content_hash', 'SELECT' + ) + AND has_column_privilege(current_user, c.oid, 'kind', 'SELECT'), + CASE WHEN id_attr.attname IS NOT NULL THEN + pg_get_serial_sequence( + format('%%I.%%I', n.nspname, c.relname), 'id' + ) + END AS id_sequence, + CASE WHEN id_attr.attname IS NOT NULL THEN + has_sequence_privilege( + current_user, + pg_get_serial_sequence( + format('%%I.%%I', n.nspname, c.relname), 'id' + ), + 'USAGE' + ) + END AS sequence_usage + FROM pg_class c + JOIN pg_namespace n ON n.oid = c.relnamespace + LEFT JOIN pg_attribute a ON a.attrelid = c.oid + AND a.attname = 'embedding' AND NOT a.attisdropped + LEFT JOIN pg_attribute id_attr ON id_attr.attrelid = c.oid + AND id_attr.attname = 'id' AND NOT id_attr.attisdropped + WHERE n.nspname = %s AND c.relname = ANY(%s) + AND c.relkind IN ('r', 'p')""", + (self._schema, list(ALLOWED_COLLECTIONS)), + ) + rows = cursor.fetchall() + present = {row[0] for row in rows} + missing_tables = sorted(ALLOWED_COLLECTIONS - present) + missing_embeddings = sorted(row[0] for row in rows if row[1] is None) + privilege_missing = sorted( + row[0] + for row in rows + if (writable and not (row[3] and row[4] and row[5])) + or (not writable and not row[2]) + ) + missing_sequences = sorted( + row[0] for row in rows if writable and row[6] is None + ) + sequence_privilege_missing = sorted( + row[0] for row in rows if writable and row[6] is not None and not row[7] + ) + problems = [] + if missing_tables: + problems.append("missing tables " + ", ".join(missing_tables)) + if missing_embeddings: + problems.append( + "missing embedding columns " + ", ".join(missing_embeddings) + ) + if privilege_missing: + authority = "write" if writable else "read" + problems.append( + f"missing {authority} privileges " + ", ".join(privilege_missing) + ) + if missing_sequences: + problems.append("missing id sequences " + ", ".join(missing_sequences)) + if sequence_privilege_missing: + problems.append( + "missing sequence privileges " + ", ".join(sequence_privilege_missing) + ) + if problems: + return False, "vector schema incomplete: " + "; ".join(problems), set() + dimensions = { + int(match.group(1)) + for _, type_name, *_ in rows + if (match := _VECTOR_DIMENSION.match(type_name)) + } + invalid_types = sorted( + row[0] + for row in rows + if row[1] is not None and not _VECTOR_DIMENSION.match(row[1]) + ) + if invalid_types: + return ( + False, + "vector schema incomplete: invalid embedding types " + + ", ".join(invalid_types), + set(), + ) + if self._expected_dimension is not None: + mismatches = sorted( + f"{name}={int(match.group(1))}" + for name, type_name, *_ in rows + if (match := _VECTOR_DIMENSION.match(type_name)) + and int(match.group(1)) != self._expected_dimension + ) + if mismatches: + return ( + False, + "embedding dimension mismatch: " + ", ".join(mismatches), + dimensions, + ) + return True, None, dimensions + finally: + raw.close() + except Exception as exc: + return False, f"vector database probe failed: {type(exc).__name__}", set() + + def health(self) -> VectorHealth: + read_ok, read_detail, read_dimensions = self._probe(self._reader, writable=False) + write_ok, write_detail, write_dimensions = self._probe(self._writer, writable=True) + dimensions = tuple(sorted(read_dimensions | write_dimensions)) + compatible = ( + None + if self._expected_dimension is None or not dimensions + else dimensions == (self._expected_dimension,) + ) + reachable = [value for value in (read_ok, write_ok) if value is not None] + details = [value for value in (read_detail, write_detail) if value] + return VectorHealth( + ok=bool(reachable) and all(reachable) and compatible is not False, + detail="; ".join(details) or None, + read_configured=self._reader is not None, + read_reachable=read_ok, + read_detail=read_detail, + write_configured=self._writer is not None, + write_reachable=write_ok, + write_detail=write_detail, + expected_dimension=self._expected_dimension, + observed_dimensions=dimensions, + dimension_compatible=compatible, + ) + + def search( + self, + collections: list[str], + embedding: list[float], + *, + limit: int, + kinds: list[str] | None = None, + metadata_filter: dict[str, object] | None = None, + ) -> list[VectorHit]: + require_positive_limit(limit) + if self._reader is None: + raise VectorReadUnavailable("Vector reader credential is not configured") + if self._expected_dimension is not None and len(embedding) != self._expected_dimension: + raise VectorStoreError("Query embedding dimension does not match configured dimension") + if kinds: + _validate_known_kinds(kinds) + hits: list[VectorHit] = [] + raw = None + try: + raw = self._reader.raw_connection() + with raw.cursor() as cursor: + for collection in collections: + table = _collection(self._schema, collection) + collection_kinds = ( + sorted(set(kinds) & COLLECTION_KINDS[collection]) if kinds else None + ) + if kinds and not collection_kinds: + continue + clauses = [] + filter_params = [] + if collection_kinds: + clauses.append(sql.SQL("kind = ANY(%s)")) + filter_params.append(collection_kinds) + if metadata_filter is not None: + if collection != "evidence" or set(metadata_filter) != { + "vector_generation", "document_ids", "workspace_id" + }: + raise VectorStoreError("Unsupported vector metadata filter") + generation = metadata_filter["vector_generation"] + document_ids = metadata_filter["document_ids"] + workspace_id = metadata_filter["workspace_id"] + if not isinstance(generation, str) or not isinstance(document_ids, list) or not isinstance(workspace_id, str): + raise VectorStoreError("Invalid vector metadata filter") + clauses.append(sql.SQL("metadata->>'vector_generation' = %s")) + clauses.append(sql.SQL("metadata->>'document_id' = ANY(%s)")) + clauses.append(sql.SQL("metadata->>'workspace_id' = %s")) + filter_params.extend((generation, document_ids, workspace_id)) + where = ( + sql.SQL(" WHERE ") + sql.SQL(" AND ").join(clauses) + if clauses else sql.SQL("") + ) + query = sql.SQL( + "SELECT metadata, 1 - (embedding {} %s::{}) AS similarity " + "FROM {}{} ORDER BY embedding {} %s::{}, record_key LIMIT %s" + ).format( + _cosine_operator(self._schema), + _vector_type(self._schema), + table, + where, + _cosine_operator(self._schema), + _vector_type(self._schema), + ) + params = [_vector_literal(embedding)] + params.extend(filter_params) + params.extend((_vector_literal(embedding), limit)) + cursor.execute(query, params) + hits.extend(hit_from_metadata(row[1], row[0]) for row in cursor.fetchall()) + except VectorStoreError: + raise + except Exception as exc: + raise VectorReadUnavailable("Vector read operation unavailable") from exc + finally: + if raw is not None: + raw.close() + return sorted(hits, key=lambda hit: (-hit.similarity, hit.id))[:limit] + + def _require_writer(self) -> Engine: + if self._writer is None: + raise VectorWriteUnavailable("Vector writer credential is not configured") + return self._writer + + def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]: + engine = self._require_writer() + table = _collection(self._schema, collection) + _validate_collection_kinds(collection, kinds) + raw = None + try: + raw = engine.raw_connection() + with raw.cursor() as cursor: + cursor.execute( + sql.SQL("SELECT record_key, content_hash FROM {} WHERE kind = ANY(%s)").format( + table + ), + (kinds,), + ) + return dict(cursor.fetchall()) + except VectorStoreError: + raise + except Exception as exc: + raise VectorWriteUnavailable("Vector write operation unavailable") from exc + finally: + if raw is not None: + raw.close() + + def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int: + engine = self._require_writer() + table = _collection(self._schema, collection) + for write_record in records: + _validate_collection_kinds(collection, [write_record.record.kind]) + if ( + self._expected_dimension is not None + and len(write_record.embedding) != self._expected_dimension + ): + raise VectorStoreError("Embedding dimension does not match configured dimension") + insert = sql.SQL( + "INSERT INTO {} (record_key, kind, content_hash, metadata, embedding) " + "VALUES (%s, %s, %s, %s::jsonb, %s::{}) " + "ON CONFLICT (record_key) DO NOTHING" + ).format(table, _vector_type(self._schema)) + update = sql.SQL( + "UPDATE {} SET kind = %s, content_hash = %s, metadata = %s::jsonb, " + "embedding = %s::{}, indexed_at = pg_catalog.now() WHERE record_key = %s" + ).format(table, _vector_type(self._schema)) + raw = None + try: + raw = engine.raw_connection() + with raw.cursor() as cursor: + for write_record in records: + record = write_record.record + metadata = { + "kind": record.kind, + "ref": record.ref, + "record_key": record.id, + "title": record.title, + "content": record.content, + **record.metadata, + } + metadata_json = json.dumps(metadata) + vector = _vector_literal(write_record.embedding) + cursor.execute( + insert, + (record.id, record.kind, write_record.content_hash, metadata_json, vector), + ) + if cursor.rowcount == 0: + cursor.execute( + update, + ( + record.kind, + write_record.content_hash, + metadata_json, + vector, + record.id, + ), + ) + raw.commit() + except VectorStoreError: + if raw is not None: + raw.rollback() + raise + except Exception as exc: + if raw is not None: + raw.rollback() + raise VectorWriteUnavailable("Vector write operation unavailable") from exc + finally: + if raw is not None: + raw.close() + return len(records) + + def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: + if collection != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None: + raise VectorStoreError("Only exact Evidence generations may be deleted") + if re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace_id) is None: + raise VectorStoreError("Invalid Evidence workspace namespace") + raw = None + try: + raw = self._require_writer().raw_connection() + with raw.cursor() as cursor: + cursor.execute( + sql.SQL( + "DELETE FROM {} WHERE kind = 'evidence' " + "AND metadata->>'vector_generation' = %s " + "AND metadata->>'workspace_id' = %s" + ).format(_collection(self._schema, collection)), + (generation, workspace_id), + ) + count = cursor.rowcount + raw.commit() + return count + except Exception as exc: + if raw is not None: + raw.rollback() + raise VectorWriteUnavailable("Vector generation cleanup unavailable") from exc + finally: + if raw is not None: + raw.close() + + def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: + if collection != "evidence": + raise VectorStoreError("Only exact Evidence generations may be listed") + if re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace_id) is None: + raise VectorStoreError("Invalid Evidence workspace namespace") + raw = None + try: + raw = self._require_writer().raw_connection() + with raw.cursor() as cursor: + cursor.execute( + sql.SQL( + "SELECT DISTINCT metadata->>'vector_generation' FROM {} " + "WHERE kind = 'evidence' AND metadata->>'vector_generation' " + "~ '^gen:[0-9a-f]{{32}}$' AND metadata->>'workspace_id' = %s ORDER BY 1" + ).format(_collection(self._schema, collection)), + (workspace_id,), + ) + return [row[0] for row in cursor.fetchall()] + except Exception as exc: + raise VectorWriteUnavailable("Vector generation inventory unavailable") from exc + finally: + if raw is not None: + raw.close() + + +__all__ = ["ALLOWED_COLLECTIONS", "PgVectorStore"] diff --git a/harness/tht/adapters/vector/thoth_http.py b/harness/tht/adapters/vector/thoth_http.py new file mode 100644 index 00000000..5a93b3c3 --- /dev/null +++ b/harness/tht/adapters/vector/thoth_http.py @@ -0,0 +1,191 @@ +"""Thoth vector HTTP adapter using distinct read and write clients.""" + +import re + +from tht.ports.vector import ( + VectorCapabilities, + VectorHealth, + VectorHit, + VectorReadUnavailable, + VectorStoreError, + VectorWriteRecord, + VectorWriteUnavailable, + require_positive_limit, +) +from tht.vectorstore.rest_client import VectorRestClient, VectorRestError +from tht.vectorstore.store import hit_from_metadata +from tht.adapters.vector.pgvector import ( + _collection, + _validate_collection_kinds, + _validate_known_kinds, +) + + +def _merge(hits: list[VectorHit], limit: int) -> list[VectorHit]: + return sorted(hits, key=lambda hit: (-hit.similarity, hit.id))[:limit] + + +class ThothHttpVectorStore: + """Vector port backed by the existing allowlisted REST RPCs.""" + + def __init__( + self, + reader: VectorRestClient | None, + writer: VectorRestClient | None, + expected_dimension: int | None = None, + ): + self._reader = reader + self._writer = writer + self._expected_dimension = expected_dimension + + @property + def capabilities(self) -> VectorCapabilities: + writable = self._writer is not None + return VectorCapabilities( + search=self._reader is not None, existing_hashes=writable, upsert=writable, + metadata_filter=self._reader is not None, delete_generation=writable, + list_evidence_generations=writable, + ) + + def health(self) -> VectorHealth: + read_reachable, read_detail, read_tables = self._probe(self._reader) + write_reachable, write_detail, write_tables = self._probe(self._writer) + dimensions = tuple(sorted({ + dimension + for row in [*read_tables, *write_tables] + if type(dimension := row.get("vector_dimensions")) is int + })) + compatible = ( + None + if self._expected_dimension is None or not dimensions + else dimensions == (self._expected_dimension,) + ) + reachable = [ + status for status in (read_reachable, write_reachable) if status is not None + ] + ok = bool(reachable) and all(reachable) and compatible is not False + details = [detail for detail in (read_detail, write_detail) if detail] + return VectorHealth( + ok=ok, + detail="; ".join(details) or None, + read_configured=self._reader is not None, + read_reachable=read_reachable, + read_detail=read_detail, + write_configured=self._writer is not None, + write_reachable=write_reachable, + write_detail=write_detail, + expected_dimension=self._expected_dimension, + observed_dimensions=dimensions, + dimension_compatible=compatible, + ) + + @staticmethod + def _probe(client: VectorRestClient | None) -> tuple[bool | None, str | None, list[dict]]: + if client is None: + return None, None, [] + try: + return True, None, client.list_tables() + except Exception as exc: + return False, str(exc), [] + + def search( + self, + collections: list[str], + embedding: list[float], + *, + limit: int, + kinds: list[str] | None = None, + metadata_filter: dict[str, object] | None = None, + ) -> list[VectorHit]: + require_positive_limit(limit) + if self._reader is None: + raise VectorReadUnavailable("Vector reader credential is not configured") + if self._expected_dimension is not None and len(embedding) != self._expected_dimension: + raise VectorStoreError("Query embedding dimension does not match configured dimension") + if kinds: + _validate_known_kinds(kinds) + hits: list[VectorHit] = [] + for collection in collections: + _collection("vectors", collection) + try: + if metadata_filter is None: + rows = self._reader.search_similar(collection, embedding, limit, kinds=kinds) + else: + rows = self._reader.search_similar( + collection, embedding, limit, kinds=kinds, + metadata_filter=metadata_filter, + ) + except VectorRestError as exc: + raise VectorStoreError(str(exc)) from exc + hits.extend( + hit_from_metadata(row.get("similarity", 0.0), row.get("metadata")) + for row in rows + ) + if kinds: + allowed = set(kinds) + hits = [hit for hit in hits if hit.kind in allowed] + return _merge(hits, limit) + + def _require_writer(self) -> VectorRestClient: + if self._writer is None: + raise VectorWriteUnavailable("Vector writer credential is not configured") + return self._writer + + def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]: + _collection("vectors", collection) + _validate_collection_kinds(collection, kinds) + try: + return self._require_writer().existing_hashes(collection, kinds) + except VectorRestError as exc: + raise VectorStoreError(str(exc)) from exc + + def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int: + writer = self._require_writer() + _collection("vectors", collection) + for record in records: + _validate_collection_kinds(collection, [record.record.kind]) + if ( + self._expected_dimension is not None + and len(record.embedding) != self._expected_dimension + ): + raise VectorStoreError("Embedding dimension does not match configured dimension") + rows = [self._row(record) for record in records] + try: + return writer.upsert_records(collection, rows) + except VectorRestError as exc: + raise VectorStoreError(str(exc)) from exc + + def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: + if collection != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None: + raise VectorStoreError("Only exact Evidence generations may be deleted") + try: + return self._require_writer().delete_generation(collection, generation, workspace_id) + except VectorRestError as exc: + raise VectorStoreError(str(exc)) from exc + + def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: + if collection != "evidence": + raise VectorStoreError("Only exact Evidence generations may be listed") + try: + return self._require_writer().list_evidence_generations(collection, workspace_id) + except VectorRestError as exc: + raise VectorWriteUnavailable("Vector generation inventory unavailable") from exc + + @staticmethod + def _row(write_record: VectorWriteRecord) -> dict: + record = write_record.record + metadata = { + "kind": record.kind, + "ref": record.ref, + "record_key": record.id, + "title": record.title, + "content": record.content, + **record.metadata, + } + return { + "record_key": record.id, + "kind": record.kind, + "content_hash": write_record.content_hash, + "metadata": metadata, + "embedding": write_record.embedding, + } diff --git a/harness/tht/cli/__init__.py b/harness/tht/cli/__init__.py index f9c20444..b95cd368 100644 --- a/harness/tht/cli/__init__.py +++ b/harness/tht/cli/__init__.py @@ -41,19 +41,23 @@ from tht.cli.cte_cmd import cte_app # noqa: E402 from tht.cli.datamart_cmd import datamart_app # noqa: E402 from tht.cli.db_cmd import db_app # noqa: E402 from tht.cli.decision_cmd import decision_app # noqa: E402 +from tht.cli.doctor_cmd import doctor # noqa: E402 from tht.cli.evidence_cmd import evidence_app # noqa: E402 from tht.cli.formula_cmd import formula_app # noqa: E402 from tht.cli.lsh_cmd import lsh_app # noqa: E402 from tht.cli.memory_cmd import memory_app # noqa: E402 from tht.cli.ollama_cmd import ollama_app # noqa: E402 from tht.cli.phase_cmd import phase_app # noqa: E402 +from tht.cli.preprocess_cmd import preprocess_app # noqa: E402 from tht.cli.schema_cmd import schema_app # noqa: E402 from tht.cli.search_cmd import search_app # noqa: E402 from tht.cli.session_cmd import session_app # noqa: E402 from tht.cli.sql_cmd import sql_app # noqa: E402 from tht.cli.vector_cmd import vector_app # noqa: E402 +import tht.cli.vector_migrate_cmd # noqa: E402, F401 app.add_typer(phase_app, name="phase") +app.add_typer(preprocess_app, name="preprocess") app.add_typer(config_app, name="config") app.add_typer(schema_app, name="schema") app.add_typer(session_app, name="session") @@ -69,3 +73,4 @@ app.add_typer(cte_app, name="cte") app.add_typer(datamart_app, name="datamart") app.add_typer(lsh_app, name="lsh") app.add_typer(ollama_app, name="ollama") +app.command("doctor")(doctor) diff --git a/harness/tht/cli/db_cmd.py b/harness/tht/cli/db_cmd.py index 549262d1..73d1b8df 100644 --- a/harness/tht/cli/db_cmd.py +++ b/harness/tht/cli/db_cmd.py @@ -1,40 +1,14 @@ from pathlib import Path import typer -from sqlalchemy.exc import OperationalError - +from tht.adapters.factory import build_dwh from tht.cli.config_cmd import CONFIG_OPT from tht.config import ConfigError, load_config -from tht.db.connection import can_create_in_schema, make_engine, ping, writable_tables from tht.db.fetch_ca import CaFetchError, describe_pem, fetch_chain_pem, parse_host_port db_app = typer.Typer(help="Operazioni sul database target") -def _ping_rest(cfg, schema: str) -> None: - """Health check via REST. Il read-only è garantito strutturalmente dall'API - (ammette solo SELECT/WITH): non serve il controllo dei privilegi di scrittura.""" - from tht.rest.client import RestClient, RestError - - try: - info = RestClient(cfg.rest).ping() - except RestError as e: - typer.secho(f"ERRORE di connessione: {e}", fg=typer.colors.RED, err=True) - raise typer.Exit(code=1) - if not info.get("db_connected") or not info.get("schema_accessible"): - typer.secho( - f"ERRORE: DWH non accessibile via REST (risposta: {info}).", - fg=typer.colors.RED, err=True, - ) - raise typer.Exit(code=1) - typer.secho( - f"OK: connesso via REST a {cfg.rest.base_url} (schema {schema})", fg=typer.colors.GREEN - ) - typer.secho( - "OK: accesso read-only garantito dall'API (solo SELECT/WITH).", fg=typer.colors.GREEN - ) - - @db_app.command("ping") def ping_cmd(config: Path = CONFIG_OPT) -> None: """Testa la connessione e verifica che l'utente sia effettivamente read-only.""" @@ -43,28 +17,31 @@ def ping_cmd(config: Path = CONFIG_OPT) -> None: except ConfigError as e: typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True) raise typer.Exit(code=1) - schema = cfg.database.db_schema - if cfg.database.transport == "rest": - _ping_rest(cfg, schema) - return - engine = make_engine(cfg.database) - try: - ping(engine) - except OperationalError as e: - typer.secho(f"ERRORE di connessione: {e.orig}", fg=typer.colors.RED, err=True) + health = build_dwh(cfg).health() + if not health.ok: + message = ( + f"ERRORE: DWH non accessibile via REST (risposta: {health.detail})." + if health.error_kind == "inaccessible" + else f"ERRORE di connessione: {health.detail}" + ) + typer.secho(message, fg=typer.colors.RED, err=True) raise typer.Exit(code=1) - typer.secho(f"OK: connesso a {cfg.database.database} (schema {schema})", fg=typer.colors.GREEN) - - writable = writable_tables(engine, schema) - can_create = can_create_in_schema(engine, schema) - if writable or can_create: + if health.endpoint: + typer.secho(f"OK: connesso via REST a {health.endpoint} (schema {health.schema})", + fg=typer.colors.GREEN) + typer.secho("OK: accesso read-only garantito dall'API (solo SELECT/WITH).", + fg=typer.colors.GREEN) + return + typer.secho(f"OK: connesso a {health.database} (schema {health.schema})", + fg=typer.colors.GREEN) + if not health.read_only: typer.secho( f"ERRORE: l'utente '{cfg.database.user}' NON e' read-only.", fg=typer.colors.RED, err=True ) - if writable: - typer.echo(f" Tabelle scrivibili: {', '.join(writable[:10])}", err=True) - if can_create: - typer.echo(f" L'utente puo' creare oggetti nello schema {schema}.", err=True) + if health.writable_tables: + typer.echo(f" Tabelle scrivibili: {', '.join(health.writable_tables[:10])}", err=True) + if health.can_create: + typer.echo(f" L'utente puo' creare oggetti nello schema {health.schema}.", err=True) typer.echo(" Crea un ruolo read-only con scripts/create_readonly_role.sql.", err=True) raise typer.Exit(code=2) typer.secho("OK: l'utente e' read-only sullo schema target.", fg=typer.colors.GREEN) diff --git a/harness/tht/cli/doctor_cmd.py b/harness/tht/cli/doctor_cmd.py new file mode 100644 index 00000000..b1c44621 --- /dev/null +++ b/harness/tht/cli/doctor_cmd.py @@ -0,0 +1,62 @@ +from __future__ import annotations + +import json +import os +from pathlib import Path +from typing import Any + +import typer +import yaml + +from tht.cli.config_cmd import CONFIG_OPT +from tht.config import ConfigError, load_config + + +def _emit(payload: dict[str, Any], as_json: bool) -> None: + if as_json: + # A single serializer call keeps stdout valid for machine consumers. + typer.echo(json.dumps(payload, sort_keys=True)) + return + for component, result in payload["components"].items(): + detail = "" + if result["status"] == "error": + detail = f" - {result['message']}" + elif component == "data_root" and result["status"] == "warning": + detail = " - set THT_DATA_ROOT to enable portable storage" + elif component == "workspace_paths" and result["status"] == "warning": + detail = " - absolute legacy roots: " + ", ".join(result["legacy_absolute"]) + typer.echo(f"{component}: {result['status']}{detail}") + + +def doctor( + config: Path = CONFIG_OPT, + as_json: bool = typer.Option(False, "--json", help="Emette diagnostica JSON."), +) -> None: + """Validate portable storage configuration without contacting external services.""" + data_root = os.environ.get("THT_DATA_ROOT") + components: dict[str, dict[str, Any]] = { + "config": {"status": "ok"}, + "data_root": {"status": "ok" if data_root else "warning"}, + } + try: + cfg = load_config(config) + except (ConfigError, yaml.YAMLError, OSError) as exc: + path_error = isinstance(exc, ConfigError) and "outside workspace" in str(exc) + target = "workspace_paths" if path_error else "config" + # Validation errors can contain Pydantic input excerpts, including credentials. + message = str(exc) if path_error else "configuration is invalid or unreadable" + components[target] = {"status": "error", "message": message} + payload = {"ok": False, "components": components} + _emit(payload, as_json) + raise typer.Exit(code=1) + + legacy_absolute = [ + name + for name in ("sessions", "artifacts", "indexes") + if getattr(cfg.roots, name).is_absolute() + ] + components["workspace_paths"] = { + "status": "warning" if legacy_absolute else "ok", + "legacy_absolute": legacy_absolute, + } + _emit({"ok": True, "components": components}, as_json) diff --git a/harness/tht/cli/lsh_cmd.py b/harness/tht/cli/lsh_cmd.py index c55c90af..d2467dbc 100644 --- a/harness/tht/cli/lsh_cmd.py +++ b/harness/tht/cli/lsh_cmd.py @@ -9,45 +9,91 @@ lsh_app = typer.Typer(help="Indice LSH su valori dei campi (derivato, rigenerabi def _lsh_dir(cfg) -> Path: - return cfg.paths.indexes / "lsh" + from tht.jobs.dwh_pipeline import resolve_dwh_snapshot + + return resolve_dwh_snapshot(cfg).lsh_dir + + +def _extract_lsh_values(dwh, physical, annotations, limit): + from tht.db.sampling import SkippedColumn, TruncatedColumn, is_text_type + from tht.mschema.eligibility import effective_eligibility + + values, skipped, truncated = {}, [], [] + for table_name, table in physical.tables.items(): + table_ann = annotations.tables.get(table_name) + for column_name, column in table.columns.items(): + ann_col = table_ann.columns.get(column_name) if table_ann else None + if not is_text_type(column.type) or not effective_eligibility(column, ann_col)[0]: + continue + try: + distinct = dwh.distinct_values(table_name, column_name, limit=limit) + except Exception as exc: + skipped.append(SkippedColumn(table_name, column_name, f"errore: {exc}")) + continue + vals = [str(value) for value in distinct.values if value not in (None, "")] + if vals: + values.setdefault(table_name, {})[column_name] = vals + if distinct.truncated: + truncated.append(TruncatedColumn(table_name, column_name, len(vals))) + return values, skipped, truncated + + +def build_lsh_artifacts( + cfg, *, dwh=None, verbose: bool = False, physical_file: Path | None = None, + output_dir: Path | None = None, +): + """Run the existing LSH extraction/build algorithm and persist its outputs.""" + from tht.adapters.factory import build_dwh + from tht.cli.schema_cmd import annotations_path + from tht.lshindex import build_index, save_index + from tht.mschema.models import Annotations, PhysicalSchema + + phys_file = physical_file or physical_path(cfg) + if not phys_file.exists(): + raise FileNotFoundError("physical catalog is missing; run schema introspect first") + physical = PhysicalSchema.from_yaml(phys_file) + annotations = Annotations.from_yaml(annotations_path(cfg)) + target = dwh if dwh is not None else build_dwh(cfg) + values, skipped, truncated = _extract_lsh_values( + target, physical, annotations, cfg.lsh.max_values_per_column + ) + lsh, minhashes = build_index(values, cfg.lsh, verbose=verbose) + save_index( + lsh, minhashes, cfg.lsh, output_dir or (cfg.paths.indexes / "lsh"), + name=cfg.database.db_schema, + ) + return minhashes, skipped, truncated, values @lsh_app.command("build") def build_cmd(config: Path = CONFIG_OPT) -> None: """Costruisce l'indice LSH dai valori del database e lo salva su pickle.""" - from tht.lshindex import build_index, save_index - from tht.mschema.models import Annotations, PhysicalSchema - cfg = _load_config_or_exit(config) - phys_file = physical_path(cfg) - if not phys_file.exists(): - typer.secho( - f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", - fg=typer.colors.RED, err=True, - ) - raise typer.Exit(code=1) - physical = PhysicalSchema.from_yaml(phys_file) - from tht.cli.schema_cmd import annotations_path - - annotations = Annotations.from_yaml(annotations_path(cfg)) - + dwh_root = cfg.paths.artifacts.parent / ".tht-dwh" + initialized = dwh_root.exists() or dwh_root.is_symlink() + if initialized: + phys_file = physical_path(cfg) + if not phys_file.exists(): + typer.secho( + f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", + fg=typer.colors.RED, err=True, + ) + raise typer.Exit(code=1) typer.echo("Estrazione valori (i più frequenti) dalle colonne testuali eligible...") - if cfg.database.transport == "rest": - from tht.db.sampling import unique_values_for_lsh_rest - from tht.rest.client import RestClient + from tht.cli.preprocess_cmd import run_dwh_from_config + from tht.lshindex import load_index - values, skipped, truncated = unique_values_for_lsh_rest( - RestClient(cfg.rest), physical, cfg.lsh, annotations - ) - else: - from tht.db.connection import make_engine - from tht.db.sampling import unique_values_for_lsh - - values, skipped, truncated = unique_values_for_lsh( - make_engine(cfg.database), physical, cfg.lsh, annotations - ) - n_values = sum(len(v) for t in values.values() for v in t.values()) - typer.echo(f" {n_values} valori da {sum(len(t) for t in values.values())} colonne") + report = run_dwh_from_config( + config, steps=("lsh",) if initialized else ("introspect", "lsh") + ) + if report.status != "succeeded": + typer.secho("ERRORE: DWH preprocessing failed", fg=typer.colors.RED, err=True) + raise typer.Exit(code=1) + _, minhashes, _ = load_index(_lsh_dir(cfg), name=cfg.database.db_schema) + skipped, truncated = [], [] + n_values = len(minhashes) + n_columns = len({(entry[1], entry[2]) for entry in minhashes.values()}) + typer.echo(f" {n_values} valori da {n_columns} colonne") for s in skipped: typer.secho(f" saltata {s.table}.{s.column}: {s.reason}", fg=typer.colors.YELLOW) for t in truncated: @@ -57,8 +103,6 @@ def build_cmd(config: Path = CONFIG_OPT) -> None: fg=typer.colors.YELLOW, ) - lsh, minhashes = build_index(values, cfg.lsh, verbose=True) - save_index(lsh, minhashes, cfg.lsh, _lsh_dir(cfg), name=cfg.database.db_schema) typer.secho( f"OK: indice LSH ({len(minhashes)} entry) -> {_lsh_dir(cfg)}", fg=typer.colors.GREEN ) diff --git a/harness/tht/cli/memory_cmd.py b/harness/tht/cli/memory_cmd.py index f0ee7571..982f1abb 100644 --- a/harness/tht/cli/memory_cmd.py +++ b/harness/tht/cli/memory_cmd.py @@ -153,9 +153,9 @@ def save_one_cmd( """ import json as _json + from tht.adapters.factory import build_vector_store from tht.cli.vector_cmd import make_embedder from tht.memory import load_registry, promote, save_one_memory - from tht.vectorstore.rest_client import VectorRestClient cfg = _load_config_or_exit(config) manifest = load_session_or_exit(cfg, session) @@ -167,6 +167,7 @@ def save_one_cmd( fg=typer.colors.RED, err=True, ) raise typer.Exit(code=4) + store = build_vector_store(cfg, require_write=True) sdir = session_dir(cfg, session) # Promuove la decisione scelta nel registro locale (idempotente: salta se gia' presente @@ -174,9 +175,8 @@ def save_one_cmd( promote(sdir, manifest, seqs=[decision], registry_path=registry_path(cfg)) records = [r for r in load_registry(registry_path(cfg)) if r.session_id == manifest.id] - writer = VectorRestClient(cfg.vector_write_rest) embedder = make_embedder(cfg.embeddings) - count = save_one_memory(records, decision, writer=writer, embedder=embedder) + count = save_one_memory(records, decision, store=store, embedder=embedder) msg = ( f"{count} memoria salvata su pgvector (decision_seq {decision})." @@ -449,23 +449,24 @@ def index_solved_session(cfg, session_id: str) -> int: Solleva RuntimeError se manca la writer key e SolvedIndexError se mancano gli artefatti: il finalize li degrada a warning, il comando CLI li converte in errori espliciti.""" + from tht.adapters.factory import build_vector_store from tht.cli.sql_cmd import promoted_tables_for from tht.cli.vector_cmd import make_embedder from tht.solved import build_solved_record, save_solved_question - from tht.vectorstore.rest_client import VectorRestClient if not has_vector_write_rest(cfg): raise RuntimeError( "vector_write_rest assente: la coppia domanda->SQL si indicizza solo con la " "writer key configurata nel workspace yaml" ) + store = build_vector_store(cfg, require_write=True) manifest = load_session_or_exit(cfg, session_id) record = build_solved_record( session_dir(cfg, session_id), manifest, promoted_tables_for(cfg, session_id) ) return save_solved_question( record, - writer=VectorRestClient(cfg.vector_write_rest), + store=store, embedder=make_embedder(cfg.embeddings), ) diff --git a/harness/tht/cli/preprocess_cmd.py b/harness/tht/cli/preprocess_cmd.py new file mode 100644 index 00000000..f1aca27a --- /dev/null +++ b/harness/tht/cli/preprocess_cmd.py @@ -0,0 +1,222 @@ +"""One-shot preprocessing commands.""" + +from __future__ import annotations + +import json +import re +import hashlib +from pathlib import Path + +import typer + +from tht.cli.config_cmd import CONFIG_OPT + + +preprocess_app = typer.Typer(help="Materialize versioned preprocessing artifacts") + + +def run_dwh_from_config( + config: Path, *, steps: tuple[str, ...], resume: str | None = None, +): + from tht.cli.lsh_cmd import build_lsh_artifacts + from tht.cli.schema_cmd import _load_config_or_exit, refresh_catalog + from tht.jobs.dwh_pipeline import ( + DwhPreprocessPipeline, config_dwh_binding, + ) + + cfg = _load_config_or_exit(config) + binding = config_dwh_binding(cfg) + workspace_root = cfg.paths.artifacts.parent + lsh_names = ( + f"{cfg.database.db_schema}_lsh.pkl", + f"{cfg.database.db_schema}_minhashes.pkl", + f"{cfg.database.db_schema}_meta.json", + ) + pipeline = DwhPreprocessPipeline( + workspace_id=binding["workspace_id"], + workspace_root=workspace_root, + config_fingerprint=binding["config_fingerprint"], + input_fingerprint=binding["input_fingerprint"], + introspect=lambda output: refresh_catalog(cfg, output_path=output), + build_lsh=lambda physical, output: build_lsh_artifacts( + cfg, physical_file=physical, output_dir=output + ), + lsh_filenames=lsh_names, + current_physical=cfg.paths.artifacts / "mschema" / "physical.yaml", + current_lsh_dir=cfg.paths.indexes / "lsh", + ) + return pipeline.run(steps, resume_run_id=resume) + + +def _parse_dwh_steps(value: str) -> tuple[str, ...]: + allowed = ("introspect", "lsh") + steps = tuple(part.strip() for part in value.split(",") if part.strip()) + if ( + not steps + or len(steps) != len(set(steps)) + or any(step not in allowed for step in steps) + or tuple(sorted(steps, key=allowed.index)) != steps + ): + raise ValueError("steps must be a unique ordered subset of introspect,lsh") + return steps + + +def run_from_config(config: Path, *, dry_run: bool = False, resume: str | None = None): + from tht.adapters.factory import build_evidence_sources, build_vector_store + from tht.cli.schema_cmd import _load_config_or_exit + from tht.cli.vector_cmd import make_embedder + from tht.corpus.chunk import ChunkPolicy + from tht.corpus.pipeline import CorpusPipeline + from tht.corpus.store import CorpusStore + + cfg = _load_config_or_exit(config) + if cfg.embeddings is None: + raise RuntimeError("embeddings are not configured") + corpus_root = cfg.paths.artifacts.parent / "corpus" + pipeline = CorpusPipeline( + store=CorpusStore(corpus_root), sources=build_evidence_sources(cfg), + embedder=make_embedder(cfg.embeddings), + vector_store=build_vector_store(cfg, require_write=True), + embedding_model=cfg.embeddings.model, embedding_dimensions=cfg.embeddings.dim, + chunk_policy=ChunkPolicy(version="chunk-v1", max_chars=cfg.vector.max_chunk_chars), + pipeline_version="evidence-v1", + retain_published_generations=cfg.vector.retain_published_generations, + ) + def fingerprint(value: str) -> str: + return "sha256:" + hashlib.sha256(value.encode()).hexdigest() + + return pipeline.run_as_job( + workspace_id=config.stem.lower().replace(".", "-").replace("_", "-"), + workspace_root=corpus_root.parent, + config_fingerprint=fingerprint(cfg.model_dump_json()), + input_fingerprint=fingerprint(config.resolve().as_posix()), + dry_run=dry_run, + resume_run_id=resume, + ) + + +def gc_from_config(config: Path, *, dry_run: bool = False): + from tht.adapters.factory import build_evidence_sources, build_vector_store + from tht.cli.schema_cmd import _load_config_or_exit + from tht.cli.vector_cmd import make_embedder + from tht.corpus.chunk import ChunkPolicy + from tht.corpus.pipeline import CorpusPipeline + from tht.corpus.store import CorpusStore + + cfg = _load_config_or_exit(config) + if cfg.embeddings is None: + raise RuntimeError("embeddings are not configured") + corpus_root = cfg.paths.artifacts.parent / "corpus" + pipeline = CorpusPipeline( + store=CorpusStore(corpus_root), sources=build_evidence_sources(cfg), + embedder=make_embedder(cfg.embeddings), vector_store=build_vector_store(cfg, require_write=True), + embedding_model=cfg.embeddings.model, embedding_dimensions=cfg.embeddings.dim, + chunk_policy=ChunkPolicy(version="chunk-v1", max_chars=cfg.vector.max_chunk_chars), + pipeline_version="evidence-v1", + retain_published_generations=cfg.vector.retain_published_generations, + ) + pipeline.workspace_id = config.stem.lower().replace(".", "-").replace("_", "-") + return pipeline.gc(workspace_root=corpus_root.parent, dry_run=dry_run) + + +@preprocess_app.command("evidence") +def evidence_cmd( + action: str | None = typer.Argument(None), + config: Path = CONFIG_OPT, + dry_run: bool = typer.Option(False, "--dry-run"), + resume: str | None = typer.Option(None, "--resume"), + json_output: bool = typer.Option(False, "--json"), +) -> None: + if action is not None and action != "gc": + raise typer.BadParameter("only the optional 'gc' action is supported") + if action == "gc": + try: + payload = gc_from_config(config, dry_run=dry_run) + except Exception: + payload = {"status": "failed", "error": "evidence cleanup failed"} + if json_output: + typer.echo(json.dumps(payload, sort_keys=True)) + else: + typer.secho("ERRORE: evidence cleanup failed", fg=typer.colors.RED, err=True) + raise typer.Exit(code=1) from None + if json_output: + typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True)) + else: + typer.echo(f"OK: evicted={len(payload['evicted'])} failures={len(payload['failures'])}") + return + if resume is not None and re.fullmatch(r"[0-9a-f]{32}", resume) is None: + payload = {"status": "failed", "error": "resume requires a preprocessing run id"} + if json_output: + typer.echo(json.dumps(payload, sort_keys=True)) + else: + typer.secho("ERRORE: resume requires a preprocessing run id", fg=typer.colors.RED, err=True) + raise typer.Exit(code=2) + try: + result = run_from_config(config, dry_run=dry_run, resume=resume) + except Exception: + payload = {"status": "failed", "error": "preprocessing failed"} + if json_output: + typer.echo(json.dumps(payload, sort_keys=True)) + else: + typer.secho("ERRORE: preprocessing failed", fg=typer.colors.RED, err=True) + raise typer.Exit(code=1) from None + payload = result.model_dump(mode="json") + if payload.get("status") != "succeeded": + payload["error"] = "preprocessing job failed" + if json_output: + typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True)) + else: + typer.secho("ERRORE: preprocessing job failed", fg=typer.colors.RED, err=True) + raise typer.Exit(code=1) + if json_output: + typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True)) + else: + counts = payload["counts"] + typer.echo( + f"OK: run={payload['run_id']} generation={payload['generation']} " + f"changed={counts['changed']} unchanged={counts['unchanged']} " + f"removed={counts['removed']}" + ) + + +@preprocess_app.command("dwh") +def dwh_cmd( + config: Path = CONFIG_OPT, + steps: str = typer.Option("introspect,lsh", "--steps"), + resume: str | None = typer.Option(None, "--resume"), + json_output: bool = typer.Option(False, "--json"), +) -> None: + try: + selected = _parse_dwh_steps(steps) + except ValueError: + payload = {"status": "failed", "error": "invalid DWH preprocessing steps"} + if json_output: + typer.echo(json.dumps(payload, sort_keys=True)) + else: + typer.secho("ERRORE: invalid DWH preprocessing steps", fg=typer.colors.RED, err=True) + raise typer.Exit(code=2) from None + if resume is not None and re.fullmatch(r"[0-9a-f]{32}", resume) is None: + payload = {"status": "failed", "error": "resume requires a preprocessing run id"} + if json_output: + typer.echo(json.dumps(payload, sort_keys=True)) + else: + typer.secho("ERRORE: resume requires a preprocessing run id", fg=typer.colors.RED, err=True) + raise typer.Exit(code=2) + try: + result = run_dwh_from_config(config, steps=selected, resume=resume) + except Exception: + payload = {"status": "failed", "error": "DWH preprocessing failed"} + if json_output: + typer.echo(json.dumps(payload, sort_keys=True)) + else: + typer.secho("ERRORE: DWH preprocessing failed", fg=typer.colors.RED, err=True) + raise typer.Exit(code=1) from None + payload = result.model_dump(mode="json") + if json_output: + typer.echo(json.dumps(payload, ensure_ascii=False, sort_keys=True)) + elif result.status == "succeeded": + typer.echo(f"OK: run={result.run_id} stages={','.join(selected)}") + else: + typer.secho(f"ERRORE: run={result.run_id} DWH preprocessing failed", fg=typer.colors.RED, err=True) + if result.status != "succeeded": + raise typer.Exit(code=1) diff --git a/harness/tht/cli/schema_cmd.py b/harness/tht/cli/schema_cmd.py index e2e671a1..f8b3c1b7 100644 --- a/harness/tht/cli/schema_cmd.py +++ b/harness/tht/cli/schema_cmd.py @@ -1,16 +1,31 @@ from pathlib import Path +import logging import typer -from sqlalchemy.exc import OperationalError - +from tht.adapters.factory import build_dwh from tht.cli.config_cmd import CONFIG_OPT from tht.config import ConfigError, load_config -from tht.db.connection import make_engine -from tht.db.introspect import IntrospectionError, introspect -from tht.db.sampling import add_examples +from tht.db.sampling import is_text_type from tht.mschema.eligibility import classify_all schema_app = typer.Typer(help="Gestione mschema (rappresentazione canonica dello schema)") +logger = logging.getLogger(__name__) + + +def _add_examples(dwh, phys, examples) -> None: + for table_name, table in phys.tables.items(): + for column_name, column in table.columns.items(): + if not is_text_type(column.type): + continue + try: + sampled = dwh.sample_column( + table_name, column_name, limit=examples.max_per_column + ) + except Exception as exc: + logger.warning("Campionamento saltato per %s.%s: %s", + table_name, column_name, exc) + continue + column.examples = [str(value) for value in sampled if value not in (None, "")] def _load_config_or_exit(config: Path): @@ -22,13 +37,27 @@ def _load_config_or_exit(config: Path): def physical_path(cfg) -> Path: - return cfg.paths.artifacts / "mschema" / "physical.yaml" + from tht.jobs.dwh_pipeline import resolve_dwh_snapshot + + if not (cfg.paths.artifacts.parent / ".tht-dwh").exists(): + return cfg.paths.artifacts / "mschema" / "physical.yaml" + return resolve_dwh_snapshot(cfg).physical def annotations_path(cfg) -> Path: return cfg.paths.artifacts / "mschema" / "annotations.yaml" +def refresh_catalog(cfg, *, dwh=None, output_path: Path | None = None): + """Run the existing catalog algorithm and persist its canonical output.""" + target = dwh if dwh is not None else build_dwh(cfg) + physical = target.introspect() + _add_examples(target, physical, cfg.examples) + classify_all(physical, cfg.eligibility) + physical.to_yaml(output_path or (cfg.paths.artifacts / "mschema" / "physical.yaml")) + return physical + + @schema_app.command("introspect") def introspect_cmd( config: Path = CONFIG_OPT, @@ -43,8 +72,11 @@ def introspect_cmd( Se physical.yaml esiste già, esce subito (cache); usa --refresh per rigenerarlo. """ cfg = _load_config_or_exit(config) - out = physical_path(cfg) - if out.exists() and not refresh: + dwh_root = cfg.paths.artifacts.parent / ".tht-dwh" + out = cfg.paths.artifacts / "mschema" / "physical.yaml" + if dwh_root.exists() or dwh_root.is_symlink(): + out = physical_path(cfg) + if (dwh_root.exists() or dwh_root.is_symlink()) and out.exists() and not refresh: from datetime import UTC, datetime from tht.mschema.models import PhysicalSchema @@ -64,33 +96,18 @@ def introspect_cmd( fg=typer.colors.GREEN, ) return - if cfg.database.transport == "rest": - from tht.db.introspect import introspect_rest - from tht.db.sampling import add_examples_rest - from tht.rest.client import RestClient, RestError + try: + from tht.cli.preprocess_cmd import run_dwh_from_config + from tht.mschema.models import PhysicalSchema - client = RestClient(cfg.rest) - try: - phys = introspect_rest( - client, database=cfg.database.database, schema=cfg.database.db_schema - ) - add_examples_rest(client, phys, cfg.examples) - classify_all(phys, cfg.eligibility) - except RestError as e: - typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True) - raise typer.Exit(code=1) - else: - engine = make_engine(cfg.database) - try: - phys = introspect( - engine, database=cfg.database.database, schema=cfg.database.db_schema - ) - add_examples(engine, phys, cfg.examples) - classify_all(phys, cfg.eligibility) - except (OperationalError, IntrospectionError) as e: - typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True) - raise typer.Exit(code=1) - phys.to_yaml(out) + report = run_dwh_from_config(config, steps=("introspect",)) + if report.status != "succeeded": + raise RuntimeError("DWH preprocessing failed") + out = physical_path(cfg) + phys = PhysicalSchema.from_yaml(out) + except Exception as e: + typer.secho(f"ERRORE: {e}", fg=typer.colors.RED, err=True) + raise typer.Exit(code=1) n_cols = sum(len(t.columns) for t in phys.tables.values()) n_ignored = sum( 1 for t in phys.tables.values() for c in t.columns.values() if not c.eligible diff --git a/harness/tht/cli/search_cmd.py b/harness/tht/cli/search_cmd.py index 082bc6f6..2b100542 100644 --- a/harness/tht/cli/search_cmd.py +++ b/harness/tht/cli/search_cmd.py @@ -21,8 +21,18 @@ DEFAULT_TOP_FALLBACK = 10 search_app = typer.Typer(help="Ricerca semantica (evidence/schema/values) nel vectorstore") +def _leased_dwh_snapshot(cfg, context: typer.Context): + from tht.jobs.dwh_pipeline import lease_dwh_snapshot + + lease = lease_dwh_snapshot(cfg) + snapshot = lease.__enter__() + context.call_on_close(lambda: lease.__exit__(None, None, None)) + return snapshot + + @search_app.command("find") def search_cmd( + ctx: typer.Context, keyword: str = typer.Argument(..., help="Termine da cercare, es. 'ablazione'."), config: Path = CONFIG_OPT, top: int | None = typer.Option( @@ -47,7 +57,18 @@ def search_cmd( from tht.search import combined_search cfg = _load_config_or_exit(config) + from tht.search.evidence import validate_corpus_workspace + + workspace_id = config.stem.lower().replace(".", "-").replace("_", "-") + validate_corpus_workspace(cfg, workspace_id) + dwh_snapshot = _leased_dwh_snapshot(cfg, ctx) require_vector_cfg(cfg) + from tht.search.evidence import active_searcher + + runtime_searcher = active_searcher( + cfg, open_searcher(cfg), + workspace_id=workspace_id, + ) if kind is not None and kind not in KIND_MAP: typer.secho( f"ERRORE: --kind sconosciuto: {kind} (validi: {', '.join(KIND_MAP)})", @@ -85,7 +106,7 @@ def search_cmd( lsh_hits = None try: lsh, minhashes, meta = load_index( - cfg.paths.indexes / "lsh", name=cfg.database.db_schema + dwh_snapshot.lsh_dir, name=cfg.database.db_schema ) hits = query_index(lsh, minhashes, keyword, meta, top_n=top * 3) lsh_hits = [(h.table, h.column, h.value, h.score) for h in hits] @@ -97,12 +118,12 @@ def search_cmd( ) if kind == "schema": - from tht.cli.schema_cmd import annotations_path, physical_path + from tht.cli.schema_cmd import annotations_path from tht.mschema.models import Annotations, PhysicalSchema from tht.mschema.render import to_mschema_text from tht.search import schema_tables - phys_file = physical_path(cfg) + phys_file = dwh_snapshot.physical if not phys_file.exists(): typer.secho( f"ERRORE: {phys_file} non trovato. Esegui prima `tht schema introspect`.", @@ -112,7 +133,7 @@ def search_cmd( candidates = combined_search( keyword=keyword, lsh_hits=lsh_hits, - store=open_searcher(cfg), embedder=make_embedder(cfg.embeddings), + store=runtime_searcher, embedder=make_embedder(cfg.embeddings), top=cfg.search.schema_chunk_pool, rrf_k=cfg.search.rrf_k, kinds=KIND_MAP["schema"], ) @@ -156,7 +177,7 @@ def search_cmd( kinds = KIND_MAP.get(kind) if kind else None results = combined_search( keyword=keyword, lsh_hits=lsh_hits if kind != "evidence" else None, - store=open_searcher(cfg), embedder=make_embedder(cfg.embeddings), + store=runtime_searcher, embedder=make_embedder(cfg.embeddings), top=top, rrf_k=cfg.search.rrf_k, kinds=kinds, ) @@ -216,6 +237,7 @@ PACK_EXCERPT_CHARS = 400 @search_app.command("pack") def pack_cmd( + ctx: typer.Context, question: str = typer.Argument(..., help="La domanda in linguaggio naturale."), config: Path = CONFIG_OPT, session: str = typer.Option( @@ -238,6 +260,11 @@ def pack_cmd( from tht.vectorstore.rest_client import VectorRestError cfg = _load_config_or_exit(config) + from tht.search.evidence import validate_corpus_workspace + + workspace_id = config.stem.lower().replace(".", "-").replace("_", "-") + validate_corpus_workspace(cfg, workspace_id) + dwh_snapshot = _leased_dwh_snapshot(cfg, ctx) require_vector_cfg(cfg) tables: list[dict] = [] @@ -249,17 +276,20 @@ def pack_cmd( vec = None searcher = embedder = None try: - searcher = open_searcher(cfg) + from tht.search.evidence import active_searcher + + searcher = active_searcher( + cfg, open_searcher(cfg), + workspace_id=workspace_id, + ) embedder = make_embedder(cfg.embeddings) vec = embedder.embed_query(question) except degrade as e: warnings.append(f"retrieval non disponibile ({e}): prosegui con le ricerche live") if vec is not None: - from tht.cli.schema_cmd import physical_path - descriptions: dict[str, str] = {} - phys_file = physical_path(cfg) + phys_file = dwh_snapshot.physical if phys_file.exists(): from tht.mschema.models import PhysicalSchema diff --git a/harness/tht/cli/sql_cmd.py b/harness/tht/cli/sql_cmd.py index 5fd1a0d3..e1dc609b 100644 --- a/harness/tht/cli/sql_cmd.py +++ b/harness/tht/cli/sql_cmd.py @@ -90,46 +90,18 @@ def validate_or_exit(cfg, sql: str, session_id: str | None): return result -def _ro_engine(cfg): - """Engine sul target con search_path impostato allo schema (nomi non qualificati).""" - from sqlalchemy import create_engine - - db = cfg.database - url = f"postgresql+psycopg2://{db.user}:{db.password}@{db.host}:{db.port}/{db.database}" - return create_engine( - url, echo=False, - connect_args={"options": f"-csearch_path={db.db_schema}"}, - ) - - -def _rest_client(cfg): - from tht.rest.client import RestClient - - return RestClient(cfg.rest) - - def do_explain(cfg, sql: str): - """EXPLAIN secondo il transport configurato (direct|rest).""" - if cfg.database.transport == "rest": - from tht.rest.execute import explain_rest + """EXPLAIN through the configured DWH adapter.""" + from tht.adapters.factory import build_dwh - return explain_rest(_rest_client(cfg), sql) - from tht.execute import explain - - return explain(_ro_engine(cfg), sql, timeout_ms=cfg.execution.statement_timeout_ms) + return build_dwh(cfg).explain(sql) def _run_transport(cfg, sql: str, *, limit: int): - """Dispatch all'esecutore controllato secondo il transport (direct|rest).""" - if cfg.database.transport == "rest": - from tht.rest.execute import run_controlled_rest + """Dispatch through the configured DWH adapter.""" + from tht.adapters.factory import build_dwh - return run_controlled_rest(_rest_client(cfg), sql, limit=limit) - from tht.execute import run_controlled - - return run_controlled( - _ro_engine(cfg), sql, limit=limit, timeout_ms=cfg.execution.statement_timeout_ms - ) + return build_dwh(cfg).run_query(sql, limit=limit) def do_run(cfg, sql: str, *, limit: int, offset: int = 0): diff --git a/harness/tht/cli/vector_cmd.py b/harness/tht/cli/vector_cmd.py index 102cf0f9..f54546bc 100644 --- a/harness/tht/cli/vector_cmd.py +++ b/harness/tht/cli/vector_cmd.py @@ -46,35 +46,27 @@ def open_store(cfg, table: str): Sul server preferisce la connessione diretta. In profilo workstation usa `vector_write_rest` se configurato, con upsert remoto non distruttivo. """ - if has_vector_write_rest(cfg) and (cfg.profile == "workstation" or cfg.vector_db is None): - from tht.vectorstore.rest_client import VectorRestClient - from tht.vectorstore.rest_writer import RestVectorWriter + from tht.adapters.factory import build_vector_loader - return RestVectorWriter(VectorRestClient(cfg.vector_write_rest), table=table) - - from tht.db.connection import make_engine - from tht.vectorstore.store import VectorStore - - engine = make_engine(cfg.vector_db) - return VectorStore( - engine, schema=cfg.vector_db.db_schema, table=table, dim=cfg.embeddings.dim - ) + return build_vector_loader(cfg, table) def open_searcher(cfg): """Searcher per la LETTURA (similarity search): via REST se `vector_rest` è configurato, altrimenti connessione diretta (dev/test).""" - if cfg.vector_rest is not None: - from tht.vectorstore.reader import RestSearcher - from tht.vectorstore.rest_client import VectorRestClient + from tht.adapters.factory import build_vector_store + from tht.vectorstore.reader import tables_for_kinds - return RestSearcher(VectorRestClient(cfg.vector_rest)) - from tht.db.connection import make_engine - from tht.vectorstore.reader import DirectSearcher + store = build_vector_store(cfg) - return DirectSearcher( - make_engine(cfg.vector_db), schema=cfg.vector_db.db_schema, dim=cfg.embeddings.dim - ) + class AdapterSearcher: + def search(self, query_vec, top_n=10, kinds=None, metadata_filter=None): + return store.search( + tables_for_kinds(kinds), query_vec, limit=top_n, kinds=kinds, + metadata_filter=metadata_filter, + ) + + return AdapterSearcher() def _print_stats(stats) -> None: diff --git a/harness/tht/cli/vector_migrate_cmd.py b/harness/tht/cli/vector_migrate_cmd.py new file mode 100644 index 00000000..3ed4243b --- /dev/null +++ b/harness/tht/cli/vector_migrate_cmd.py @@ -0,0 +1,227 @@ +"""Versioned, transactional migrations for the direct pgvector schema.""" + +from __future__ import annotations + +import hashlib +import json +import re +from dataclasses import dataclass +from importlib.resources import files +from importlib.resources.abc import Traversable +from pathlib import Path + +import typer +from sqlalchemy import create_engine, text +from sqlalchemy.exc import SQLAlchemyError + +from tht.cli.vector_cmd import vector_app + +MIGRATIONS_DIR = files("tht").joinpath("migrations", "vector") +_MIGRATION_NAME = re.compile(r"^(?P\d+)_(?P[a-z0-9_]+)\.sql$") +_LOCK_KEY = 7_304_708_654_221_909_028 + + +class MigrationError(RuntimeError): + """Raised when migration discovery or application is unsafe.""" + + +@dataclass(frozen=True) +class Migration: + version: str + name: str + path: Traversable + checksum: str + + +@dataclass(frozen=True) +class MigrationStatus: + applied: tuple[Migration, ...] + pending: tuple[Migration, ...] + drifted: tuple[Migration, ...] + + +def _migration_source(directory: Traversable | Path | str) -> Traversable: + return Path(directory) if isinstance(directory, (str, Path)) else directory + + +def _discover(directory: Traversable | Path | str) -> tuple[Migration, ...]: + source = _migration_source(directory) + migrations = [] + seen_versions: set[int] = set() + paths = [path for path in source.iterdir() if path.name.endswith(".sql")] + parsed = [] + for path in paths: + match = _MIGRATION_NAME.fullmatch(path.name) + if match is None: + raise MigrationError(f"Invalid migration filename: {path.name}") + version = match.group("version") + numeric_version = int(version) + if numeric_version in seen_versions: + raise MigrationError(f"Duplicate migration version: {numeric_version}") + seen_versions.add(numeric_version) + parsed.append((numeric_version, version, match.group("name"), path)) + for _, version, name, path in sorted(parsed, key=lambda item: item[0]): + migrations.append( + Migration( + version=version, + name=name, + path=path, + checksum=hashlib.sha256(path.read_bytes()).hexdigest(), + ) + ) + if not migrations: + raise MigrationError(f"No migrations found in {source}") + return tuple(migrations) + + +def _applied(connection) -> dict[str, str]: + exists = connection.execute( + text("SELECT pg_catalog.to_regclass('public.tht_vector_migrations')") + ).scalar() + if exists is None: + return {} + return dict( + connection.execute( + text("SELECT version, checksum FROM public.tht_vector_migrations") + ).all() + ) + + +def _reject_unknown_versions( + migrations: tuple[Migration, ...], applied_checksums: dict[str, str] +) -> None: + local_versions = {migration.version for migration in migrations} + unknown = sorted( + set(applied_checksums) - local_versions, + key=lambda version: (0, int(version)) if version.isdigit() else (1, version), + ) + if unknown: + raise MigrationError( + "Database migration versions absent from local manifest: " + ", ".join(unknown) + ) + + +def migration_status( + database_url: str, migrations_dir: Traversable | Path | str = MIGRATIONS_DIR +) -> MigrationStatus: + migrations = _discover(migrations_dir) + engine = create_engine(database_url) + try: + with engine.connect() as connection: + connection.exec_driver_sql("SET LOCAL search_path = pg_catalog, pg_temp") + applied_checksums = _applied(connection) + finally: + engine.dispose() + _reject_unknown_versions(migrations, applied_checksums) + applied = tuple( + migration + for migration in migrations + if applied_checksums.get(migration.version) == migration.checksum + ) + drifted = tuple( + migration + for migration in migrations + if migration.version in applied_checksums + and applied_checksums[migration.version] != migration.checksum + ) + pending = tuple( + migration for migration in migrations if migration.version not in applied_checksums + ) + return MigrationStatus(applied=applied, pending=pending, drifted=drifted) + + +def migrate( + database_url: str, migrations_dir: Traversable | Path | str = MIGRATIONS_DIR +) -> MigrationStatus: + migrations = _discover(migrations_dir) + engine = create_engine(database_url) + current: Migration | None = None + try: + with engine.begin() as connection: + connection.exec_driver_sql("SET LOCAL search_path = pg_catalog, pg_temp") + connection.execute( + text("SELECT pg_catalog.pg_advisory_xact_lock(:key)"), {"key": _LOCK_KEY} + ) + connection.exec_driver_sql( + """CREATE TABLE IF NOT EXISTS public.tht_vector_migrations ( + version text PRIMARY KEY, + name text NOT NULL, + checksum text NOT NULL, + applied_at timestamptz NOT NULL DEFAULT pg_catalog.now() + )""" + ) + connection.exec_driver_sql( + "REVOKE ALL ON public.tht_vector_migrations FROM PUBLIC" + ) + applied_checksums = _applied(connection) + _reject_unknown_versions(migrations, applied_checksums) + drifted = [ + item + for item in migrations + if item.version in applied_checksums + and applied_checksums[item.version] != item.checksum + ] + if drifted: + versions = ", ".join(item.version for item in drifted) + raise MigrationError(f"Migration checksum drift: {versions}") + for current in migrations: + if current.version in applied_checksums: + continue + connection.exec_driver_sql(current.path.read_text()) + connection.execute( + text( + "INSERT INTO public.tht_vector_migrations (version, name, checksum) " + "VALUES (:version, :name, :checksum)" + ), + { + "version": current.version, + "name": current.name, + "checksum": current.checksum, + }, + ) + except MigrationError: + raise + except SQLAlchemyError as exc: + filename = current.path.name if current is not None else "migration setup" + raise MigrationError(f"Failed to apply {filename}: {type(exc).__name__}") from exc + finally: + engine.dispose() + return migration_status(database_url, migrations_dir) + + +def _payload(status: MigrationStatus) -> dict[str, list[str]]: + return { + "applied": [item.version for item in status.applied], + "drifted": [item.version for item in status.drifted], + "pending": [item.version for item in status.pending], + } + + +@vector_app.command("migrate") +def migrate_cmd( + database_url: str = typer.Option( + ..., "--database-url", envvar="THT_VECTOR_ADMIN_URL", help="Admin PostgreSQL URL." + ), + status_only: bool = typer.Option(False, "--status", help="Inspect without applying."), + json_output: bool = typer.Option(False, "--json", help="Emit pristine JSON."), +) -> None: + """Apply or inspect the local pgvector schema migrations.""" + try: + status = migration_status(database_url) if status_only else migrate(database_url) + except (MigrationError, SQLAlchemyError) as exc: + if json_output: + typer.echo(json.dumps({"error": str(exc)}, sort_keys=True)) + else: + typer.echo(f"ERROR: {exc}", err=True) + raise typer.Exit(code=1) from None + payload = _payload(status) + if json_output: + typer.echo(json.dumps(payload, sort_keys=True)) + else: + typer.echo( + f"Applied: {len(status.applied)}; pending: {len(status.pending)}; " + f"drifted: {len(status.drifted)}" + ) + + +__all__ = ["MigrationError", "MigrationStatus", "migrate", "migration_status"] diff --git a/harness/tht/config.py b/harness/tht/config.py index 58f2f277..808a73cc 100644 --- a/harness/tht/config.py +++ b/harness/tht/config.py @@ -1,10 +1,13 @@ import os import re +import warnings from pathlib import Path -from typing import Any, Literal +from typing import Annotated, Any, Literal import yaml -from pydantic import BaseModel, Field, ValidationError +from pydantic import BaseModel, Field, PrivateAttr, SecretStr, model_validator, ValidationError + +from tht.config_compat import translate_legacy_config _ENV_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}") @@ -15,6 +18,7 @@ class ConfigError(Exception): def _expand_env(value: Any) -> Any: if isinstance(value, str): + def repl(m: re.Match) -> str: var = m.group(1) if var not in os.environ: @@ -32,6 +36,29 @@ def _expand_env(value: Any) -> Any: return value +def _resolve_secret_files(value: Any) -> Any: + if isinstance(value, dict): + resolved = {key: _resolve_secret_files(item) for key, item in value.items()} + for secret_name in ("password", "access_key", "secret_key", "session_token"): + file_name = f"{secret_name}_file" + if file_name not in resolved: + continue + if secret_name in resolved: + raise ConfigError(f"{secret_name} and {file_name} are mutually exclusive") + path = Path(resolved.pop(file_name)) + try: + secret = path.read_text() + except (OSError, UnicodeError) as exc: + raise ConfigError(f"Cannot read secret file: {path}") from exc + if not secret or any(char.isspace() for char in secret) or "\x00" in secret: + raise ConfigError(f"Invalid secret file: {path}") + resolved[secret_name] = secret + return resolved + if isinstance(value, list): + return [_resolve_secret_files(item) for item in value] + return value + + class DatabaseConfig(BaseModel): host: str = "localhost" port: int = 5432 @@ -56,12 +83,68 @@ class RestConfig(BaseModel): ssl_ca: str | None = None # path al certificato CA (per server con CA interna) +class DatabaseIdentityConfig(BaseModel): + database: str + db_schema: str = Field(alias="schema") + + model_config = {"populate_by_name": True} + + +class PostgresDwhConfig(BaseModel): + type: Literal["postgres_direct"] + connection: DatabaseConfig + + +class ThothRestDwhConfig(BaseModel): + type: Literal["thoth_rest"] + database: DatabaseIdentityConfig + endpoint: RestConfig + + +DwhResourceConfig = Annotated[ + PostgresDwhConfig | ThothRestDwhConfig, + Field(discriminator="type"), +] + + +class PgvectorDirectConfig(BaseModel): + type: Literal["pgvector_direct"] + reader: DatabaseConfig | None = None + writer: DatabaseConfig | None = None + # Deprecated compatibility: a single direct connection historically meant read-only. + connection: DatabaseConfig | None = None + + @model_validator(mode="after") + def validate_connections(self): + if self.reader is None and self.writer is None and self.connection is None: + raise ValueError("pgvector_direct requires a reader or writer connection") + return self + + +class ThothVectorHttpConfig(BaseModel): + type: Literal["thoth_vector_http"] + reader: RestConfig | None = None + writer: RestConfig | None = None + # Transitional direct loading path used by the server profile. + direct: DatabaseConfig | None = None + + +VectorResourceConfig = Annotated[ + PgvectorDirectConfig | ThothVectorHttpConfig, + Field(discriminator="type"), +] + + class PathsConfig(BaseModel): artifacts: Path = Path("artifacts") indexes: Path = Path("indexes") sessions: Path = Path("sessions") +class WorkspaceRoots(PathsConfig): + pass + + class ExamplesConfig(BaseModel): max_per_column: int = 10 @@ -84,21 +167,73 @@ class LshConfig(BaseModel): class EligibilityConfig(BaseModel): # Soglie del principio di column eligibility (testo ampio ignorato ovunque). - max_declared_len: int = 128 # char/varchar dichiarati <= soglia: eligible senza campionare - max_avg_length: int = 40 # fallback data-driven: lunghezza media valori campionati - max_sampled_len: int = 200 # fallback data-driven: lunghezza massima valore campionato + max_declared_len: int = 128 # char/varchar dichiarati <= soglia: eligible senza campionare + max_avg_length: int = 40 # fallback data-driven: lunghezza media valori campionati + max_sampled_len: int = 200 # fallback data-driven: lunghezza massima valore campionato # Colonne di servizio sempre ignorate per nome (match case-insensitive), a prescindere # dal tipo: metadati ETL/audit non analitici (es. timestamp di ultimo aggiornamento). ignore_columns: list[str] = ["etl_last_update"] +class FilesystemEvidenceSourceConfig(BaseModel): + type: Literal["filesystem"] + root: Path + patterns: list[str] = ["**/*.md"] + max_bytes: int = Field(default=10 * 1024 * 1024, gt=0) + + +class HttpEvidenceSourceConfig(BaseModel): + type: Literal["http"] + # Manifest URLs may contain signed query parameters. Treat the complete transport URL as + # secret-bearing configuration; adapters derive a query-free provenance URI from it. + urls: list[SecretStr] = Field(min_length=1) + connect_timeout: float = Field(default=5, gt=0) + read_timeout: float = Field(default=30, gt=0) + max_bytes: int = Field(default=10 * 1024 * 1024, gt=0) + max_redirects: int = Field(default=5, ge=0) + allow_private_hosts: bool = False + max_cache_bytes: int = Field(default=64 * 1024 * 1024, gt=0) + + +class S3EvidenceSourceConfig(BaseModel): + type: Literal["s3"] + bucket: str = Field(min_length=1) + prefix: str = "" + endpoint_url: str | None = None + region: str | None = None + access_key: SecretStr | None = None + secret_key: SecretStr | None = None + session_token: SecretStr | None = None + trusted_endpoint: bool = False + allow_private_endpoint: bool = False + allow_insecure_endpoint: bool = False + max_bytes: int = Field(default=10 * 1024 * 1024, gt=0) + max_objects: int = Field(default=10_000, gt=0) + max_pages: int = Field(default=100, gt=0) + page_size: int = Field(default=1000, gt=0, le=1000) + + +EvidenceSourceConfig = Annotated[ + FilesystemEvidenceSourceConfig | HttpEvidenceSourceConfig | S3EvidenceSourceConfig, + Field(discriminator="type"), +] + + class EvidenceSourcesConfig(BaseModel): - source_root: Path + # Legacy curated-tree configuration remains accepted during migration. + source_root: Path | None = None # cartella curata a mano nell'ETL (relativa a source_root): unica fonte delle # evidence. Niente piu' estrazione automatica dalle schede tabella: i documenti # qui dentro sono gia' evidence pronte (frontmatter + corpo), scelte e arricchite # dall'autore ETL e organizzate in sottocartelle per dominio. evidence_dir: str = "evidence" + sources: list[EvidenceSourceConfig] = [] + + @model_validator(mode="after") + def require_a_source(self): + if self.source_root is None and not self.sources: + raise ValueError("evidence requires source_root or sources") + return self class EmbeddingsConfig(BaseModel): @@ -114,11 +249,13 @@ class EmbeddingsConfig(BaseModel): class VectorConfig(BaseModel): max_chunk_chars: int = 4000 + # ACTIVE plus the two most recent rollback generations by default. + retain_published_generations: int = Field(default=3, ge=1) class SearchConfig(BaseModel): rrf_k: int = 60 - top_schema_tables: int = 12 # default `--top` per `tht search --kind schema` (n. tabelle) + top_schema_tables: int = 12 # default `--top` per `tht search --kind schema` (n. tabelle) schema_chunk_pool: int = 150 # chunk tabella/colonna fusi prima dell'aggregazione a tabella @@ -131,13 +268,27 @@ class ExecutionConfig(BaseModel): max_aggregate_cells: int = 20 max_export_rows: int = 100000 forbidden_functions: list[str] = [ - "setval", "nextval", "pg_advisory_lock", "pg_advisory_xact_lock", - "dblink", "dblink_exec", "pg_terminate_backend", "pg_cancel_backend", - "lo_import", "lo_export", "pg_reload_conf", + "setval", + "nextval", + "pg_advisory_lock", + "pg_advisory_xact_lock", + "dblink", + "dblink_exec", + "pg_terminate_backend", + "pg_cancel_backend", + "lo_import", + "lo_export", + "pg_reload_conf", ] class Config(BaseModel): + _workspace_id: str = PrivateAttr(default="default") + _config_source: str = PrivateAttr(default="direct") + dwh: DwhResourceConfig + vectors: VectorResourceConfig | None = None + roots: WorkspaceRoots = WorkspaceRoots() + # Compatibility views retained until all call sites consume typed resources. database: DatabaseConfig # Profilo dell'installazione, letto da THT_PROFILE (.env), non dallo yaml versionato. # server: ricostruisce i derivati (artefatti, LSH, vettori schema nel vectordb). @@ -166,6 +317,15 @@ class Config(BaseModel): # una API key separata dalla lettura; espone solo upsert/hash via RPC allowlist. vector_write_rest: RestConfig | None = None + @model_validator(mode="before") + @classmethod + def accept_legacy_constructor_fields(cls, value: Any) -> Any: + if not isinstance(value, dict) or "dwh" in value: + return value + translated, _ = translate_legacy_config(value) + _populate_legacy_views(translated) + return translated + def load_config(path: Path) -> Config: if not path.exists(): @@ -173,8 +333,11 @@ def load_config(path: Path) -> Config: raw = yaml.safe_load(path.read_text()) if not isinstance(raw, dict): raise ConfigError(f"Configurazione non valida (atteso un mapping YAML): {path}") + expanded = _resolve_secret_files(_expand_env(raw)) + translated, used_legacy = translate_legacy_config(expanded) + _populate_legacy_views(translated) try: - cfg = Config.model_validate(_expand_env(raw)) + cfg = Config.model_validate(translated) except ValidationError as e: raise ConfigError(f"Configurazione non valida in {path}:\n{e}") from e env_profile = os.environ.get("THT_PROFILE") @@ -188,4 +351,62 @@ def load_config(path: Path) -> Config: raise ConfigError( f"transport: rest richiede la sezione `rest` (base_url, api_key) in {path}." ) + data_root = os.environ.get("THT_DATA_ROOT") + if data_root: + # Import locally: paths owns resolution, while ConfigError remains the public + # configuration exception callers already handle. + from tht.paths import resolve_workspace_paths + + resolved = resolve_workspace_paths(path, cfg, Path(data_root)) + cfg = cfg.model_copy( + update={ + "paths": PathsConfig( + sessions=resolved.sessions, + artifacts=resolved.artifacts, + indexes=resolved.indexes, + ) + } + ) + elif not used_legacy: + # Modern `roots` replace `paths`; without a mounted data root retain the old + # working-directory-relative behavior used by local development. + cfg = cfg.model_copy(update={"paths": PathsConfig(**cfg.roots.model_dump())}) + if used_legacy: + warnings.warn( + "DEPRECATION: legacy workspace resource keys are deprecated; " + "use dwh, vectors, and roots.", + FutureWarning, + stacklevel=2, + ) + cfg._workspace_id = path.stem.lower().replace(".", "-").replace("_", "-") + cfg._config_source = path.resolve().as_posix() return cfg + + +def _populate_legacy_views(raw: dict[str, Any]) -> None: + """Populate old Config attributes for command compatibility during migration.""" + dwh = raw.get("dwh") + if "database" not in raw and isinstance(dwh, dict): + if dwh.get("type") == "postgres_direct": + raw["database"] = {**dwh["connection"], "transport": "direct"} + elif dwh.get("type") == "thoth_rest": + raw["database"] = { + **dwh["database"], + "user": "rest", + "password": "", + "transport": "rest", + } + raw["rest"] = dwh["endpoint"] + + vectors = raw.get("vectors") + if isinstance(vectors, dict): + if vectors.get("type") == "pgvector_direct": + raw.setdefault( + "vector_db", + vectors.get("writer") or vectors.get("reader") or vectors.get("connection"), + ) + elif vectors.get("type") == "thoth_vector_http": + raw.setdefault("vector_rest", vectors.get("reader")) + raw.setdefault("vector_write_rest", vectors.get("writer")) + raw.setdefault("vector_db", vectors.get("direct")) + raw.setdefault("paths", raw.get("roots", {})) diff --git a/harness/tht/config_compat.py b/harness/tht/config_compat.py new file mode 100644 index 00000000..b0276e8e --- /dev/null +++ b/harness/tht/config_compat.py @@ -0,0 +1,74 @@ +from __future__ import annotations + +from copy import deepcopy +from typing import Any + + +_LEGACY_RESOURCE_KEYS = { + "database", + "rest", + "vector_db", + "vector_rest", + "vector_write_rest", + "paths", +} + + +def _as_mapping(value: Any) -> dict[str, Any] | None: + if isinstance(value, dict): + return deepcopy(value) + model_dump = getattr(value, "model_dump", None) + if callable(model_dump): + return model_dump(by_alias=True) + return None + + +def translate_legacy_config(raw: dict[str, Any]) -> tuple[dict[str, Any], bool]: + """Translate the legacy flat resource keys without validating their contents.""" + translated = deepcopy(raw) + legacy = any(key in raw for key in _LEGACY_RESOURCE_KEYS) + if not legacy: + return translated, False + + database = _as_mapping(raw.get("database")) + rest = raw.get("rest") + if "dwh" not in translated and database is not None: + if database.get("transport", "direct") == "rest": + identity = { + key: database[key] + for key in ("database", "schema") + if key in database + } + translated["dwh"] = { + "type": "thoth_rest", + "database": identity, + "endpoint": rest, + } + else: + connection = database + connection.pop("transport", None) + translated["dwh"] = { + "type": "postgres_direct", + "connection": connection, + } + + if "vectors" not in translated: + vector_db = raw.get("vector_db") + reader = raw.get("vector_rest") + writer = raw.get("vector_write_rest") + if reader is not None or writer is not None: + translated["vectors"] = { + "type": "thoth_vector_http", + "reader": reader, + "writer": writer, + "direct": vector_db, + } + elif vector_db is not None: + translated["vectors"] = { + "type": "pgvector_direct", + "connection": vector_db, + } + + if "roots" not in translated and "paths" in raw: + translated["roots"] = deepcopy(raw["paths"]) + return translated, True diff --git a/harness/tht/corpus/__init__.py b/harness/tht/corpus/__init__.py new file mode 100644 index 00000000..34ac3790 --- /dev/null +++ b/harness/tht/corpus/__init__.py @@ -0,0 +1 @@ +"""Canonical, transport-independent Evidence corpus.""" diff --git a/harness/tht/corpus/chunk.py b/harness/tht/corpus/chunk.py new file mode 100644 index 00000000..3c297435 --- /dev/null +++ b/harness/tht/corpus/chunk.py @@ -0,0 +1,84 @@ +"""Versioned deterministic chunking for canonical corpus documents.""" + +import hashlib +import json +import re +from dataclasses import asdict, dataclass + +from tht.corpus.models import CanonicalChunk, CanonicalDocument + + +@dataclass(frozen=True, slots=True) +class ChunkPolicy: + version: str + max_chars: int + + def __post_init__(self) -> None: + if not self.version: + raise ValueError("chunk policy version must not be empty") + if self.max_chars <= 0: + raise ValueError("max_chars must be greater than zero") + + +def _hash(text: str) -> str: + return hashlib.sha256(text.encode("utf-8")).hexdigest() + + +def _contents(content: str, maximum: int) -> list[str]: + result: list[str] = [] + start = 0 + while start < len(content): + end = min(start + maximum, len(content)) + if end < len(content): + boundaries = list(re.finditer(r"\s+", content[start:end])) + if boundaries: + end = start + boundaries[-1].end() + result.append(content[start:end]) + start = end + return result + + +def _policy_fingerprint(policy: ChunkPolicy) -> str: + serialized = json.dumps(asdict(policy), ensure_ascii=False, sort_keys=True, separators=(",", ":")) + return f"sha256:{_hash(serialized)}" + + +def chunk(document: CanonicalDocument, policy: ChunkPolicy) -> list[CanonicalChunk]: + """Split canonical text with stable character-count boundaries and identifiers.""" + chunks: list[CanonicalChunk] = [] + policy_fingerprint = _policy_fingerprint(policy) + for ordinal, content in enumerate(_contents(document.content, policy.max_chars)): + chunk_hash = f"sha256:{_hash(content)}" + identifier = _hash( + ":".join( + ( + document.document_id, + document.content_hash, + policy_fingerprint, + str(ordinal), + chunk_hash, + ) + ) + ) + chunks.append( + CanonicalChunk( + chunk_id=f"chunk:{identifier}", + document_id=document.document_id, + ordinal=ordinal, + content=content, + content_hash=chunk_hash, + source_uri=document.source_uri, + pipeline_version=document.pipeline_version, + metadata={ + "chunk_policy": { + "version": policy.version, + "max_chars": policy.max_chars, + "fingerprint": policy_fingerprint, + }, + "document": document.model_dump(mode="json")["metadata"], + "source_fingerprint": document.source_fingerprint, + "title": document.title, + }, + ) + ) + return chunks diff --git a/harness/tht/corpus/models.py b/harness/tht/corpus/models.py new file mode 100644 index 00000000..4c4f93f5 --- /dev/null +++ b/harness/tht/corpus/models.py @@ -0,0 +1,163 @@ +"""Immutable records emitted by the Evidence preprocessing pipeline.""" + +import hashlib +import re +from collections.abc import Mapping +from datetime import UTC, datetime +from typing import Self + +from pydantic import BaseModel, ConfigDict, Field, JsonValue, field_validator, model_validator + +from tht.ports.evidence import ( + canonical_provenance_uri, + normalize_aware_datetime, + validate_namespaced_value, + validate_safe_metadata, +) + + +_NAMESPACED_ID = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$") +_SHA256 = re.compile(r"^sha256:[0-9a-f]{64}$") + + +def _validate_namespaced_id(value: str) -> str: + if not _NAMESPACED_ID.fullmatch(value): + raise ValueError("identifier must be namespaced as ':'") + return value + + +def _validate_hash(value: str) -> str: + if not _SHA256.fullmatch(value): + raise ValueError("content hash must be 'sha256:' followed by 64 lowercase hex digits") + return value + + +def _require_content_hash(content: str, content_hash: str) -> None: + expected = f"sha256:{hashlib.sha256(content.encode('utf-8')).hexdigest()}" + if content_hash != expected: + raise ValueError("content_hash must match the exact canonical UTF-8 content") + + +class _CanonicalValue(BaseModel): + model_config = ConfigDict( + frozen=True, extra="forbid", validate_default=True, revalidate_instances="always" + ) + + def model_copy(self, *, update: Mapping[str, object] | None = None, deep: bool = False) -> Self: + """Copy through full field and model validation, including manifest invariants.""" + data = self.model_dump(round_trip=True) + if update: + data.update(update) + return type(self).model_validate(data) + + +class _WithMetadata(_CanonicalValue): + metadata: dict[str, JsonValue] = Field(default_factory=dict) + _frozen_metadata = field_validator("metadata")(validate_safe_metadata) + + +class CanonicalDocument(_WithMetadata): + """Normalized text whose hash covers the exact stored UTF-8 content bytes.""" + document_id: str + source_id: str + source_uri: str + source_fingerprint: str = Field(min_length=1) + content_hash: str + title: str = "" + content: str + media_type: str = "text/plain" + modified_at: datetime | None = None + pipeline_version: str = Field(min_length=1) + + _document_id = field_validator("document_id")(_validate_namespaced_id) + _source_id = field_validator("source_id")(_validate_namespaced_id) + _source_uri = field_validator("source_uri")(canonical_provenance_uri) + _source_fingerprint = field_validator("source_fingerprint")(validate_namespaced_value) + _content_hash = field_validator("content_hash")(_validate_hash) + _modified_at = field_validator("modified_at")(normalize_aware_datetime) + + @model_validator(mode="after") + def content_hash_matches(self) -> "CanonicalDocument": + _require_content_hash(self.content, self.content_hash) + return self + + +class CanonicalChunk(_WithMetadata): + """Chunk text whose hash covers the exact stored UTF-8 content bytes.""" + chunk_id: str + document_id: str + ordinal: int = Field(ge=0) + content: str + content_hash: str + source_uri: str + pipeline_version: str = Field(min_length=1) + + _chunk_id = field_validator("chunk_id")(_validate_namespaced_id) + _document_id = field_validator("document_id")(_validate_namespaced_id) + _content_hash = field_validator("content_hash")(_validate_hash) + _source_uri = field_validator("source_uri")(canonical_provenance_uri) + + @model_validator(mode="after") + def content_hash_matches(self) -> "CanonicalChunk": + _require_content_hash(self.content, self.content_hash) + return self + + +class CorpusManifest(_WithMetadata): + """Description of one internally consistent publishable generation.""" + + schema_version: int = Field(default=1, ge=1) + manifest_id: str | None = None + created_at: datetime = Field(default_factory=lambda: datetime.now(UTC)) + pipeline_version: str = Field(default="evidence-v1", min_length=1) + embedding_model: str | None = None + embedding_dimensions: int | None = Field(default=None, gt=0) + vector_generation: str | None = None + documents: tuple[CanonicalDocument, ...] = Field(default_factory=tuple) + chunks: tuple[CanonicalChunk, ...] = Field(default_factory=tuple) + + _manifest_id = field_validator("manifest_id")( + lambda value: _validate_namespaced_id(value) if value is not None else None + ) + _vector_generation = field_validator("vector_generation")( + lambda value: _validate_namespaced_id(value) if value is not None else None + ) + _created_at = field_validator("created_at")(normalize_aware_datetime) + + @model_validator(mode="after") + def validate_generation(self) -> "CorpusManifest": + if (self.embedding_model is None) != (self.embedding_dimensions is None): + raise ValueError("embedding_model and embedding_dimensions must be set together") + if self.vector_generation is not None and self.embedding_model is None: + raise ValueError("vector_generation requires embedding model and dimension compatibility") + + document_ids = [document.document_id for document in self.documents] + source_ids = [document.source_id for document in self.documents] + chunk_ids = [chunk.chunk_id for chunk in self.chunks] + self._require_unique("document_id", document_ids) + self._require_unique("source_id", source_ids) + self._require_unique("chunk_id", chunk_ids) + + documents = {document.document_id: document for document in self.documents} + ordinals: dict[str, list[int]] = {} + for document in self.documents: + if document.pipeline_version != self.pipeline_version: + raise ValueError("document pipeline_version must match manifest pipeline_version") + for chunk in self.chunks: + document = documents.get(chunk.document_id) + if document is None: + raise ValueError(f"chunk references unknown document: {chunk.document_id}") + if chunk.pipeline_version != self.pipeline_version: + raise ValueError("chunk pipeline_version must match manifest pipeline_version") + if chunk.source_uri != document.source_uri: + raise ValueError("chunk source_uri must match its document provenance") + ordinals.setdefault(chunk.document_id, []).append(chunk.ordinal) + for document_id, values in ordinals.items(): + if sorted(values) != list(range(len(values))): + raise ValueError(f"chunk ordinals must be unique and contiguous for {document_id}") + return self + + @staticmethod + def _require_unique(field: str, values: list[str]) -> None: + if len(values) != len(set(values)): + raise ValueError(f"{field} values must be unique") diff --git a/harness/tht/corpus/normalize.py b/harness/tht/corpus/normalize.py new file mode 100644 index 00000000..54076826 --- /dev/null +++ b/harness/tht/corpus/normalize.py @@ -0,0 +1,147 @@ +"""Pure, deterministic conversion of acquired bytes into canonical text.""" + +import hashlib +import re +import unicodedata +from collections.abc import Mapping + +import yaml +from pydantic import JsonValue, TypeAdapter, ValidationError +from yaml.events import AliasEvent +from yaml.nodes import MappingNode + +from tht.corpus.models import CanonicalDocument +from tht.ports.evidence import AcquiredDocument, canonical_provenance_uri + + +MAX_DOCUMENT_BYTES = 10 * 1024 * 1024 +_CHARSET = re.compile(r"(?:^|;)\s*charset\s*=\s*[\"']?([^;\s\"']+)", re.IGNORECASE) +_FRONTMATTER = re.compile(r"\A---\n(.*?)\n---(?:\n|\Z)", re.DOTALL) +_JSON_OBJECT = TypeAdapter(dict[str, JsonValue]) +_MAX_FRONTMATTER_DEPTH = 20 +_MAX_FRONTMATTER_NODES = 1000 + + +class _FrontmatterLoader(yaml.SafeLoader): + """SafeLoader with bounded structure and no YAML graph features.""" + + def __init__(self, stream) -> None: + super().__init__(stream) + self._depth = 0 + self._nodes = 0 + + def compose_node(self, parent, index): + event = self.peek_event() + if isinstance(event, AliasEvent) or getattr(event, "anchor", None) is not None: + raise yaml.constructor.ConstructorError(None, None, "aliases are not allowed") + self._depth += 1 + self._nodes += 1 + if self._depth > _MAX_FRONTMATTER_DEPTH or self._nodes > _MAX_FRONTMATTER_NODES: + raise yaml.constructor.ConstructorError(None, None, "frontmatter is too complex") + try: + return super().compose_node(parent, index) + finally: + self._depth -= 1 + + def construct_mapping(self, node, deep=False): + if not isinstance(node, MappingNode): + return super().construct_mapping(node, deep=deep) + seen: set[object] = set() + for key_node, _ in node.value: + key = self.construct_object(key_node, deep=deep) + try: + duplicate = key in seen + seen.add(key) + except TypeError as error: + raise yaml.constructor.ConstructorError( + None, None, "mapping keys must be scalar" + ) from error + if duplicate: + raise yaml.constructor.ConstructorError(None, None, "duplicate mapping key") + return super().construct_mapping(node, deep=deep) + + +class PermanentNormalizationError(ValueError): + """A deterministic input failure which retrying cannot repair.""" + + def __init__(self, reason: str) -> None: + super().__init__(f"document normalization failed: {reason}") + self.reason = reason + self.permanent = True + + +def _sha256(value: str) -> str: + return hashlib.sha256(value.encode("utf-8")).hexdigest() + + +def _decode(acquired: AcquiredDocument) -> str: + if len(acquired.content) > MAX_DOCUMENT_BYTES: + raise PermanentNormalizationError("oversized") + + media_type = acquired.media_type or "text/plain" + charset = _CHARSET.search(media_type) + if charset and charset.group(1).lower().replace("_", "-") not in { + "utf-8", + "utf8", + "us-ascii", + "ascii", + }: + raise PermanentNormalizationError("unsupported_charset") + try: + return acquired.content.decode("utf-8-sig", errors="strict") + except UnicodeDecodeError as error: + raise PermanentNormalizationError("undecodable") from error + + +def _frontmatter(text: str) -> tuple[dict[str, JsonValue], str]: + match = _FRONTMATTER.match(text) + if match is None: + return {}, text + try: + loaded = yaml.load(match.group(1), Loader=_FrontmatterLoader) + if loaded is None: + loaded = {} + if not isinstance(loaded, Mapping): + raise TypeError("frontmatter is not a mapping") + metadata = _JSON_OBJECT.validate_python(dict(loaded)) + except (TypeError, UnicodeError, ValidationError, yaml.YAMLError) as error: + raise PermanentNormalizationError("invalid_frontmatter") from error + return metadata, text[match.end() :] + + +def normalize(acquired: AcquiredDocument, pipeline_version: str) -> CanonicalDocument: + """Normalize one transport result without I/O or implicit data loss.""" + if not pipeline_version: + raise ValueError("pipeline_version must not be empty") + + decoded = _decode(acquired) + canonical = unicodedata.normalize("NFC", decoded.replace("\r\n", "\n").replace("\r", "\n")) + frontmatter, content = _frontmatter(canonical) + source_uri = canonical_provenance_uri(acquired.source.uri) + identity = f"{acquired.source.source_id}\n{source_uri}" + media_type = (acquired.media_type or "text/plain").split(";", 1)[0].strip().lower() + metadata: dict[str, JsonValue] = { + "source": acquired.source.model_dump(mode="json")["metadata"], + "acquisition": acquired.model_dump(mode="json")["metadata"], + } + if frontmatter: + metadata["frontmatter"] = frontmatter + + try: + return CanonicalDocument( + document_id=f"doc:{_sha256(identity)}", + source_id=acquired.source.source_id, + source_uri=source_uri, + source_fingerprint=acquired.source.fingerprint, + content_hash=f"sha256:{_sha256(content)}", + title=str(frontmatter.get("title", "")), + content=content, + media_type=media_type, + modified_at=acquired.source.modified_at, + pipeline_version=pipeline_version, + metadata=metadata, + ) + except ValidationError as error: + if frontmatter: + raise PermanentNormalizationError("invalid_frontmatter") from error + raise diff --git a/harness/tht/corpus/pipeline.py b/harness/tht/corpus/pipeline.py new file mode 100644 index 00000000..625a5ff1 --- /dev/null +++ b/harness/tht/corpus/pipeline.py @@ -0,0 +1,806 @@ +"""Incremental Evidence preprocessing with generation-isolated vector writes.""" + +from __future__ import annotations + +import hashlib +import json +import re +import uuid +from collections.abc import Mapping, Sequence +from dataclasses import asdict, dataclass, field +from datetime import UTC +from pathlib import Path + +from tht.corpus.chunk import ChunkPolicy, chunk +from tht.corpus.models import CanonicalChunk, CanonicalDocument, CorpusManifest +from tht.corpus.normalize import normalize +from tht.corpus.store import CorpusStore +from tht.ports.evidence import EvidenceSource, SourceObject, canonical_provenance_uri +from tht.ports.vector import VectorStore, VectorWriteRecord +from tht.vectorstore.records import VectorRecord +from tht.jobs.models import JobSpec +from tht.jobs.runner import JobContext, StageArtifacts, run_job, seal_stage_artifacts + + +EVIDENCE_STAGE_IDS = ( + "discover", + "acquire_normalize_chunk", + "embed", + "vector_upsert", + "stage_validate", + "publish", + "retention_cleanup", +) + + +class PipelineError(RuntimeError): + """Credential-free failure at the preprocessing boundary.""" + + +@dataclass(frozen=True) +class PipelineResult: + status: str + generation: str | None + published: bool + changed: tuple[str, ...] + unchanged: tuple[str, ...] + removed: tuple[str, ...] + manifest: CorpusManifest = field(repr=False) + run_id: str | None = None + resumed_from: str | None = None + + def __repr__(self) -> str: + counts = { + "changed": len(self.changed), + "unchanged": len(self.unchanged), + "removed": len(self.removed), + } + return ( + f"PipelineResult(status={self.status!r}, generation={self.generation!r}, " + f"published={self.published!r}, counts={counts!r}, " + f"run_id={self.run_id!r}, resumed_from={self.resumed_from!r})" + ) + + def model_dump(self, mode=None): + def bounded(values: tuple[str, ...]) -> list[str]: + return [value[:200] for value in values[:100]] + + return { + "status": self.status, + "generation": self.generation, + "published": self.published, + "changed": bounded(self.changed), + "unchanged": bounded(self.unchanged), + "removed": bounded(self.removed), + "counts": { + "changed": len(self.changed), + "unchanged": len(self.unchanged), + "removed": len(self.removed), + "documents": len(self.manifest.documents), + "chunks": len(self.manifest.chunks), + }, + "manifest_id": self.manifest.manifest_id, + "run_id": self.run_id, + "resumed_from": self.resumed_from, + } + + +def _fingerprint(value) -> str: + payload = json.dumps(value, sort_keys=True, separators=(",", ":"), default=str) + return "sha256:" + hashlib.sha256(payload.encode()).hexdigest() + + +def _canonical_json(value): + if isinstance(value, Mapping): + return {str(key): _canonical_json(value[key]) for key in sorted(value)} + if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)): + return [_canonical_json(child) for child in value] + return value + + +def _source_snapshot(discovered) -> dict[str, dict]: + snapshot = {} + for _, item in discovered: + modified_at = item.modified_at.astimezone(UTC) if item.modified_at else None + metadata = _canonical_json(item.metadata) + snapshot[item.source_id] = { + "source_id": item.source_id, + "uri": item.uri, + "fingerprint": item.fingerprint, + "modified_at": modified_at.isoformat().replace("+00:00", "Z") if modified_at else None, + "metadata": metadata, + "media_type": metadata.get("media_type"), + "size": metadata.get("size"), + } + return snapshot + + +class CorpusPipeline: + def __init__( + self, *, store: CorpusStore, sources: list[EvidenceSource], embedder, + vector_store: VectorStore, embedding_model: str, embedding_dimensions: int, + chunk_policy: ChunkPolicy, pipeline_version: str, retain_published_generations: int = 3, + workspace_id: str | None = None, + ) -> None: + self.store = store + self.sources = sources + self.embedder = embedder + self.vector_store = vector_store + self.embedding_model = embedding_model + self.embedding_dimensions = embedding_dimensions + self.chunk_policy = chunk_policy + self.pipeline_version = pipeline_version + if isinstance(retain_published_generations, bool) or retain_published_generations < 1: + raise ValueError("retain_published_generations must be at least 1") + self.retain_published_generations = retain_published_generations + self.workspace_id = workspace_id + + def _assert_workspace_binding(self) -> None: + manifest = self.store.active_manifest() + if manifest is None: + if self.workspace_id is None: + self.workspace_id = "default" + return + persisted = manifest.metadata.get("workspace_id") + if not isinstance(persisted, str) or re.fullmatch( + r"[a-z][a-z0-9_-]{0,63}", persisted + ) is None: + raise PipelineError( + "corpus workspace ownership is missing or invalid; use a new corpus root or explicit rebuild" + ) + if self.workspace_id is None and isinstance(persisted, str): + self.workspace_id = persisted + return + if persisted != self.workspace_id: + raise PipelineError( + "corpus belongs to a different workspace; use a new corpus root or explicit rebuild" + ) + + def _protected_generations(self, workspace_root: Path) -> set[str]: + protected = {value for value in (self.store.active_generation(),) if value} + runs = workspace_root / ".tht-jobs" / "evidence" / "runs" + for checkpoint in runs.glob("*/checkpoint.json") if runs.exists() else (): + try: + state = json.loads(checkpoint.read_text(encoding="utf-8")) + if state.get("status") not in {"running", "failed"}: + continue + plan = checkpoint.parent / "artifacts" / "plan.json" + generation = json.loads(plan.read_text(encoding="utf-8")).get("generation") + if isinstance(generation, str): + protected.add(generation) + except (OSError, ValueError): + continue + return protected + + def gc(self, *, workspace_root: Path, dry_run: bool = False) -> dict: + with self.store.writer_lock(): + return self._gc(workspace_root=workspace_root, dry_run=dry_run) + + def _gc(self, *, workspace_root: Path, dry_run: bool = False) -> dict: + self._assert_workspace_binding() + published = self.store.published_generations() + list_vectors = getattr(self.vector_store, "list_evidence_generations", None) + vector_generations = set(list_vectors("evidence", self.workspace_id)) if list_vectors else set() + generations = sorted(set(published) | vector_generations) + job_protected = self._protected_generations(workspace_root) + active = self.store.active_generation() + rollback_count = self.retain_published_generations - 1 + rollback = [generation for generation in published if generation != active] + keep = ({active} if active else set()) | set(rollback[-rollback_count:] if rollback_count else ()) + fs_keep = keep | job_protected + vector_protected = set(fs_keep) + for generation in fs_keep: + try: + manifest = self.store.manifest(generation) + except (OSError, ValueError): + continue + vector_protected.update( + value for value in manifest.metadata.get("document_generations", {}).values() + if isinstance(value, str) + ) + evicted, failures = [], [] + filesystem_generations = set(self.store.list_generations()) + for generation in generations: + purge_vector = generation not in vector_protected + purge_filesystem = generation in filesystem_generations and generation not in fs_keep + if not purge_vector and not purge_filesystem: + continue + if dry_run: + evicted.append(generation) + continue + if purge_vector: + try: + self.vector_store.delete_generation("evidence", generation, self.workspace_id) + except Exception: + failures.append({"generation": generation, "error": "vector cleanup failed"}) + continue + try: + if purge_filesystem: + self.store.discard(generation) + evicted.append(generation) + except Exception: + failures.append({"generation": generation, "error": "filesystem cleanup failed"}) + return {"status": "partial" if failures else "succeeded", "dry_run": dry_run, + "active_generation": self.store.active_generation(), "evicted": evicted, + "protected": sorted(vector_protected), "failures": failures} + + def _discover(self) -> list[tuple[EvidenceSource, SourceObject]]: + discovered = [] + seen = set() + for source in self.sources: + for item in source.discover(): + if item.source_id in seen: + raise PipelineError("duplicate Evidence source identity") + seen.add(item.source_id) + discovered.append((source, item)) + return sorted(discovered, key=lambda pair: pair[1].source_id) + + def run(self, *, dry_run: bool = False, resume: str | None = None) -> PipelineResult: + with self.store.writer_lock(): + self._assert_workspace_binding() + return self._run(dry_run=dry_run, resume=resume) + + def run_as_job(self, **kwargs) -> PipelineResult: + self.workspace_id = kwargs["workspace_id"] + with self.store.writer_lock(): + self._assert_workspace_binding() + return self._run_as_job(**kwargs) + + def _run_as_job( + self, + *, + workspace_id: str, + workspace_root: Path, + config_fingerprint: str, + input_fingerprint: str, + dry_run: bool = False, + resume_run_id: str | None = None, + after_stage_return=None, + ) -> PipelineResult: + """Execute preprocessing through the durable shared job envelope.""" + discovered = self._discover() + discovered_fingerprint = _fingerprint( + {item.source_id: item.fingerprint for _, item in discovered} + ) + source_snapshot = _source_snapshot(discovered) + source_by_id = {item.source_id: (source, item) for source, item in discovered} + compatibility = _fingerprint({ + "pipeline": self.pipeline_version, + "model": self.embedding_model, + "dimensions": self.embedding_dimensions, + "chunk_policy": asdict(self.chunk_policy), + }) + job_binding = { + "config_fingerprint": config_fingerprint, + "input_fingerprint": input_fingerprint, + "compatibility_fingerprint": compatibility, + "pipeline_version": self.pipeline_version, + "chunk_policy_version": self.chunk_policy.version, + "embedding_model": self.embedding_model, + "embedding_dimensions": self.embedding_dimensions, + } + previous = self.store.active_manifest() + + def document_sources(manifest: CorpusManifest) -> dict[str, dict]: + return { + document.document_id: { + "document_id": document.document_id, + "source_id": document.source_id, + "source_uri": document.source_uri, + "source_fingerprint": document.source_fingerprint, + "modified_at": ( + document.modified_at.isoformat().replace("+00:00", "Z") + if document.modified_at else None + ), + "source_metadata": _canonical_json(document.metadata.get("source")), + "media_type": document.media_type, + "content_hash": document.content_hash, + "pipeline_version": document.pipeline_version, + } + for document in manifest.documents + } + + def active_assets_are_valid(manifest: CorpusManifest | None) -> bool: + if manifest is None or manifest.metadata.get("workspace_id") != workspace_id: + return False + actual_documents = {document.source_id: document for document in manifest.documents} + persisted_snapshot = _canonical_json(manifest.metadata.get("source_snapshot")) + if ( + not isinstance(persisted_snapshot, dict) + or set(persisted_snapshot) != set(actual_documents) + or manifest.metadata.get("compatibility_fingerprint") != compatibility + or _canonical_json(manifest.metadata.get("document_sources")) + != document_sources(manifest) + ): + return False + for source_id, document in actual_documents.items(): + source_payload = persisted_snapshot[source_id] + source = SourceObject.model_validate({ + "source_id": source_payload["source_id"], + "uri": source_payload["uri"], + "fingerprint": source_payload["fingerprint"], + "modified_at": source_payload["modified_at"], + "metadata": source_payload["metadata"], + }) + content = self.store.read_document(document.document_id, manifest.manifest_id) + expected_uri = canonical_provenance_uri(source.uri) + expected_id = "doc:" + hashlib.sha256( + f"{source.source_id}\n{expected_uri}".encode() + ).hexdigest() + expected_media_type = source_payload.get("media_type") + if ( + content != document.content + or document.document_id != expected_id + or document.source_id != source.source_id + or document.source_uri != expected_uri + or document.source_fingerprint != source.fingerprint + or document.modified_at != source.modified_at + or _canonical_json(document.metadata.get("source")) + != _canonical_json(source.metadata) + or ( + isinstance(expected_media_type, str) + and document.media_type != expected_media_type + ) + or document.pipeline_version != self.pipeline_version + ): + return False + expected_chunks = tuple( + part for document in manifest.documents for part in chunk(document, self.chunk_policy) + ) + if any(document.content and not chunk(document, self.chunk_policy) + for document in manifest.documents): + return False + if _canonical_json([part.model_dump(mode="json") for part in manifest.chunks]) != ( + _canonical_json([part.model_dump(mode="json") for part in expected_chunks]) + ): + return False + generations = manifest.metadata.get("document_generations") + if not isinstance(generations, Mapping): + return False + health = self.vector_store.health() + if ( + not health.ok + or health.dimension_compatible is not True + or health.expected_dimension != self.embedding_dimensions + or health.observed_dimensions != (self.embedding_dimensions,) + ): + return False + existing = self.vector_store.existing_hashes("evidence", ["evidence"]) + for part in expected_chunks: + generation = generations.get(part.document_id) + if not isinstance(generation, str): + return False + record_id = f"{workspace_id}:{generation}:{part.chunk_id}" + if existing.get(record_id) != part.content_hash: + return False + return True + + try: + active_assets_valid = active_assets_are_valid(previous) + except Exception: + active_assets_valid = False + reusable = ( + active_assets_valid + and _canonical_json(previous.metadata.get("source_snapshot")) == source_snapshot + and _canonical_json(previous.metadata.get("job_binding")) == job_binding + ) + if not dry_run and resume_run_id is None and reusable: + return PipelineResult( + "succeeded", previous.manifest_id, False, (), + tuple(sorted(item.source_id for _, item in discovered)), (), previous, + ) + spec = JobSpec( + workspace_id=workspace_id, + job_type="evidence", + workspace_root=workspace_root, + spec_version="jobs-v1", + pipeline_version=self.pipeline_version, + config_fingerprint=config_fingerprint, + input_fingerprint=_fingerprint([input_fingerprint, discovered_fingerprint]), + stage_ids=EVIDENCE_STAGE_IDS, + dry_run=dry_run, + resume_run_id=resume_run_id, + ) + + def artifact(context: JobContext, name: str) -> Path: + root = context.run_dir / "artifacts" + root.mkdir(exist_ok=True) + return root / name + + def write(context: JobContext, name: str, value) -> None: + artifact(context, name).write_text( + json.dumps(value, sort_keys=True, separators=(",", ":")), encoding="utf-8" + ) + + def read(context: JobContext, name: str): + try: + return json.loads(artifact(context, name).read_text(encoding="utf-8")) + except (OSError, ValueError) as error: + raise PipelineError("preprocessing checkpoint artifact is corrupt") from error + + def discover_stage(context: JobContext) -> None: + previous = self.store.active_manifest() + prior = {doc.source_id: doc for doc in previous.documents} if previous else {} + fingerprints = {item.source_id: item.fingerprint for _, item in discovered} + previous_snapshot = ( + _canonical_json(previous.metadata.get("source_snapshot")) if previous else {} + ) + rebuild = bool(previous and not active_assets_valid) + changed = sorted( + item.source_id for _, item in discovered + if rebuild or item.source_id not in prior + or previous_snapshot.get(item.source_id) != source_snapshot[item.source_id] + ) + unchanged = sorted(set(fingerprints) - set(changed)) + removed = sorted(set(prior) - set(fingerprints)) + write(context, "plan.json", { + "generation": f"gen:{context.run_id}", + "compatibility": compatibility, + "job_binding": job_binding, + "source_snapshot": source_snapshot, + "fingerprints": fingerprints, + "changed": changed, + "unchanged": unchanged, + "removed": removed, + "previous": previous.model_dump(mode="json") if previous else None, + }) + return StageArtifacts(("plan.json",)) + + def acquire_stage(context: JobContext) -> None: + if context.dry_run: + return StageArtifacts() + plan = read(context, "plan.json") + previous = CorpusManifest.model_validate(plan["previous"]) if plan["previous"] else None + prior = {doc.source_id: doc for doc in previous.documents} if previous else {} + documents = [prior[source_id] for source_id in plan["unchanged"]] + for source_id in plan["changed"]: + source, item = source_by_id[source_id] + documents.append(normalize(source.acquire(item), self.pipeline_version)) + documents.sort(key=lambda value: value.source_id) + chunks = [part for document in documents for part in chunk(document, self.chunk_policy)] + previous_generations = dict(previous.metadata.get("document_generations", {})) if previous else {} + changed = set(plan["changed"]) + generations = { + document.document_id: ( + plan["generation"] if document.source_id in changed + else previous_generations.get(document.document_id, previous.vector_generation) + ) for document in documents + } + manifest = CorpusManifest( + pipeline_version=self.pipeline_version, + embedding_model=self.embedding_model, + embedding_dimensions=self.embedding_dimensions, + vector_generation=plan["generation"], + documents=tuple(documents), chunks=tuple(chunks), + metadata={ + "workspace_id": self.workspace_id, + "compatibility_fingerprint": compatibility, + "job_binding": plan["job_binding"], + "source_snapshot": plan["source_snapshot"], + "document_sources": { + document.document_id: { + "document_id": document.document_id, + "source_id": document.source_id, + "source_uri": document.source_uri, + "source_fingerprint": document.source_fingerprint, + "modified_at": ( + document.modified_at.isoformat().replace("+00:00", "Z") + if document.modified_at else None + ), + "source_metadata": _canonical_json( + document.metadata.get("source") + ), + "media_type": document.media_type, + "content_hash": document.content_hash, + "pipeline_version": document.pipeline_version, + } + for document in documents + }, + "fingerprints": plan["fingerprints"], + "removed": plan["removed"], + "document_generations": generations, + }, + ) + write(context, "manifest.json", manifest.model_dump(mode="json")) + return StageArtifacts(("manifest.json",)) + + def embed_stage(context: JobContext) -> None: + if context.dry_run: + return StageArtifacts() + plan = read(context, "plan.json") + manifest = CorpusManifest.model_validate(read(context, "manifest.json")) + changed_docs = {doc.document_id for doc in manifest.documents if doc.source_id in plan["changed"]} + parts = [part for part in manifest.chunks if part.document_id in changed_docs] + embeddings = self.embedder.embed_documents([part.content for part in parts]) + if len(embeddings) != len(parts) or any( + len(vector) != self.embedding_dimensions for vector in embeddings + ): + raise PipelineError("embedding output is incompatible") + write(context, "embeddings.json", embeddings) + return StageArtifacts(("embeddings.json",)) + + def records(context: JobContext): + plan = read(context, "plan.json") + manifest = CorpusManifest.model_validate(read(context, "manifest.json")) + changed_docs = {doc.document_id for doc in manifest.documents if doc.source_id in plan["changed"]} + parts = [part for part in manifest.chunks if part.document_id in changed_docs] + embeddings = read(context, "embeddings.json") + return [self._vector_record(part, vector, plan["generation"], self.workspace_id) + for part, vector in zip(parts, embeddings, strict=True)] + + def compensate(context: JobContext) -> None: + generation = read(context, "plan.json")["generation"] + if self.store.active_generation() != generation: + self.store.discard(generation) + try: + self.vector_store.delete_generation("evidence", generation, self.workspace_id) + except Exception: + pass + write(context, "compensated.json", {"generation": generation}) + + def rotate_compensated_generation(context: JobContext) -> None: + marker = artifact(context, "compensated.json") + if not marker.exists(): + return + plan = read(context, "plan.json") + old = plan["generation"] + plan["generation"] = f"gen:{uuid.uuid4().hex}" + write(context, "plan.json", plan) + manifest = CorpusManifest.model_validate(read(context, "manifest.json")) + changed = set(plan["changed"]) + generations = dict(manifest.metadata["document_generations"]) + for document in manifest.documents: + if document.source_id in changed and generations.get(document.document_id) == old: + generations[document.document_id] = plan["generation"] + manifest_payload = manifest.model_dump(mode="json") + manifest_payload["metadata"]["document_generations"] = generations + manifest_payload["vector_generation"] = plan["generation"] + manifest = CorpusManifest.model_validate(manifest_payload) + write(context, "manifest.json", manifest.model_dump(mode="json")) + marker.unlink() + + def vector_stage(context: JobContext) -> None: + if context.dry_run: + return StageArtifacts() + rotate_compensated_generation(context) + values = records(context) + write(context, "vector-intent.json", { + "generation": read(context, "plan.json")["generation"], + "records": {value.record.id: value.content_hash for value in values}, + }) + seal_stage_artifacts( + context, "vector_upsert", + ("plan.json", "manifest.json", "vector-intent.json"), spec, + ) + try: + existing = self.vector_store.existing_hashes("evidence", ["evidence"]) + missing = [ + value for value in values + if existing.get(value.record.id) != value.content_hash + ] + if missing and self.vector_store.upsert("evidence", missing) != len(missing): + raise PipelineError("vector write count mismatch") + except Exception: + compensate(context) + raise + return StageArtifacts(("plan.json", "manifest.json", "vector-intent.json")) + + def stage_stage(context: JobContext) -> None: + if context.dry_run: + return StageArtifacts() + plan = read(context, "plan.json") + manifest = CorpusManifest.model_validate(read(context, "manifest.json")) + recovered = False + try: + if artifact(context, "compensated.json").exists(): + recovered = True + rotate_compensated_generation(context) + values = records(context) + existing = self.vector_store.existing_hashes("evidence", ["evidence"]) + missing = [value for value in values if existing.get(value.record.id) != value.content_hash] + if missing and self.vector_store.upsert("evidence", missing) != len(missing): + raise PipelineError("vector write count mismatch") + plan = read(context, "plan.json") + manifest = CorpusManifest.model_validate(read(context, "manifest.json")) + if not self.store.generation_path(plan["generation"]).exists(): + self.store.stage( + manifest, {doc.document_id: doc.content for doc in manifest.documents}, + generation=plan["generation"], + ) + self.store.manifest(plan["generation"]) + except Exception: + compensate(context) + raise + return StageArtifacts( + ("plan.json", "manifest.json", "vector-intent.json") if recovered else () + ) + + def publish_stage(context: JobContext) -> None: + if context.dry_run: + return StageArtifacts() + if artifact(context, "compensated.json").exists(): + rotate_compensated_generation(context) + values = records(context) + try: + existing = self.vector_store.existing_hashes("evidence", ["evidence"]) + missing = [value for value in values if existing.get(value.record.id) != value.content_hash] + if missing and self.vector_store.upsert("evidence", missing) != len(missing): + raise PipelineError("vector write count mismatch") + except Exception: + compensate(context) + raise + manifest = CorpusManifest.model_validate(read(context, "manifest.json")) + generation = read(context, "plan.json")["generation"] + try: + if not self.store.generation_path(generation).exists(): + self.store.stage( + manifest, {doc.document_id: doc.content for doc in manifest.documents}, + generation=generation, + ) + except Exception: + compensate(context) + raise + generation = read(context, "plan.json")["generation"] + try: + self.store.publish(generation) + except Exception: + compensate(context) + raise + return StageArtifacts(("plan.json", "manifest.json", "vector-intent.json")) + + def retention_stage(context: JobContext) -> None: + if not context.dry_run: + self.gc(workspace_root=workspace_root) + + report = run_job(spec, [ + discover_stage, acquire_stage, embed_stage, vector_stage, + stage_stage, publish_stage, retention_stage, + ], after_stage_return=after_stage_return) + run_dir = workspace_root / ".tht-jobs" / "evidence" / "runs" / report.run_id + plan = json.loads((run_dir / "artifacts" / "plan.json").read_text()) + if dry_run: + manifest = self.store.active_manifest() or CorpusManifest(pipeline_version=self.pipeline_version) + generation = None + published = False + elif report.status == "succeeded": + generation = plan["generation"] + manifest = self.store.manifest(generation) + published = True + else: + generation = plan["generation"] + manifest_path = run_dir / "artifacts" / "manifest.json" + manifest = (CorpusManifest.model_validate_json(manifest_path.read_text()) + if manifest_path.exists() else CorpusManifest(pipeline_version=self.pipeline_version)) + published = False + return PipelineResult( + report.status, generation, published, tuple(plan["changed"]), + tuple(plan["unchanged"]), tuple(plan["removed"]), manifest, + report.run_id, report.resumed_from, + ) + + def _run(self, *, dry_run: bool = False, resume: str | None = None) -> PipelineResult: + generation = None + vector_written = False + previous = self.store.active_manifest() + try: + discovered = self._discover() + except Exception as error: + raise PipelineError("Evidence discovery failed") from error + prior_documents = {doc.source_id: doc for doc in previous.documents} if previous else {} + fingerprints = {item.source_id: item.fingerprint for _, item in discovered} + compatibility = _fingerprint({ + "pipeline": self.pipeline_version, "model": self.embedding_model, + "dimensions": self.embedding_dimensions, "chunk_policy": asdict(self.chunk_policy), + }) + previous_compatibility = previous.metadata.get("compatibility_fingerprint") if previous else None + rebuild = previous is not None and compatibility != previous_compatibility + changed = tuple(item.source_id for _, item in discovered if rebuild or prior_documents.get(item.source_id) is None or prior_documents[item.source_id].source_fingerprint != item.fingerprint) + unchanged = tuple(item.source_id for _, item in discovered if item.source_id not in changed) + removed = tuple(sorted(set(prior_documents) - set(fingerprints))) + if dry_run: + manifest = previous or CorpusManifest(pipeline_version=self.pipeline_version) + return PipelineResult("succeeded", None, False, changed, unchanged, removed, manifest) + if previous is not None and not changed and not removed: + return PipelineResult( + "succeeded", previous.manifest_id, False, changed, unchanged, removed, previous + ) + + documents: list[CanonicalDocument] = [prior_documents[source_id] for source_id in unchanged] + changed_set = set(changed) + try: + for source, item in discovered: + if item.source_id in changed_set: + documents.append(normalize(source.acquire(item), self.pipeline_version)) + documents.sort(key=lambda document: document.source_id) + chunks: list[CanonicalChunk] = [] + for document in documents: + chunks.extend(chunk(document, self.chunk_policy)) + generation = resume or f"gen:{uuid.uuid4().hex}" + previous_generations = dict(previous.metadata.get("document_generations", {})) if previous else {} + document_generations = { + document.document_id: ( + generation if document.source_id in changed_set + else previous_generations.get(document.document_id, previous.vector_generation) + ) + for document in documents + } + manifest = CorpusManifest( + pipeline_version=self.pipeline_version, + embedding_model=self.embedding_model, + embedding_dimensions=self.embedding_dimensions, + vector_generation=generation, + documents=tuple(documents), chunks=tuple(chunks), + metadata={ + "workspace_id": self.workspace_id, + "compatibility_fingerprint": compatibility, + "fingerprints": fingerprints, + "removed": list(removed), + "document_generations": document_generations, + }, + ) + changed_documents = {document.document_id for document in documents if document.source_id in changed_set} + changed_chunks = [part for part in chunks if part.document_id in changed_documents] + embeddings = self.embedder.embed_documents([part.content for part in changed_chunks]) + if len(embeddings) != len(changed_chunks): + raise PipelineError("embedding count mismatch") + if any(len(vector) != self.embedding_dimensions for vector in embeddings): + raise PipelineError("embedding dimension mismatch") + records = [self._vector_record(part, vector, generation, self.workspace_id) for part, vector in zip(changed_chunks, embeddings, strict=True)] + if records: + written = self.vector_store.upsert("evidence", records) + vector_written = True + if written != len(records): + raise PipelineError("vector write count mismatch") + generation_path = self.store.generation_path(generation) + if resume is not None and generation_path.exists(): + staged_manifest = self.store.manifest(generation) + expected = manifest.model_dump(mode="json", exclude={"created_at", "manifest_id"}) + actual = staged_manifest.model_dump(mode="json", exclude={"created_at", "manifest_id"}) + actual["metadata"].pop("files", None) + if actual != expected: + raise PipelineError("resume generation is incompatible") + staged = generation + else: + staged = self.store.stage( + manifest, {document.document_id: document.content for document in documents}, + generation=generation, + ) + self.store.publish(staged) + self.gc(workspace_root=self.store.root.parent) + except PipelineError: + self._compensate(generation, vector_written) + raise + except Exception as error: + self._compensate(generation, vector_written) + raise PipelineError("Evidence preprocessing failed") from error + return PipelineResult("succeeded", generation, True, changed, unchanged, removed, self.store.manifest(generation)) + + def _compensate(self, generation: str | None, vector_written: bool) -> None: + if generation is None: + return + try: + self.store.discard(generation) + except Exception: + pass + if vector_written: + try: + self.vector_store.delete_generation("evidence", generation, self.workspace_id) + except Exception: + pass + + @staticmethod + def _vector_record( + chunk: CanonicalChunk, embedding: list[float], generation: str, workspace_id: str, + ): + record = VectorRecord( + id=f"{workspace_id}:{generation}:{chunk.chunk_id}", + kind="evidence", ref=chunk.document_id, + title=str(chunk.metadata.get("title", "")), content=chunk.content, + metadata={ + **dict(chunk.metadata), "document_id": chunk.document_id, + "workspace_id": workspace_id, + "source_uri": chunk.source_uri, "ordinal": chunk.ordinal, + "vector_generation": generation, + }, + ) + return VectorWriteRecord(record=record, embedding=embedding, content_hash=chunk.content_hash) diff --git a/harness/tht/corpus/store.py b/harness/tht/corpus/store.py new file mode 100644 index 00000000..0202e566 --- /dev/null +++ b/harness/tht/corpus/store.py @@ -0,0 +1,272 @@ +"""Durable immutable corpus generations and an atomic ACTIVE pointer.""" + +from __future__ import annotations + +import json +import fcntl +import os +import re +import stat +import shutil +import uuid +import hashlib +import threading +from datetime import UTC, datetime +from pathlib import Path +from contextlib import contextmanager + +from tht.corpus.models import CorpusManifest + + +_GENERATION = re.compile(r"^gen:[0-9a-f]{32}$") + + +class UnsafeCorpusPath(RuntimeError): + pass + + +def _atomic_write(path: Path, payload: bytes) -> None: + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") + fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, 0o600) + try: + with os.fdopen(fd, "wb") as stream: + stream.write(payload) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + directory = os.open(path.parent, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + os.fsync(directory) + finally: + os.close(directory) + except BaseException: + temporary.unlink(missing_ok=True) + raise + + +class CorpusStore: + def __init__(self, root: Path) -> None: + self.root = Path(root) + self.active_path = self.root / "ACTIVE" + self._replace = os.replace + self._fsync_directory = self._sync_root + self._lock_state = threading.local() + self._ensure_root() + + def _ensure_root(self) -> None: + if self.root.is_symlink(): + raise UnsafeCorpusPath("corpus root must not be a symlink") + self.root.mkdir(parents=True, exist_ok=True, mode=0o700) + info = self.root.lstat() + if not stat.S_ISDIR(info.st_mode) or info.st_uid != os.getuid(): + raise UnsafeCorpusPath("corpus root is unsafe") + + @contextmanager + def writer_lock(self): + depth = getattr(self._lock_state, "depth", 0) + if depth: + self._lock_state.depth = depth + 1 + try: + yield + finally: + self._lock_state.depth -= 1 + return + lock_path = self.root / ".writer.lock" + fd = os.open(lock_path, os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW | os.O_CLOEXEC, 0o600) + try: + info = os.fstat(fd) + if not stat.S_ISREG(info.st_mode) or info.st_uid != os.getuid() or info.st_nlink != 1: + raise UnsafeCorpusPath("corpus writer lock is unsafe") + fcntl.flock(fd, fcntl.LOCK_EX) + self._lock_state.depth = 1 + yield + finally: + self._lock_state.depth = 0 + fcntl.flock(fd, fcntl.LOCK_UN) + os.close(fd) + + def generation_path(self, generation: str) -> Path: + if not _GENERATION.fullmatch(generation): + raise UnsafeCorpusPath("invalid corpus generation") + path = self.root / generation.replace(":", "-") + if path.is_symlink(): + raise UnsafeCorpusPath("generation must not be a symlink") + return path + + def stage( + self, + manifest: CorpusManifest, + materialized: dict[str, str], + *, + generation: str | None = None, + ) -> str: + generation = generation or f"gen:{uuid.uuid4().hex}" + path = self.generation_path(generation) + try: + path.mkdir(mode=0o700) + except FileExistsError: + raise UnsafeCorpusPath("generation already exists") from None + documents = path / "documents" + documents.mkdir(mode=0o700) + files: dict[str, str] = {} + for document in manifest.documents: + relative = f"documents/{document.document_id.removeprefix('doc:')}.md" + _atomic_write(path / relative, materialized[document.document_id].encode("utf-8")) + files[document.document_id] = relative + payload = json.loads(manifest.model_dump_json()) + metadata = payload["metadata"] + metadata["files"] = files + payload.update({"manifest_id": generation, "metadata": metadata}) + staged = CorpusManifest.model_validate(payload) + _atomic_write(path / "manifest.json", (staged.model_dump_json(indent=2) + "\n").encode()) + return generation + + def publish(self, generation: str) -> str: + manifest = self.manifest(generation) + if manifest.manifest_id != generation: + raise UnsafeCorpusPath("manifest generation mismatch") + if self.active_generation() == generation: + return generation + previous = self.active_generation() + published_marker = self.generation_path(generation) / "PUBLISHED" + temporary = self.active_path.with_name(f".ACTIVE.{uuid.uuid4().hex}.tmp") + replaced = False + try: + _atomic_write(temporary, (generation + "\n").encode()) + self._replace(temporary, self.active_path) + replaced = True + self._fsync_directory() + _atomic_write( + published_marker, + (datetime.now(UTC).isoformat().replace("+00:00", "Z") + "\n").encode("ascii"), + ) + except BaseException: + temporary.unlink(missing_ok=True) + if replaced: + if previous is None: + self.active_path.unlink(missing_ok=True) + else: + rollback = self.active_path.with_name(f".ACTIVE.rollback.{uuid.uuid4().hex}.tmp") + _atomic_write(rollback, (previous + "\n").encode()) + self._replace(rollback, self.active_path) + self._sync_root() + raise + return generation + + def _sync_root(self) -> None: + directory = os.open(self.root, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + os.fsync(directory) + finally: + os.close(directory) + + def active_generation(self) -> str | None: + try: + if self.active_path.is_symlink(): + raise UnsafeCorpusPath("ACTIVE must not be a symlink") + value = self.active_path.read_text(encoding="ascii").strip() + except FileNotFoundError: + return None + if not _GENERATION.fullmatch(value): + raise UnsafeCorpusPath("ACTIVE contains an invalid generation") + return value + + def manifest(self, generation: str) -> CorpusManifest: + path = self.generation_path(generation) + manifest_path = path / "manifest.json" + if manifest_path.is_symlink(): + raise UnsafeCorpusPath("manifest must not be a symlink") + return CorpusManifest.model_validate_json(manifest_path.read_text(encoding="utf-8")) + + def discard(self, generation: str) -> None: + path = self.generation_path(generation) + if path.exists(): + if path.is_symlink() or not stat.S_ISDIR(path.lstat().st_mode): + raise UnsafeCorpusPath("generation cleanup target is unsafe") + shutil.rmtree(path) + + def active_manifest(self) -> CorpusManifest | None: + generation = self.active_generation() + return self.manifest(generation) if generation else None + + def list_generations(self) -> list[str]: + values = [] + for entry in self.root.iterdir(): + match = re.fullmatch(r"gen-([0-9a-f]{32})", entry.name) + if match and not entry.is_symlink() and stat.S_ISDIR(entry.lstat().st_mode): + values.append(f"gen:{match.group(1)}") + return sorted(values, key=lambda value: self.generation_path(value).stat().st_mtime_ns) + + def published_generations(self) -> list[str]: + active = self.active_generation() + published = [] + for generation in self.list_generations(): + path = self.generation_path(generation) + marker = path / "PUBLISHED" + if generation != active and not marker.is_file(): + continue + try: + manifest = self.manifest(generation) + if manifest.manifest_id != generation: + continue + timestamp = marker.read_text(encoding="ascii").strip() if marker.is_file() else "" + key = (timestamp or manifest.created_at.isoformat(), generation) + published.append((key, generation)) + except (OSError, ValueError): + continue + return [generation for _, generation in sorted(published)] + + def resolve_document(self, document_id: str, generation: str | None = None) -> Path | None: + generation = generation or self.active_generation() + if generation is None: + return None + manifest = self.manifest(generation) + relative = manifest.metadata.get("files", {}).get(document_id) + if not isinstance(relative, str): + return None + parts = Path(relative).parts + if Path(relative).is_absolute() or parts[:1] != ("documents",) or len(parts) != 2: + raise UnsafeCorpusPath("materialized document path is unsafe") + return self.generation_path(generation) / relative + + def read_document(self, document_id: str, generation: str | None = None) -> str | None: + generation = generation or self.active_generation() + if generation is None: + return None + manifest = self.manifest(generation) + path = self.resolve_document(document_id, generation) + document = next((item for item in manifest.documents if item.document_id == document_id), None) + if path is None or document is None: + return None + generation_fd = os.open(self.generation_path(generation), os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + documents_fd = fd = None + try: + documents_fd = os.open("documents", os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=generation_fd) + fd = os.open(path.name, os.O_RDONLY | os.O_NOFOLLOW | os.O_CLOEXEC, dir_fd=documents_fd) + info = os.fstat(fd) + if not stat.S_ISREG(info.st_mode) or info.st_uid != os.getuid() or info.st_nlink != 1: + raise UnsafeCorpusPath("materialized document is unsafe") + payload = os.read(fd, info.st_size + 1) + if len(payload) != info.st_size or "sha256:" + hashlib.sha256(payload).hexdigest() != document.content_hash: + raise UnsafeCorpusPath("materialized document hash mismatch") + return payload.decode("utf-8") + except (OSError, UnicodeError) as error: + raise UnsafeCorpusPath("materialized document read failed") from error + finally: + if fd is not None: + os.close(fd) + if documents_fd is not None: + os.close(documents_fd) + os.close(generation_fd) + + def materialize_document( + self, document_id: str, destination: Path, generation: str | None = None, + ) -> Path | None: + content = self.read_document(document_id, generation) + if content is None: + return None + destination = Path(destination) + destination.parent.mkdir(parents=True, exist_ok=True, mode=0o700) + _atomic_write(destination, content.encode("utf-8")) + destination.chmod(0o400) + return destination diff --git a/harness/tht/db/execute.py b/harness/tht/db/execute.py new file mode 100644 index 00000000..75fab3c6 --- /dev/null +++ b/harness/tht/db/execute.py @@ -0,0 +1,27 @@ +"""PostgreSQL execution operations used by the direct DWH adapter.""" + +from sqlalchemy import Engine + +from tht.execute import ( + ExecResult, + PlanSummary, + explain as _explain, + require_positive_int, + run_controlled, +) + +DEFAULT_TIMEOUT_MS = 30_000 + + +def run_query(engine: Engine, sql: str, *, limit: int, timeout_ms: int = DEFAULT_TIMEOUT_MS) -> ExecResult: + limit = require_positive_int(limit, name="limit") + return run_controlled( + engine, + sql, + limit=limit, + timeout_ms=timeout_ms, + ) + + +def explain(engine: Engine, sql: str, *, timeout_ms: int = DEFAULT_TIMEOUT_MS) -> PlanSummary: + return _explain(engine, sql, timeout_ms=timeout_ms) diff --git a/harness/tht/db/sampling.py b/harness/tht/db/sampling.py index 9fd916fb..ebc51bfb 100644 --- a/harness/tht/db/sampling.py +++ b/harness/tht/db/sampling.py @@ -4,11 +4,68 @@ from dataclasses import dataclass from sqlalchemy import Engine, text from tht.config import ExamplesConfig, LshConfig +from tht.execute import require_positive_int from tht.mschema.models import Annotations, PhysicalSchema +from tht.ports.dwh import DistinctValues logger = logging.getLogger(__name__) TEXT_TYPE_PREFIXES = ("text", "varchar", "character", "char") +DEFAULT_DISTINCT_VALUES_LIMIT = 1000 + + +def _quoted_top_values_query(engine: Engine, schema: str, table: str, column: str): + quote = engine.dialect.identifier_preparer.quote + identifier = quote(column) + return text( + f"SELECT {identifier} AS value FROM {quote(schema)}.{quote(table)} " + f"WHERE {identifier} IS NOT NULL GROUP BY {identifier} " + f"ORDER BY count(*) DESC, {identifier} LIMIT :lim" + ) + + +def sample_column( + engine: Engine, schema: str, table: str, column: str, *, limit: int +) -> list[object]: + limit = require_positive_int(limit, name="limit") + query = _quoted_top_values_query(engine, schema, table, column) + with engine.connect() as conn: + rows = conn.execute(query, {"lim": limit}).fetchall() + return [row[0] for row in rows] + + +def sample_column_rest( + client, schema: str, table: str, column: str, *, limit: int +) -> list[object]: + limit = require_positive_int(limit, name="limit") + rows = client.top_values(schema, table, column, limit) + return [row["value"] for row in rows if row.get("value") is not None] + + +def distinct_values( + engine: Engine, + schema: str, + table: str, + column: str, + *, + max_values: int = DEFAULT_DISTINCT_VALUES_LIMIT, +) -> DistinctValues: + max_values = require_positive_int(max_values, name="max_values") + values = sample_column(engine, schema, table, column, limit=max_values + 1) + return DistinctValues(values=values[:max_values], truncated=len(values) > max_values) + + +def distinct_values_rest( + client, + schema: str, + table: str, + column: str, + *, + max_values: int = DEFAULT_DISTINCT_VALUES_LIMIT, +) -> DistinctValues: + max_values = require_positive_int(max_values, name="max_values") + values = sample_column_rest(client, schema, table, column, limit=max_values + 1) + return DistinctValues(values=values[:max_values], truncated=len(values) > max_values) def is_text_type(pg_type: str) -> bool: diff --git a/harness/tht/execute/__init__.py b/harness/tht/execute/__init__.py index 6d3dd14c..57c17757 100644 --- a/harness/tht/execute/__init__.py +++ b/harness/tht/execute/__init__.py @@ -26,6 +26,13 @@ class PlanSummary: node_types: list[str] +def require_positive_int(value: object, *, name: str) -> int: + """Return a validated positive integer, excluding booleans and numeric lookalikes.""" + if type(value) is not int or value <= 0: + raise ValueError(f"{name} must be a positive integer") + return value + + def _inject_limit(sql: str, limit: int) -> tuple[str, bool]: """Aggiunge LIMIT limit+1 se assente (il +1 serve a rilevare il troncamento). Se la query ha gia' un suo LIMIT, lo si rispetta.""" diff --git a/harness/tht/jobs/__init__.py b/harness/tht/jobs/__init__.py new file mode 100644 index 00000000..ad825ac1 --- /dev/null +++ b/harness/tht/jobs/__init__.py @@ -0,0 +1,15 @@ +"""Shared execution envelope for resumable preprocessing jobs.""" + +from tht.jobs.locking import JobAlreadyRunningError, WorkspaceJobLock +from tht.jobs.models import JobReport, JobRun, JobSpec +from tht.jobs.runner import JobContext, run_job + +__all__ = [ + "JobAlreadyRunningError", + "JobContext", + "JobReport", + "JobRun", + "JobSpec", + "WorkspaceJobLock", + "run_job", +] diff --git a/harness/tht/jobs/dwh_pipeline.py b/harness/tht/jobs/dwh_pipeline.py new file mode 100644 index 00000000..a483a404 --- /dev/null +++ b/harness/tht/jobs/dwh_pipeline.py @@ -0,0 +1,1072 @@ +"""Crash-safe, resumable DWH catalog and LSH preprocessing stages.""" + +from __future__ import annotations + +import hashlib +import json +import fcntl +import atexit +import os +import re +import shutil +import stat +import tempfile +import uuid +from collections.abc import Callable +from dataclasses import dataclass +from pathlib import Path + +from tht.jobs.models import JobReport, JobSpec +from tht.jobs.runner import ( + CorruptCheckpointError, + JobContext, + StageArtifacts, + run_job, + seal_stage_artifacts, +) + + +DWH_STAGE_IDS = ("introspect", "lsh") +_RUN_ID = re.compile(r"^[0-9a-f]{32}$") +_SAFE_FILE = re.compile(r"^[A-Za-z0-9_-]+\.(?:pkl|json)$") +GENERATION_MANIFEST = "generation-manifest.json" +OWNER_MARKER = "OWNER.json" +_SNAPSHOT_DIRS: set[Path] = set() + + +def _cleanup_snapshot_dirs() -> None: + for path in tuple(_SNAPSHOT_DIRS): + shutil.rmtree(path, ignore_errors=True) + _SNAPSHOT_DIRS.discard(path) + + +atexit.register(_cleanup_snapshot_dirs) + + +def config_dwh_binding(cfg) -> dict[str, str]: + workspace_id = getattr(cfg, "_workspace_id", None) + config_source = getattr(cfg, "_config_source", None) + if not isinstance(workspace_id, str) or not isinstance(config_source, str): + raise CorruptCheckpointError("DWH workspace identity is unavailable; reload configuration") + return { + "workspace_id": workspace_id, + "config_fingerprint": fingerprint(cfg.model_dump_json()), + "input_fingerprint": fingerprint(config_source), + } + + +def _binding_digest(binding: dict[str, str]) -> str: + payload = json.dumps(binding, sort_keys=True, separators=(",", ":")) + return hashlib.sha256(payload.encode("utf-8")).hexdigest() + + +def _read_root_binding_fd(root_fd: int) -> dict[str, str]: + try: + fd = os.open(OWNER_MARKER, os.O_RDONLY | os.O_NOFOLLOW, dir_fd=root_fd) + try: + info = os.fstat(fd) + if ( + not stat.S_ISREG(info.st_mode) + or info.st_uid != os.getuid() + or info.st_nlink != 1 + or stat.S_IMODE(info.st_mode) != 0o400 + ): + raise OSError("unsafe DWH ownership marker") + chunks = [] + while chunk := os.read(fd, 1024 * 1024): + chunks.append(chunk) + finally: + os.close(fd) + payload = json.loads(b"".join(chunks).decode("utf-8")) + binding = payload["binding"] + if ( + payload.get("schema_version") != 1 + or not isinstance(binding, dict) + or set(binding) != {"workspace_id", "config_fingerprint", "input_fingerprint"} + or payload.get("binding_sha256") != _binding_digest(binding) + ): + raise ValueError + return binding + except (OSError, KeyError, TypeError, ValueError, UnicodeDecodeError) as error: + raise CorruptCheckpointError("DWH workspace ownership marker is missing or invalid") from error + + +def _validate_root_binding_fd(root_fd: int, expected: dict[str, str]) -> None: + if _read_root_binding_fd(root_fd) != expected: + raise CorruptCheckpointError("DWH artifacts belong to a different workspace configuration") + entries = set(os.listdir(root_fd)) + active_exists = "ACTIVE" in entries + try: + generations_fd = os.open( + "generations", os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=root_fd + ) + except FileNotFoundError: + generation_entries = [] + except OSError as error: + raise CorruptCheckpointError("DWH generations directory is invalid") from error + else: + try: + info = os.fstat(generations_fd) + if info.st_uid != os.getuid() or stat.S_IMODE(info.st_mode) != 0o700: + raise OSError("unsafe DWH generations directory") + generation_entries = os.listdir(generations_fd) + finally: + os.close(generations_fd) + if generation_entries and not active_exists: + raise CorruptCheckpointError("DWH generations exist without a consistent ACTIVE pointer") + + +def _claim_or_validate_root_binding( + root_fd: int, binding: dict[str, str] +) -> None: + temporary: str | None = None + try: + entries = set(os.listdir(root_fd)) + if OWNER_MARKER in entries: + _validate_root_binding_fd(root_fd, binding) + return + if not entries <= {"generation.lock", "generations"} or "generation.lock" not in entries: + raise CorruptCheckpointError( + "DWH artifacts are unbound; migrate them explicitly or use an empty root" + ) + if "generations" in entries: + generations_fd = os.open( + "generations", os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=root_fd + ) + try: + info = os.fstat(generations_fd) + if info.st_uid != os.getuid() or stat.S_IMODE(info.st_mode) != 0o700: + raise OSError("unsafe DWH generations directory") + if os.listdir(generations_fd): + raise CorruptCheckpointError( + "DWH artifacts are unbound; migrate them explicitly or use an empty root" + ) + finally: + os.close(generations_fd) + payload = { + "schema_version": 1, + "binding": binding, + "binding_sha256": _binding_digest(binding), + } + temporary = f".{OWNER_MARKER}.{uuid.uuid4().hex}.tmp" + fd = os.open( + temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=root_fd, + ) + try: + data = (json.dumps(payload, sort_keys=True, separators=(",", ":")) + "\n").encode() + offset = 0 + while offset < len(data): + offset += os.write(fd, data[offset:]) + os.fsync(fd) + os.fchmod(fd, 0o400) + os.fsync(fd) + finally: + os.close(fd) + os.replace(temporary, OWNER_MARKER, src_dir_fd=root_fd, dst_dir_fd=root_fd) + temporary = None + os.fsync(root_fd) + except OSError as error: + raise CorruptCheckpointError("DWH workspace root is invalid") from error + finally: + if temporary is not None: + try: + os.unlink(temporary, dir_fd=root_fd) + except FileNotFoundError: + pass + + +@dataclass(frozen=True) +class DwhArtifactSnapshot: + generation: str | None + physical: Path + lsh_dir: Path + _holder: object | None = None + + +@dataclass +class _GenerationLease: + root_fd: int + lock_fd: int + + def assert_root_identity(self, root: Path) -> None: + try: + path_info = os.stat(root, follow_symlinks=False) + opened_info = os.fstat(self.root_fd) + except OSError as error: + raise CorruptCheckpointError("DWH workspace root changed while locked") from error + if ( + not stat.S_ISDIR(path_info.st_mode) + or (path_info.st_dev, path_info.st_ino) + != (opened_info.st_dev, opened_info.st_ino) + ): + raise CorruptCheckpointError("DWH workspace root changed while locked") + + def close(self) -> None: + try: + fcntl.flock(self.lock_fd, fcntl.LOCK_UN) + finally: + try: + os.close(self.lock_fd) + finally: + os.close(self.root_fd) + + +class DwhSnapshotLease: + def __init__(self, cfg) -> None: + self.cfg = cfg + self._lease: _GenerationLease | None = None + self.snapshot: DwhArtifactSnapshot | None = None + + def __enter__(self) -> DwhArtifactSnapshot: + binding = config_dwh_binding(self.cfg) + self._lease = _acquire_existing_generation_lock( + self.cfg.paths.artifacts.parent, exclusive=False + ) + try: + self.snapshot = _resolve_dwh_snapshot_locked( + self.cfg, binding, self._lease.root_fd + ) + return self.snapshot + except BaseException: + self.__exit__(None, None, None) + raise + + def __exit__(self, *_args) -> None: + if self._lease is not None: + lease, self._lease = self._lease, None + lease.close() + if self.snapshot is not None and isinstance(self.snapshot._holder, Path): + shutil.rmtree(self.snapshot._holder, ignore_errors=True) + _SNAPSHOT_DIRS.discard(self.snapshot._holder) + + +def lease_dwh_snapshot(cfg) -> DwhSnapshotLease: + return DwhSnapshotLease(cfg) + + +def _acquire_generation_lock(workspace_root: Path, *, exclusive: bool) -> _GenerationLease: + root = workspace_root / ".tht-dwh" + DwhPreprocessPipeline._ensure_owned_dir(root) + root_fd = os.open(root, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + fd = os.open( + "generation.lock", os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW, + 0o600, dir_fd=root_fd, + ) + except BaseException: + os.close(root_fd) + raise + try: + info = os.fstat(fd) + if ( + not stat.S_ISREG(info.st_mode) + or info.st_uid != os.getuid() + or info.st_nlink != 1 + or stat.S_IMODE(info.st_mode) != 0o600 + ): + raise OSError("unsafe DWH generation lock") + fcntl.flock(fd, fcntl.LOCK_EX if exclusive else fcntl.LOCK_SH) + return _GenerationLease(root_fd, fd) + except BaseException: + os.close(fd) + os.close(root_fd) + raise + + +def _acquire_existing_generation_lock( + workspace_root: Path, *, exclusive: bool +) -> _GenerationLease: + root = workspace_root / ".tht-dwh" + try: + root_fd = os.open(root, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + root_info = os.fstat(root_fd) + if root_info.st_uid != os.getuid() or stat.S_IMODE(root_info.st_mode) != 0o700: + raise OSError("unsafe DWH workspace root") + fd = os.open("generation.lock", os.O_RDWR | os.O_NOFOLLOW, dir_fd=root_fd) + except BaseException: + os.close(root_fd) + raise + try: + info = os.fstat(fd) + if ( + not stat.S_ISREG(info.st_mode) + or info.st_uid != os.getuid() + or info.st_nlink != 1 + or stat.S_IMODE(info.st_mode) != 0o600 + ): + raise OSError("unsafe DWH generation lock") + fcntl.flock(fd, fcntl.LOCK_EX if exclusive else fcntl.LOCK_SH) + return _GenerationLease(root_fd, fd) + except BaseException: + os.close(fd) + os.close(root_fd) + raise + except OSError as error: + raise CorruptCheckpointError( + "DWH workspace ownership is not initialized; run preprocessing first" + ) from error + + +def _digest(path: Path) -> str: + return hashlib.sha256(path.read_bytes()).hexdigest() + + +def _read_owned_at(directory_fd: int, name: str, *, readonly: bool) -> bytes: + fd = os.open(name, os.O_RDONLY | os.O_NOFOLLOW, dir_fd=directory_fd) + try: + info = os.fstat(fd) + if ( + not stat.S_ISREG(info.st_mode) + or info.st_uid != os.getuid() + or info.st_nlink != 1 + or (readonly and bool(info.st_mode & 0o222)) + ): + raise OSError("unsafe DWH generation file") + chunks = [] + while chunk := os.read(fd, 1024 * 1024): + chunks.append(chunk) + return b"".join(chunks) + finally: + os.close(fd) + + +def _open_generations_fd(root_fd: int, *, create: bool = False) -> int: + if create: + try: + os.mkdir("generations", 0o700, dir_fd=root_fd) + os.fsync(root_fd) + except FileExistsError: + pass + fd = os.open( + "generations", os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=root_fd + ) + info = os.fstat(fd) + if ( + not stat.S_ISDIR(info.st_mode) + or info.st_uid != os.getuid() + or stat.S_IMODE(info.st_mode) != 0o700 + ): + os.close(fd) + raise CorruptCheckpointError("DWH generations directory is invalid") + return fd + + +def _open_generation_fd(generations_fd: int, generation: str) -> int: + if not _RUN_ID.fullmatch(generation): + raise CorruptCheckpointError("DWH generation identity is invalid") + try: + return os.open( + generation, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=generations_fd, + ) + except OSError as error: + raise CorruptCheckpointError("active DWH generation is missing") from error + + +def _active_generation_fd( + root_fd: int, expected_binding: dict[str, str], *, validate: bool = True +) -> tuple[str, int] | None: + _validate_root_binding_fd(root_fd, expected_binding) + try: + generation = _read_owned_at(root_fd, "ACTIVE", readonly=False).decode().strip() + except FileNotFoundError: + try: + generations_fd = _open_generations_fd(root_fd) + except FileNotFoundError: + return None + try: + if os.listdir(generations_fd): + raise CorruptCheckpointError( + "DWH generations exist without a consistent ACTIVE pointer" + ) + finally: + os.close(generations_fd) + return None + except (OSError, UnicodeDecodeError) as error: + raise CorruptCheckpointError("DWH ACTIVE pointer is invalid") from error + if not _RUN_ID.fullmatch(generation): + raise CorruptCheckpointError("DWH ACTIVE pointer is invalid") + generations_fd = _open_generations_fd(root_fd) + try: + generation_fd = _open_generation_fd(generations_fd, generation) + finally: + os.close(generations_fd) + try: + if validate: + validate_generation_fd(generation_fd, generation, expected_binding) + return generation, generation_fd + except BaseException: + os.close(generation_fd) + raise + + +def _materialize_generation_fd( + generation_fd: int, generation: str, binding: dict[str, str] +) -> tuple[Path, Path]: + holder = Path(tempfile.mkdtemp(prefix="tht-dwh-snapshot-")) + _SNAPSHOT_DIRS.add(holder) + snapshot_root = holder + try: + _, payloads = _read_validated_generation_fd(generation_fd, generation, binding) + for name, payload in payloads.items(): + destination = snapshot_root / name + destination.write_bytes(payload) + destination.chmod(0o400) + return holder, snapshot_root + except BaseException: + shutil.rmtree(holder, ignore_errors=True) + _SNAPSHOT_DIRS.discard(holder) + raise + + +def validate_generation_fd( + directory_fd: int, generation: str, expected_binding: dict[str, str] | None = None, +) -> dict: + manifest, _ = _read_validated_generation_fd( + directory_fd, generation, expected_binding + ) + return manifest + + +def _read_validated_generation_fd( + directory_fd: int, generation: str, expected_binding: dict[str, str] | None = None, +) -> tuple[dict, dict[str, bytes]]: + try: + directory_info = os.fstat(directory_fd) + if ( + not stat.S_ISDIR(directory_info.st_mode) + or directory_info.st_uid != os.getuid() + or stat.S_IMODE(directory_info.st_mode) != 0o700 + ): + raise ValueError + manifest = json.loads( + _read_owned_at(directory_fd, GENERATION_MANIFEST, readonly=True).decode("utf-8") + ) + files = manifest["files"] + if ( + manifest["generation"] != generation + or not isinstance(files, dict) + or not re.fullmatch(r"sha256:[0-9a-f]{64}", manifest["job_spec_fingerprint"]) + or not re.fullmatch(r"[0-9a-f]{64}", manifest["artifact_manifest_sha256"]) + or not re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", manifest["workspace_id"]) + or not re.fullmatch(r"sha256:[0-9a-f]{64}", manifest["config_fingerprint"]) + or not re.fullmatch(r"sha256:[0-9a-f]{64}", manifest["input_fingerprint"]) + ): + raise ValueError + if expected_binding is not None and any( + manifest.get(key) != value for key, value in expected_binding.items() + ): + raise CorruptCheckpointError( + "DWH artifacts belong to a different workspace configuration" + ) + if set(os.listdir(directory_fd)) != set(files) | {GENERATION_MANIFEST}: + raise ValueError + payloads = {} + for name, expected in files.items(): + if name != "physical.yaml" and not _SAFE_FILE.fullmatch(name): + raise ValueError + payload = _read_owned_at(directory_fd, name, readonly=True) + if hashlib.sha256(payload).hexdigest() != expected: + raise ValueError + payloads[name] = payload + return manifest, payloads + except (OSError, KeyError, TypeError, ValueError, UnicodeDecodeError, json.JSONDecodeError) as error: + raise CorruptCheckpointError("published DWH generation is invalid") from error + + +def resolve_dwh_snapshot(cfg) -> DwhArtifactSnapshot: + binding = config_dwh_binding(cfg) + lease = _acquire_existing_generation_lock( + cfg.paths.artifacts.parent, exclusive=False + ) + try: + return _resolve_dwh_snapshot_locked(cfg, binding, lease.root_fd) + finally: + lease.close() + + +def _resolve_dwh_snapshot_locked( + cfg, binding: dict[str, str], root_fd: int, +) -> DwhArtifactSnapshot: + active = _active_generation_fd(root_fd, binding, validate=False) + if active is None: + return DwhArtifactSnapshot( + None, cfg.paths.artifacts / "mschema" / "physical.yaml", cfg.paths.indexes / "lsh" + ) + generation, generation_fd = active + try: + holder, snapshot_root = _materialize_generation_fd( + generation_fd, generation, binding + ) + finally: + os.close(generation_fd) + return DwhArtifactSnapshot( + generation, snapshot_root / "physical.yaml", snapshot_root, holder + ) + + +def active_generation_dir( + workspace_root: Path, expected_binding: dict[str, str] +) -> Path | None: + lease = _acquire_existing_generation_lock(workspace_root, exclusive=False) + try: + active = _active_generation_fd(lease.root_fd, expected_binding) + if active is None: + return None + generation, generation_fd = active + os.close(generation_fd) + lease.assert_root_identity(workspace_root / ".tht-dwh") + return workspace_root / ".tht-dwh" / "generations" / generation + finally: + lease.close() + + +class DwhPreprocessPipeline: + """Stage a complete artifact bundle, then publish it through one atomic pointer.""" + + def __init__( + self, + *, + workspace_id: str, + workspace_root: Path, + config_fingerprint: str, + input_fingerprint: str, + introspect: Callable[[Path], object], + build_lsh: Callable[[Path, Path], object], + lsh_filenames: tuple[str, str, str] | None = None, + current_physical: Path | None = None, + current_lsh_dir: Path | None = None, + after_publish: Callable[[str], object] | None = None, + retain_generations: int = 3, + ) -> None: + self.workspace_id = workspace_id + self.workspace_root = workspace_root + self.config_fingerprint = config_fingerprint + self.input_fingerprint = input_fingerprint + self.introspect = introspect + self.build_lsh = build_lsh + self.lsh_filenames = lsh_filenames or ( + f"{workspace_id}_lsh.pkl", f"{workspace_id}_minhashes.pkl", + f"{workspace_id}_meta.json", + ) + self.current_physical = current_physical + self.current_lsh_dir = current_lsh_dir + self._snapshot_holder = None + self.after_publish = after_publish + if isinstance(retain_generations, bool) or retain_generations < 1: + raise ValueError("retain_generations must be positive") + self.retain_generations = retain_generations + if len(set(self.lsh_filenames)) != 3 or any( + not _SAFE_FILE.fullmatch(name) for name in self.lsh_filenames + ): + raise ValueError("LSH filenames must be unique flat safe names") + + @property + def binding(self) -> dict[str, str]: + return { + "workspace_id": self.workspace_id, + "config_fingerprint": self.config_fingerprint, + "input_fingerprint": self.input_fingerprint, + } + + def _assert_active_binding(self, root_fd: int) -> None: + _validate_root_binding_fd(root_fd, self.binding) + active = _active_generation_fd(root_fd, self.binding) + if active is not None: + _, generation_fd = active + os.close(generation_fd) + + def run( + self, steps: tuple[str, ...] = DWH_STAGE_IDS, *, resume_run_id: str | None = None + ) -> JobReport: + self._validate_steps(steps) + lease = _acquire_generation_lock(self.workspace_root, exclusive=True) + try: + self._assert_no_legacy_artifacts(lease.root_fd) + _claim_or_validate_root_binding(lease.root_fd, self.binding) + lease.assert_root_identity(self.workspace_root / ".tht-dwh") + self._assert_active_binding(lease.root_fd) + active = _active_generation_fd(lease.root_fd, self.binding, validate=False) + if active is not None: + generation, generation_fd = active + try: + self._snapshot_holder, snapshot_root = _materialize_generation_fd( + generation_fd, generation, self.binding + ) + finally: + os.close(generation_fd) + self.current_physical = snapshot_root / "physical.yaml" + self.current_lsh_dir = snapshot_root + finally: + lease.close() + try: + if resume_run_id is not None: + self._validate_resume_publication(resume_run_id) + spec = JobSpec( + workspace_id=self.workspace_id, + job_type="dwh", + workspace_root=self.workspace_root, + spec_version="jobs-v1", + pipeline_version="dwh-v2", + config_fingerprint=self.config_fingerprint, + input_fingerprint=self.input_fingerprint, + stage_ids=steps, + resume_run_id=resume_run_id, + ) + except BaseException: + self._release_pipeline_snapshot() + raise + + def introspect_stage(context: JobContext): + artifacts = self._artifacts(context) + physical = artifacts / "physical.yaml" + try: + self.introspect(physical) + except BaseException: + physical.unlink(missing_ok=True) + raise + if steps[-1] == "introspect": + self._copy_current_lsh(artifacts) + return self._publish_stage(context, "introspect", spec) + return StageArtifacts(("physical.yaml",)) + + def lsh_stage(context: JobContext): + artifacts = self._artifacts(context) + physical = artifacts / "physical.yaml" + if not physical.exists(): + source = self.current_physical + if source is None or not source.is_file(): + raise FileNotFoundError("physical catalog is missing") + shutil.copyfile(source, physical) + try: + self.build_lsh(physical, artifacts) + except BaseException: + for name in self.lsh_filenames: + (artifacts / name).unlink(missing_ok=True) + raise + return self._publish_stage(context, "lsh", spec) + + implementations = {"introspect": introspect_stage, "lsh": lsh_stage} + try: + return run_job( + spec, tuple(implementations[step] for step in steps), + reconcile_effects=self._reconcile_effects, + ) + finally: + self._release_pipeline_snapshot() + + def _release_pipeline_snapshot(self) -> None: + holder, self._snapshot_holder = self._snapshot_holder, None + if isinstance(holder, Path): + shutil.rmtree(holder, ignore_errors=True) + _SNAPSHOT_DIRS.discard(holder) + if self.current_physical is not None and holder in self.current_physical.parents: + self.current_physical = None + if self.current_lsh_dir is not None and ( + self.current_lsh_dir == holder or holder in self.current_lsh_dir.parents + ): + self.current_lsh_dir = None + + def _reconcile_effects(self, source, run_dir: Path) -> set[str]: + running = next( + (stage for stage in source.stages if stage.status == "running" and stage.effect_state == "intent"), + None, + ) + if running is None: + return set() + lease = _acquire_existing_generation_lock(self.workspace_root, exclusive=True) + try: + generations_fd = _open_generations_fd(lease.root_fd) + try: + self._validate_published_fd( + generations_fd, source.run_id, + run_dir / "artifacts", running.artifact_files, + ) + active = _active_generation_fd(lease.root_fd, self.binding) + try: + if active is None or active[0] != source.run_id: + raise CorruptCheckpointError("sealed DWH publication is not ACTIVE") + finally: + if active is not None: + os.close(active[1]) + finally: + os.close(generations_fd) + finally: + lease.close() + return {running.name} + + def _validate_resume_publication(self, run_id: str) -> None: + run_dir = self.workspace_root / ".tht-jobs" / "dwh" / "runs" / run_id + try: + checkpoint = json.loads((run_dir / "checkpoint.json").read_text(encoding="utf-8")) + except (OSError, ValueError, TypeError) as error: + raise CorruptCheckpointError("checkpoint is invalid and cannot be resumed") from error + lease = _acquire_existing_generation_lock(self.workspace_root, exclusive=True) + try: + try: + generations_fd = _open_generations_fd(lease.root_fd) + except FileNotFoundError: + if checkpoint.get("status") == "succeeded": + raise CorruptCheckpointError("published DWH generation is missing") + return + try: + if run_id not in os.listdir(generations_fd): + if checkpoint.get("status") == "succeeded": + raise CorruptCheckpointError("published DWH generation is missing") + return + try: + manifest = json.loads( + (run_dir / "artifacts" / "artifact-manifest.json").read_text( + encoding="utf-8" + ) + ) + required = tuple( + name + for stage in manifest["stages"].values() + for name in stage["required"] + ) + except (OSError, KeyError, ValueError, TypeError) as error: + raise CorruptCheckpointError( + "resume artifact manifest is invalid" + ) from error + self._validate_published_fd( + generations_fd, run_id, run_dir / "artifacts", required + ) + finally: + os.close(generations_fd) + finally: + lease.close() + + def _assert_no_legacy_artifacts(self, root_fd: int) -> None: + if OWNER_MARKER in os.listdir(root_fd): + return + legacy_physical = self.current_physical is not None and ( + self.current_physical.exists() or self.current_physical.is_symlink() + ) + legacy_lsh = self.current_lsh_dir is not None and any( + (self.current_lsh_dir / name).exists() + or (self.current_lsh_dir / name).is_symlink() + for name in self.lsh_filenames + ) + if legacy_physical or legacy_lsh: + raise CorruptCheckpointError( + "DWH legacy artifacts are unbound; migrate them explicitly or use an empty root" + ) + + @staticmethod + def _artifacts(context: JobContext) -> Path: + root = context.run_dir / "artifacts" + root.mkdir(exist_ok=True) + return root + + def _required(self, artifacts: Path) -> tuple[str, ...]: + required = ("physical.yaml",) + tuple( + name for name in self.lsh_filenames if (artifacts / name).is_file() + ) + if not (artifacts / "physical.yaml").is_file(): + raise CorruptCheckpointError("staged physical catalog is missing") + lsh_count = len(required) - 1 + if lsh_count not in (0, len(self.lsh_filenames)): + raise CorruptCheckpointError("staged LSH artifact set is incomplete") + return required + + def _copy_current_lsh(self, artifacts: Path) -> None: + if self.current_lsh_dir is None: + return + existing = [self.current_lsh_dir / name for name in self.lsh_filenames] + if not any(path.exists() for path in existing): + return + if not all(path.is_file() for path in existing): + raise CorruptCheckpointError("current LSH artifact set is incomplete") + for source in existing: + shutil.copyfile(source, artifacts / source.name) + + def _publish_stage(self, context: JobContext, stage: str, spec: JobSpec): + artifacts = self._artifacts(context) + required = self._required(artifacts) + seal_stage_artifacts(context, stage, required, spec) + lease = _acquire_generation_lock(self.workspace_root, exclusive=True) + try: + _validate_root_binding_fd(lease.root_fd, self.binding) + self._assert_active_binding(lease.root_fd) + self._publish(context.run_id, artifacts, required, lease.root_fd) + if self.after_publish is not None: + self.after_publish(context.run_id) + self._cleanup_generations_fd(lease.root_fd) + finally: + lease.close() + return StageArtifacts(required) + + def _publish( + self, generation: str, artifacts: Path, required: tuple[str, ...], root_fd: int + ) -> None: + generations_fd = _open_generations_fd(root_fd, create=True) + temporary = f".{generation}.{uuid.uuid4().hex}.tmp" + try: + if generation in os.listdir(generations_fd): + self._validate_published_fd(generations_fd, generation, artifacts, required) + else: + os.mkdir(temporary, 0o700, dir_fd=generations_fd) + temporary_fd = os.open( + temporary, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=generations_fd, + ) + try: + for name in required: + self._write_readonly_at( + temporary_fd, name, (artifacts / name).read_bytes() + ) + artifact_manifest = artifacts / "artifact-manifest.json" + manifest = { + "schema_version": 1, + "generation": generation, + "files": {name: _digest(artifacts / name) for name in required}, + "job_spec_fingerprint": json.loads( + artifact_manifest.read_text(encoding="utf-8") + )["spec_fingerprint"], + "artifact_manifest_sha256": _digest(artifact_manifest), + **self.binding, + } + self._write_readonly_at( + temporary_fd, GENERATION_MANIFEST, + (json.dumps(manifest, sort_keys=True, separators=(",", ":")) + "\n").encode(), + ) + os.fsync(temporary_fd) + finally: + os.close(temporary_fd) + os.rename( + temporary, generation, + src_dir_fd=generations_fd, dst_dir_fd=generations_fd, + ) + temporary = "" + os.fsync(generations_fd) + previous = None + try: + previous = _read_owned_at(root_fd, "ACTIVE", readonly=False) + except FileNotFoundError: + pass + self._replace_active_at(root_fd, (generation + "\n").encode(), previous) + finally: + if temporary: + self._safe_delete_generation(generations_fd, temporary, allow_temporary=True) + os.close(generations_fd) + + @staticmethod + def _write_readonly_at(directory_fd: int, name: str, payload: bytes) -> None: + fd = os.open( + name, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=directory_fd, + ) + try: + offset = 0 + while offset < len(payload): + offset += os.write(fd, payload[offset:]) + os.fsync(fd) + os.fchmod(fd, 0o400) + os.fsync(fd) + finally: + os.close(fd) + + def _replace_active_at(self, root_fd: int, payload: bytes, previous: bytes | None) -> None: + temporary = f".ACTIVE.{uuid.uuid4().hex}.tmp" + fd = os.open( + temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=root_fd, + ) + try: + offset = 0 + while offset < len(payload): + offset += os.write(fd, payload[offset:]) + os.fsync(fd) + finally: + os.close(fd) + try: + os.replace(temporary, "ACTIVE", src_dir_fd=root_fd, dst_dir_fd=root_fd) + temporary = "" + try: + os.fsync(root_fd) + except BaseException: + self._restore_pointer_at(root_fd, previous) + raise + finally: + if temporary: + try: + os.unlink(temporary, dir_fd=root_fd) + except FileNotFoundError: + pass + + def _validate_published_fd( + self, generations_fd: int, generation: str, + artifacts: Path, required: tuple[str, ...], + ) -> None: + generation_fd = _open_generation_fd(generations_fd, generation) + try: + manifest = validate_generation_fd(generation_fd, generation, self.binding) + if set(manifest["files"]) != set(required): + raise CorruptCheckpointError("published DWH generation is invalid") + artifact_digest = _digest(artifacts / "artifact-manifest.json") + if manifest["artifact_manifest_sha256"] != artifact_digest: + raise CorruptCheckpointError("published DWH job manifest digest mismatch") + for name in required: + if hashlib.sha256((artifacts / name).read_bytes()).hexdigest() != manifest["files"][name]: + raise CorruptCheckpointError("published DWH artifact digest mismatch") + finally: + os.close(generation_fd) + + def _restore_pointer_at(self, root_fd: int, previous: bytes | None) -> None: + if previous is None: + try: + os.unlink("ACTIVE", dir_fd=root_fd) + except FileNotFoundError: + pass + else: + restore = f".ACTIVE.restore.{uuid.uuid4().hex}.tmp" + fd = os.open( + restore, os.O_WRONLY | os.O_CREAT | os.O_EXCL | os.O_NOFOLLOW, + 0o600, dir_fd=root_fd, + ) + try: + offset = 0 + while offset < len(previous): + offset += os.write(fd, previous[offset:]) + os.fsync(fd) + finally: + os.close(fd) + os.replace(restore, "ACTIVE", src_dir_fd=root_fd, dst_dir_fd=root_fd) + os.fsync(root_fd) + + def _cleanup_generations(self) -> None: + lease = _acquire_existing_generation_lock(self.workspace_root, exclusive=True) + try: + self._cleanup_generations_fd(lease.root_fd) + finally: + lease.close() + + def _cleanup_generations_fd(self, root_fd: int) -> None: + _validate_root_binding_fd(root_fd, self.binding) + active = _active_generation_fd(root_fd, self.binding) + active_name = active[0] if active else None + if active is not None: + os.close(active[1]) + protected = {active_name} if active_name else set() + runs = self.workspace_root / ".tht-jobs" / "dwh" / "runs" + for checkpoint in runs.glob("*/checkpoint.json") if runs.exists() else (): + try: + status = json.loads(checkpoint.read_text(encoding="utf-8"))["status"] + if status in {"running", "failed"}: + protected.add(checkpoint.parent.name) + except (OSError, KeyError, ValueError): + continue + generations = [] + try: + generations_fd = _open_generations_fd(root_fd) + except FileNotFoundError: + return + try: + for name in os.listdir(generations_fd): + if not _RUN_ID.fullmatch(name): + continue + try: + candidate_fd = os.open( + name, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=generations_fd, + ) + except OSError: + continue + try: + info = os.fstat(candidate_fd) + validate_generation_fd(candidate_fd, name, self.binding) + except CorruptCheckpointError: + continue + finally: + os.close(candidate_fd) + generations.append((info.st_mtime_ns, name)) + generations.sort(key=lambda value: (value[0], value[1])) + rollback = [value for value in generations if value[1] != active_name] + rollback_count = self.retain_generations - 1 + keep_recent = { + name for _, name in (rollback[-rollback_count:] if rollback_count else ()) + } + for _, name in generations: + if name in protected | keep_recent: + continue + self._safe_delete_generation(generations_fd, name) + os.fsync(generations_fd) + finally: + os.close(generations_fd) + + @staticmethod + def _safe_delete_generation( + root_fd: int, name: str, *, allow_temporary: bool = False + ) -> None: + if not (_RUN_ID.fullmatch(name) or (allow_temporary and name.startswith("."))): + return + try: + generation_fd = os.open( + name, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=root_fd + ) + except OSError: + return + try: + info = os.fstat(generation_fd) + if not stat.S_ISDIR(info.st_mode) or info.st_uid != os.getuid(): + return + entries = os.listdir(generation_fd) + opened = [] + try: + for child in entries: + try: + fd = os.open( + child, os.O_RDONLY | os.O_NOFOLLOW, dir_fd=generation_fd + ) + except OSError: + return + child_info = os.fstat(fd) + if ( + not stat.S_ISREG(child_info.st_mode) + or child_info.st_uid != os.getuid() + or child_info.st_nlink != 1 + ): + os.close(fd) + return + opened.append((child, fd)) + for child, fd in opened: + os.fchmod(fd, 0o600) + os.close(fd) + os.unlink(child, dir_fd=generation_fd) + opened.clear() + finally: + for _, fd in opened: + os.close(fd) + finally: + os.close(generation_fd) + try: + os.rmdir(name, dir_fd=root_fd) + except OSError: + return + + @staticmethod + def _ensure_owned_dir(path: Path) -> None: + try: + path.mkdir(mode=0o700) + except FileExistsError: + pass + info = path.lstat() + if not stat.S_ISDIR(info.st_mode) or info.st_uid != os.getuid(): + raise OSError("unsafe DWH publication directory") + path.chmod(0o700) + + @staticmethod + def _validate_steps(steps: tuple[str, ...]) -> None: + if not steps or len(steps) != len(set(steps)) or any( + step not in DWH_STAGE_IDS for step in steps + ): + raise ValueError("DWH preprocessing steps must be unique introspect/lsh stages") + if tuple(sorted(steps, key=DWH_STAGE_IDS.index)) != steps: + raise ValueError("DWH preprocessing steps must follow introspect,lsh order") + + +def fingerprint(value: str) -> str: + return "sha256:" + hashlib.sha256(value.encode("utf-8")).hexdigest() diff --git a/harness/tht/jobs/locking.py b/harness/tht/jobs/locking.py new file mode 100644 index 00000000..26ed43d1 --- /dev/null +++ b/harness/tht/jobs/locking.py @@ -0,0 +1,115 @@ +"""Crash-safe interprocess locking scoped by workspace and job type.""" + +from __future__ import annotations + +import fcntl +import hashlib +import os +import re +import stat +from pathlib import Path +from types import TracebackType + + +class JobAlreadyRunningError(RuntimeError): + pass + + +_JOB_KEY = re.compile(r"^[a-z][a-z0-9_-]{0,63}$") + + +def _lock_name(workspace_id: str, job_type: str) -> str: + if not _JOB_KEY.fullmatch(workspace_id) or not _JOB_KEY.fullmatch(job_type): + raise ValueError("lock identifiers must be lowercase filesystem-safe keys") + workspace_key = hashlib.sha256(workspace_id.encode("utf-8")).hexdigest()[:16] + return f"{workspace_key}-{job_type}.lock" + + +class WorkspaceJobLock: + """Advisory kernel lock; the inode remains stable and is never deleted by PID.""" + + def __init__(self, workspace_root: Path, workspace_id: str, job_type: str) -> None: + self.path = workspace_root / ".tht-jobs" / ".locks" / _lock_name( + workspace_id, job_type + ) + self._fd: int | None = None + + def acquire(self) -> "WorkspaceJobLock": + if self._fd is not None: + raise RuntimeError("job lock is already held by this object") + root_fd = os.open(self.path.parents[2], os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + jobs_fd = _open_owned_directory(root_fd, ".tht-jobs") + try: + locks_fd = _open_owned_directory(jobs_fd, ".locks") + try: + fd = os.open( + self.path.name, + os.O_RDWR | os.O_CREAT | os.O_NOFOLLOW | os.O_CLOEXEC, + 0o600, + dir_fd=locks_fd, + ) + try: + info = os.fstat(fd) + if ( + not stat.S_ISREG(info.st_mode) + or info.st_uid != os.getuid() + or info.st_nlink != 1 + ): + raise OSError("unsafe job lock file") + os.fchmod(fd, 0o600) + try: + fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + except BlockingIOError as error: + raise JobAlreadyRunningError( + "this workspace job is already running" + ) from error + except BaseException: + os.close(fd) + raise + finally: + os.close(locks_fd) + finally: + os.close(jobs_fd) + finally: + os.close(root_fd) + self._fd = fd + return self + + def release(self) -> None: + if self._fd is None: + return + fd, self._fd = self._fd, None + try: + fcntl.flock(fd, fcntl.LOCK_UN) + finally: + os.close(fd) + + def __enter__(self) -> "WorkspaceJobLock": + return self.acquire() + + def __exit__( + self, + exc_type: type[BaseException] | None, + exc: BaseException | None, + traceback: TracebackType | None, + ) -> None: + self.release() + + +def _open_owned_directory(parent_fd: int, name: str) -> int: + try: + os.mkdir(name, 0o700, dir_fd=parent_fd) + os.fsync(parent_fd) + except FileExistsError: + pass + fd = os.open(name, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, dir_fd=parent_fd) + try: + info = os.fstat(fd) + if not stat.S_ISDIR(info.st_mode) or info.st_uid != os.getuid(): + raise OSError("unsafe job lock directory") + os.fchmod(fd, 0o700) + except BaseException: + os.close(fd) + raise + return fd diff --git a/harness/tht/jobs/models.py b/harness/tht/jobs/models.py new file mode 100644 index 00000000..e85cd8a1 --- /dev/null +++ b/harness/tht/jobs/models.py @@ -0,0 +1,213 @@ +"""Immutable, secret-free records for preprocessing execution.""" + +from __future__ import annotations + +import re +from datetime import UTC, datetime +from pathlib import Path +from typing import Literal, Self + +from pydantic import BaseModel, ConfigDict, Field, field_serializer, field_validator, model_validator + + +_JOB_KEY = re.compile(r"^[a-z][a-z0-9_-]{0,63}$") +_RUN_ID = re.compile(r"^[0-9a-f]{32}$") +_FINGERPRINT = re.compile(r"^sha256:[0-9a-f]{64}$") +JobStatus = Literal["pending", "running", "succeeded", "failed"] +StageStatus = Literal["pending", "running", "succeeded", "failed"] +EffectState = Literal["intent", "completed"] + + +def utc_now() -> datetime: + return datetime.now(UTC) + + +def _validate_job_key(value: str) -> str: + if not _JOB_KEY.fullmatch(value): + raise ValueError("job identifiers must be lowercase filesystem-safe keys") + return value + + +def _validate_run_id(value: str | None) -> str | None: + if value is not None and not _RUN_ID.fullmatch(value): + raise ValueError("run id must contain 32 lowercase hexadecimal characters") + return value + + +class _FrozenModel(BaseModel): + model_config = ConfigDict(frozen=True, extra="forbid", validate_default=True) + + def model_copy(self, *, update=None, deep: bool = False) -> Self: + data = self.model_dump(round_trip=True) + if update: + data.update(update) + return type(self).model_validate(data) + + @field_serializer("*", when_used="json", check_fields=False) + def serialize_utc(self, value): + if isinstance(value, datetime): + return value.astimezone(UTC).isoformat().replace("+00:00", "Z") + return value + + +class JobSpec(_FrozenModel): + """Execution input. The local root is deliberately excluded from serialization.""" + + workspace_id: str + job_type: str + workspace_root: Path = Field(exclude=True) + spec_version: str = Field(min_length=1, max_length=64) + pipeline_version: str = Field(min_length=1, max_length=64) + config_fingerprint: str + input_fingerprint: str + stage_ids: tuple[str, ...] + dry_run: bool = False + resume_run_id: str | None = None + + _workspace_key = field_validator("workspace_id")(_validate_job_key) + _job_type_key = field_validator("job_type")(_validate_job_key) + _version_keys = field_validator("spec_version", "pipeline_version")(_validate_job_key) + _resume_id = field_validator("resume_run_id")(_validate_run_id) + _config_fingerprint = field_validator("config_fingerprint")( + lambda value: value if _FINGERPRINT.fullmatch(value) else _invalid_fingerprint() + ) + _input_fingerprint = field_validator("input_fingerprint")( + lambda value: value if _FINGERPRINT.fullmatch(value) else _invalid_fingerprint() + ) + _stage_ids = field_validator("stage_ids")( + lambda values: tuple(_validate_job_key(value) for value in values) + ) + + def model_copy(self, *, update=None, deep: bool = False) -> Self: + data = { + "workspace_id": self.workspace_id, + "job_type": self.job_type, + "workspace_root": self.workspace_root, + "spec_version": self.spec_version, + "pipeline_version": self.pipeline_version, + "config_fingerprint": self.config_fingerprint, + "input_fingerprint": self.input_fingerprint, + "stage_ids": self.stage_ids, + "dry_run": self.dry_run, + "resume_run_id": self.resume_run_id, + } + if update: + data.update(update) + return type(self).model_validate(data) + + def with_resume(self, run_id: str) -> "JobSpec": + return self.model_copy(update={"resume_run_id": run_id}) + + +class StageError(_FrozenModel): + category: Literal["internal"] = "internal" + code: Literal["stage_exception"] = "stage_exception" + message: Literal["stage execution failed"] = "stage execution failed" + + +class StageRun(_FrozenModel): + name: str + status: StageStatus = "pending" + started_at: datetime | None = None + finished_at: datetime | None = None + error: StageError | None = None + effect_state: EffectState | None = None + artifact_manifest_digest: str | None = None + artifact_files: tuple[str, ...] = () + + _name_key = field_validator("name")(_validate_job_key) + _artifact_digest = field_validator("artifact_manifest_digest")( + lambda value: value if value is None or _FINGERPRINT.fullmatch(value) else _invalid_fingerprint() + ) + + @model_validator(mode="after") + def state_shape(self) -> "StageRun": + if self.status == "pending" and any( + value is not None for value in ( + self.started_at, self.finished_at, self.error, self.effect_state, + self.artifact_manifest_digest, + ) + ): + raise ValueError("pending stage cannot contain timestamps or error") + if self.status == "running" and ( + self.started_at is None or self.finished_at is not None or self.error is not None + ): + raise ValueError("running stage requires only started_at") + if self.status == "succeeded" and ( + self.started_at is None or self.finished_at is None or self.error is not None + ): + raise ValueError("succeeded stage requires timestamps and no error") + if self.status == "failed" and ( + self.started_at is None or self.finished_at is None or self.error is None + ): + raise ValueError("failed stage requires timestamps and safe error") + if (self.effect_state is None) != (self.artifact_manifest_digest is None): + raise ValueError("effect state and artifact manifest digest must be persisted together") + if self.artifact_files and self.effect_state is None: + raise ValueError("artifact files require a persisted effect state") + return self + + +class JobRun(_FrozenModel): + """Durable checkpoint, persisted after every state transition.""" + + schema_version: Literal[1] = 1 + run_id: str + compatibility_fingerprint: str + workspace_fingerprint: str + job_type: str + spec_version: str + pipeline_version: str + config_fingerprint: str + input_fingerprint: str + dry_run: bool + status: JobStatus + started_at: datetime + finished_at: datetime | None = None + resumed_from: str | None = None + stages: tuple[StageRun, ...] = () + + _run_id = field_validator("run_id")(_validate_run_id) + _compatibility = field_validator("compatibility_fingerprint", "workspace_fingerprint")( + lambda value: value if _FINGERPRINT.fullmatch(value) else _invalid_fingerprint() + ) + _input_fingerprints = field_validator("config_fingerprint", "input_fingerprint")( + lambda value: value if _FINGERPRINT.fullmatch(value) else _invalid_fingerprint() + ) + _job_type = field_validator("job_type")(_validate_job_key) + _persisted_versions = field_validator("spec_version", "pipeline_version")(_validate_job_key) + _resumed_from = field_validator("resumed_from")(_validate_run_id) + + @model_validator(mode="after") + def ledger_shape(self) -> "JobRun": + names = [stage.name for stage in self.stages] + if len(names) != len(set(names)): + raise ValueError("stage identifiers must be unique") + statuses = [stage.status for stage in self.stages] + first_incomplete = next( + (index for index, status in enumerate(statuses) if status != "succeeded"), + len(statuses), + ) + if any(status != "pending" for status in statuses[first_incomplete + 1 :]): + raise ValueError("stage ledger must be an ordered execution prefix") + if self.status == "succeeded" and ( + self.finished_at is None or any(status != "succeeded" for status in statuses) + ): + raise ValueError("succeeded job requires a complete succeeded ledger") + if self.status == "failed" and ( + self.finished_at is None + or first_incomplete == len(statuses) + or statuses[first_incomplete] != "failed" + ): + raise ValueError("failed job requires the first incomplete stage to be failed") + if self.status == "running" and self.finished_at is not None: + raise ValueError("running job cannot have finished_at") + return self + + +class JobReport(JobRun): + """Public machine-readable terminal report (contains no paths or stage outputs).""" + + +def _invalid_fingerprint(): + raise ValueError("fingerprint must be sha256 followed by 64 lowercase hexadecimal characters") diff --git a/harness/tht/jobs/runner.py b/harness/tht/jobs/runner.py new file mode 100644 index 00000000..89f73b46 --- /dev/null +++ b/harness/tht/jobs/runner.py @@ -0,0 +1,427 @@ +"""Resumable stage runner with durable atomic checkpoints and reports.""" + +from __future__ import annotations + +import json +import hashlib +import os +import uuid +import stat +import shutil +from collections.abc import Callable, Sequence +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +from pydantic import ValidationError + +from tht.jobs.locking import WorkspaceJobLock +from tht.jobs.models import JobReport, JobRun, JobSpec, StageError, StageRun, utc_now + + +class CorruptCheckpointError(RuntimeError): + pass + + +@dataclass(frozen=True) +class StageArtifacts: + required: tuple[str, ...] = () + + +@dataclass(frozen=True) +class JobContext: + run_id: str + job_type: str + dry_run: bool + workspace_root: Path + run_dir: Path + _record_artifacts: Callable[[str, tuple[str, ...], str], None] | None = None + + def record_artifacts( + self, stage: str, required: tuple[str, ...], effect_state: str, + ) -> None: + if self._record_artifacts is None: + raise RuntimeError("artifact recorder is unavailable") + self._record_artifacts(stage, required, effect_state) + + +Stage = Callable[[JobContext], Any] + + +def _artifact_digest(path: Path) -> dict[str, Any]: + payload = path.read_bytes() + return {"sha256": hashlib.sha256(payload).hexdigest(), "size": len(payload)} + + +def _seal_artifacts(context: JobContext, stage: str, result: Any, spec: JobSpec) -> str: + required = result.required if isinstance(result, StageArtifacts) else () + root = context.run_dir / "artifacts" + root.mkdir(exist_ok=True) + manifest_path = root / "artifact-manifest.json" + manifest = json.loads(manifest_path.read_text()) if manifest_path.exists() else { + "schema_version": 1, + "spec_fingerprint": _compatibility_fingerprint(spec, list(spec.stage_ids)), + "stages": {}, + } + files = {} + for relative in required: + candidate = root / relative + if Path(relative).is_absolute() or ".." in Path(relative).parts or candidate.is_symlink(): + raise CorruptCheckpointError("artifact path is unsafe") + if not candidate.is_file(): + raise CorruptCheckpointError("required stage artifact is missing") + files[relative] = _artifact_digest(candidate) + for prior in manifest["stages"].values(): + if relative in prior.get("required", []): + prior["required"].remove(relative) + prior["files"].pop(relative, None) + manifest["stages"][stage] = {"required": list(required), "files": files} + canonical = json.dumps(manifest, sort_keys=True, separators=(",", ":")) + "\n" + _atomic_write(manifest_path, canonical) + return _value_fingerprint(canonical) + + +def seal_stage_artifacts( + context: JobContext, stage: str, required: tuple[str, ...], spec: JobSpec, +) -> None: + """Durably record external-effect intent before a stage performs that effect.""" + context.record_artifacts(stage, required, "intent") + + +def _validate_artifacts(run_dir: Path, spec: JobSpec, source: JobRun) -> set[str]: + root = run_dir / "artifacts" + manifest_path = root / "artifact-manifest.json" + successful = {stage.name for stage in source.stages if stage.status == "succeeded"} + if not successful and not manifest_path.exists(): + return set() + try: + manifest = json.loads(manifest_path.read_text()) + root_digest = _value_fingerprint(manifest_path.read_text()) + if manifest["spec_fingerprint"] != _compatibility_fingerprint(spec, list(spec.stage_ids)): + raise ValueError + sealed = set(manifest["stages"]) + allowed = {"artifact-manifest.json"} + for stage, record in manifest["stages"].items(): + for relative in record["required"]: + candidate = root / relative + if Path(relative).is_absolute() or ".." in Path(relative).parts or candidate.is_symlink(): + raise ValueError + if not candidate.is_file() or _artifact_digest(candidate) != record["files"][relative]: + raise ValueError + allowed.add(relative) + if not successful.issubset(sealed): + raise ValueError + for stage in source.stages: + if stage.effect_state is None: + continue + record = manifest["stages"].get(stage.name) + if ( + stage.artifact_manifest_digest != root_digest + or record is None + or tuple(record["required"]) != stage.artifact_files + ): + raise ValueError + entries = list(root.iterdir()) + if any(path.is_symlink() or not path.is_file() for path in entries): + raise ValueError + actual = {path.name for path in entries} + incomplete = next((stage for stage in source.stages if stage.status != "succeeded"), None) + marker = root / "compensated.json" + if incomplete is not None and incomplete.status in {"failed", "running"} and marker.is_file(): + payload = json.loads(marker.read_text()) + if not isinstance(payload.get("generation"), str): + raise ValueError + allowed.add("compensated.json") + if actual != allowed: + raise ValueError + return { + stage.name for stage in source.stages + if stage.effect_state == "completed" + } + except (OSError, KeyError, TypeError, ValueError, json.JSONDecodeError) as error: + raise CorruptCheckpointError("resume artifact manifest is invalid") from error + + +def _atomic_write(path: Path, payload: str) -> None: + temporary = path.with_name(f".{path.name}.{uuid.uuid4().hex}.tmp") + fd = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL, 0o600) + try: + with os.fdopen(fd, "w", encoding="utf-8") as stream: + stream.write(payload) + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + directory_fd = os.open(path.parent, os.O_RDONLY) + try: + os.fsync(directory_fd) + finally: + os.close(directory_fd) + except BaseException: + try: + temporary.unlink() + except FileNotFoundError: + pass + raise + + +def _persist(path: Path, run: JobRun) -> None: + _atomic_write(path, run.model_dump_json(indent=2) + "\n") + + +def _load_checkpoint(path: Path) -> JobRun: + try: + return JobRun.model_validate_json(path.read_text(encoding="utf-8")) + except (OSError, ValidationError, ValueError, json.JSONDecodeError) as error: + raise CorruptCheckpointError("checkpoint is invalid and cannot be resumed") from error + + +def _new_run(spec: JobSpec, run_id: str, stages: Sequence[Stage]) -> JobRun: + names = list(spec.stage_ids) + if len(names) != len(stages): + raise ValueError("stage_ids must identify every stage exactly once") + if len(names) != len(set(names)): + raise ValueError("stage names must be unique") + return JobRun( + run_id=run_id, + compatibility_fingerprint=_compatibility_fingerprint(spec, names), + workspace_fingerprint=_value_fingerprint(spec.workspace_id), + job_type=spec.job_type, + spec_version=spec.spec_version, + pipeline_version=spec.pipeline_version, + config_fingerprint=spec.config_fingerprint, + input_fingerprint=spec.input_fingerprint, + dry_run=spec.dry_run, + status="running", + started_at=utc_now(), + resumed_from=spec.resume_run_id, + stages=tuple(StageRun(name=name) for name in names), + ) + + +def _resume_run( + spec: JobSpec, run_id: str, stages: Sequence[Stage], source: JobRun, + effect_completed: set[str] | None = None, +) -> JobRun: + requested_names = list(spec.stage_ids) + if len(requested_names) != len(stages) or len(requested_names) != len(set(requested_names)): + raise CorruptCheckpointError("resume checkpoint is incompatible with requested stages") + source_fingerprint = _source_compatibility_fingerprint(source) + if source.compatibility_fingerprint != source_fingerprint: + raise CorruptCheckpointError("resume checkpoint compatibility fingerprint is invalid") + expected = _compatibility_fingerprint(spec, requested_names) + if source_fingerprint != expected or [stage.name for stage in source.stages] != requested_names: + raise CorruptCheckpointError( + "resume checkpoint is incompatible; start an intentional new run without resume" + ) + source_by_name = {stage.name: stage for stage in source.stages} + resumed_stages = [] + for name in requested_names: + previous = source_by_name.get(name) + if previous is not None and previous.status == "succeeded": + resumed_stages.append(previous) + elif previous is not None and previous.status == "running" and name in (effect_completed or set()): + resumed_stages.append(StageRun( + name=name, status="succeeded", started_at=previous.started_at or utc_now(), + finished_at=utc_now(), + effect_state="completed", + artifact_manifest_digest=previous.artifact_manifest_digest, + artifact_files=previous.artifact_files, + )) + else: + resumed_stages.append(StageRun(name=name)) + return JobRun( + run_id=run_id, + compatibility_fingerprint=source.compatibility_fingerprint, + workspace_fingerprint=source.workspace_fingerprint, + job_type=spec.job_type, + spec_version=spec.spec_version, + pipeline_version=spec.pipeline_version, + config_fingerprint=spec.config_fingerprint, + input_fingerprint=spec.input_fingerprint, + dry_run=spec.dry_run, + status="running", + started_at=utc_now(), + resumed_from=source.run_id, + stages=tuple(resumed_stages), + ) + + +def run_job( + spec: JobSpec, stages: Sequence[Stage], *, + after_stage_return: Callable[[JobContext, str], Any] | None = None, + reconcile_effects: Callable[[JobRun, Path], set[str]] | None = None, +) -> JobReport: + """Run stages once, returning a terminal report instead of leaking stage exceptions.""" + with WorkspaceJobLock(spec.workspace_root, spec.workspace_id, spec.job_type): + jobs_root = spec.workspace_root / ".tht-jobs" / spec.job_type / "runs" + if spec.resume_run_id is None: + source = None + else: + source_path = jobs_root / spec.resume_run_id / "checkpoint.json" + if not source_path.exists(): + matches = list( + (spec.workspace_root / ".tht-jobs").glob( + f"*/runs/{spec.resume_run_id}/checkpoint.json" + ) + ) + if len(matches) == 1: + source_path = matches[0] + source = _load_checkpoint(source_path) + _validate_resume_source(spec, stages, source) + effect_completed = _validate_artifacts(source_path.parent, spec, source) + if reconcile_effects is not None: + effect_completed |= reconcile_effects(source, source_path.parent) + + run_id = uuid.uuid4().hex + run_dir = jobs_root / run_id + _prepare_run_directory(spec.workspace_root, spec.job_type, run_id) + checkpoint_path = run_dir / "checkpoint.json" + if source is None: + run = _new_run(spec, run_id, stages) + else: + run = _resume_run(spec, run_id, stages, source, effect_completed) + source_artifacts = jobs_root / source.run_id / "artifacts" + if source_artifacts.exists(): + shutil.copytree(source_artifacts, run_dir / "artifacts") + _persist(checkpoint_path, run) + current_index = -1 + + def record_artifacts(stage_name: str, required: tuple[str, ...], effect_state: str) -> None: + nonlocal run + if current_index < 0 or run.stages[current_index].name != stage_name: + raise CorruptCheckpointError("artifact producer does not match running stage") + digest = _seal_artifacts(context, stage_name, StageArtifacts(required), spec) + manifest = json.loads((run_dir / "artifacts" / "artifact-manifest.json").read_text()) + updated = [] + for position, value in enumerate(run.stages): + record = manifest["stages"].get(value.name) + if record is not None and (value.status == "succeeded" or position == current_index): + state = effect_state if position == current_index else value.effect_state + updated.append(value.model_copy(update={ + "effect_state": state, + "artifact_manifest_digest": digest, + "artifact_files": tuple(record["required"]), + })) + else: + updated.append(value) + run = run.model_copy(update={"stages": tuple(updated)}) + _persist(checkpoint_path, run) + + context = JobContext( + run_id, spec.job_type, spec.dry_run, spec.workspace_root, run_dir, + record_artifacts, + ) + + for index, stage_callable in enumerate(stages): + current_index = index + if run.stages[index].status == "succeeded": + continue + stage = run.stages[index].model_copy( + update={"status": "running", "started_at": utc_now()} + ) + run = run.model_copy( + update={"stages": run.stages[:index] + (stage,) + run.stages[index + 1 :]} + ) + _persist(checkpoint_path, run) + try: + stage_result = stage_callable(context) + except Exception: + failed = stage.model_copy( + update={ + "status": "failed", + "finished_at": utc_now(), + "error": StageError(), + } + ) + run = run.model_copy( + update={ + "status": "failed", + "finished_at": utc_now(), + "stages": run.stages[:index] + (failed,) + run.stages[index + 1 :], + } + ) + _persist(checkpoint_path, run) + break + required = stage_result.required if isinstance(stage_result, StageArtifacts) else () + context.record_artifacts(stage.name, required, "completed") + stage = run.stages[index] + if after_stage_return is not None: + after_stage_return(context, stage.name) + succeeded = stage.model_copy(update={"status": "succeeded", "finished_at": utc_now()}) + run = run.model_copy( + update={"stages": run.stages[:index] + (succeeded,) + run.stages[index + 1 :]} + ) + _persist(checkpoint_path, run) + else: + run = run.model_copy(update={"status": "succeeded", "finished_at": utc_now()}) + _persist(checkpoint_path, run) + + report = JobReport.model_validate(run.model_dump()) + _atomic_write(run_dir / "report.json", report.model_dump_json(indent=2) + "\n") + return report + + +def _value_fingerprint(value: str) -> str: + return "sha256:" + hashlib.sha256(value.encode("utf-8")).hexdigest() + + +def _compatibility_fingerprint(spec: JobSpec, stage_ids: list[str]) -> str: + payload = { + "schema_version": 1, + "workspace": _value_fingerprint(spec.workspace_id), + "job_type": spec.job_type, + "dry_run": spec.dry_run, + "spec_version": spec.spec_version, + "pipeline_version": spec.pipeline_version, + "config_fingerprint": spec.config_fingerprint, + "input_fingerprint": spec.input_fingerprint, + "stage_ids": stage_ids, + } + canonical = json.dumps(payload, sort_keys=True, separators=(",", ":")) + return _value_fingerprint(canonical) + + +def _source_compatibility_fingerprint(source: JobRun) -> str: + payload = { + "schema_version": source.schema_version, + "workspace": source.workspace_fingerprint, + "job_type": source.job_type, + "dry_run": source.dry_run, + "spec_version": source.spec_version, + "pipeline_version": source.pipeline_version, + "config_fingerprint": source.config_fingerprint, + "input_fingerprint": source.input_fingerprint, + "stage_ids": [stage.name for stage in source.stages], + } + canonical = json.dumps(payload, sort_keys=True, separators=(",", ":")) + return _value_fingerprint(canonical) + + +def _validate_resume_source(spec: JobSpec, stages: Sequence[Stage], source: JobRun) -> None: + _resume_run(spec, "0" * 32, stages, source) + + +def _prepare_run_directory(workspace_root: Path, job_type: str, run_id: str) -> None: + parent_fd = os.open(workspace_root, os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW) + try: + for component in (".tht-jobs", job_type, "runs", run_id): + try: + os.mkdir(component, 0o700, dir_fd=parent_fd) + os.fsync(parent_fd) + except FileExistsError: + pass + child_fd = os.open( + component, + os.O_RDONLY | os.O_DIRECTORY | os.O_NOFOLLOW, + dir_fd=parent_fd, + ) + info = os.fstat(child_fd) + if not stat.S_ISDIR(info.st_mode) or info.st_uid != os.getuid(): + os.close(child_fd) + raise OSError("unsafe job run directory") + os.fchmod(child_fd, 0o700) + os.close(parent_fd) + parent_fd = child_fd + os.fsync(parent_fd) + finally: + os.close(parent_fd) diff --git a/harness/tht/memory.py b/harness/tht/memory.py index e081b478..8a6acc78 100644 --- a/harness/tht/memory.py +++ b/harness/tht/memory.py @@ -255,7 +255,7 @@ def memory_vector_record_for_decision( def save_one_memory( - records: list[MemoryRecord], decision_seq: int, *, writer, embedder + records: list[MemoryRecord], decision_seq: int, *, store, embedder ) -> int: """Targeted one-row upsert of a promoted decision to pgvector via the writer key (spec D11). This is NOT a full vectorstore resync: it embeds and pushes a single @@ -267,27 +267,22 @@ def save_one_memory( the writer's existing_vector_hashes; the embedding (Ollama round-trip) and the upsert are skipped when the content is unchanged. Idempotent by construction. - `writer` is a VectorRestClient (writer key); `embedder` an embeddings client. + `store` is the configured writable VectorStore; `embedder` an embeddings client. The destructive cleanup (sync's delete-stale step) is intentionally absent: it remains a server-side-only operation via the direct vectordb connection. """ - from tht.vectorstore.rest_writer import pack_metadata + from tht.ports.vector import VectorWriteRecord from tht.vectorstore.store import content_hash record = memory_vector_record_for_decision(records, decision_seq) if record is None: return 0 new_hash = content_hash(record.content) - existing = writer.existing_hashes("memory", ["memory"]) + existing = store.existing_hashes("memory", ["memory"]) if existing.get(record.id) == new_hash: return 0 # unchanged: skip embedding + upsert embedding = embedder.embed_documents([record.content])[0] - row = { - "record_key": record.id, - "kind": record.kind, - "content_hash": new_hash, - "metadata": pack_metadata(record), - "embedding": embedding, - } - return writer.upsert_records("memory", [row]) - + return store.upsert( + "memory", + [VectorWriteRecord(record=record, embedding=embedding, content_hash=new_hash)], + ) diff --git a/harness/tht/migrations/vector/001_extensions.sql b/harness/tht/migrations/vector/001_extensions.sql new file mode 100644 index 00000000..f64f725a --- /dev/null +++ b/harness/tht/migrations/vector/001_extensions.sql @@ -0,0 +1,3 @@ +CREATE SCHEMA IF NOT EXISTS vectors; +REVOKE ALL ON SCHEMA vectors FROM PUBLIC; +CREATE EXTENSION IF NOT EXISTS vector WITH SCHEMA vectors; diff --git a/harness/tht/migrations/vector/002_schema_tables.sql b/harness/tht/migrations/vector/002_schema_tables.sql new file mode 100644 index 00000000..2cffaea8 --- /dev/null +++ b/harness/tht/migrations/vector/002_schema_tables.sql @@ -0,0 +1,32 @@ +CREATE TABLE IF NOT EXISTS vectors.schema_records ( + id bigserial PRIMARY KEY, + record_key text UNIQUE NOT NULL, + kind text NOT NULL, + content_hash text NOT NULL, + metadata jsonb NOT NULL, + embedding vectors.vector(768) NOT NULL, + indexed_at timestamptz NOT NULL DEFAULT pg_catalog.now() +); + +CREATE TABLE IF NOT EXISTS vectors.evidence ( + id bigserial PRIMARY KEY, + record_key text UNIQUE NOT NULL, + kind text NOT NULL, + content_hash text NOT NULL, + metadata jsonb NOT NULL, + embedding vectors.vector(768) NOT NULL, + indexed_at timestamptz NOT NULL DEFAULT pg_catalog.now() +); + +CREATE TABLE IF NOT EXISTS vectors.memory ( + id bigserial PRIMARY KEY, + record_key text UNIQUE NOT NULL, + kind text NOT NULL, + content_hash text NOT NULL, + metadata jsonb NOT NULL, + embedding vectors.vector(768) NOT NULL, + indexed_at timestamptz NOT NULL DEFAULT pg_catalog.now() +); + +REVOKE ALL ON ALL TABLES IN SCHEMA vectors FROM PUBLIC; +REVOKE ALL ON ALL SEQUENCES IN SCHEMA vectors FROM PUBLIC; diff --git a/harness/tht/migrations/vector/003_roles.sql b/harness/tht/migrations/vector/003_roles.sql new file mode 100644 index 00000000..ba88bd2b --- /dev/null +++ b/harness/tht/migrations/vector/003_roles.sql @@ -0,0 +1,23 @@ +DO $roles$ +BEGIN + IF NOT EXISTS (SELECT 1 FROM pg_catalog.pg_roles WHERE rolname = 'vector_reader') THEN + CREATE ROLE vector_reader NOLOGIN; + END IF; + IF NOT EXISTS (SELECT 1 FROM pg_catalog.pg_roles WHERE rolname = 'vector_writer') THEN + CREATE ROLE vector_writer NOLOGIN; + END IF; +END +$roles$; + +REVOKE ALL ON SCHEMA vectors FROM vector_reader, vector_writer; +REVOKE ALL ON ALL TABLES IN SCHEMA vectors FROM vector_reader, vector_writer; +REVOKE ALL ON ALL SEQUENCES IN SCHEMA vectors FROM vector_reader, vector_writer; + +GRANT USAGE ON SCHEMA vectors TO vector_reader, vector_writer; +GRANT SELECT ON ALL TABLES IN SCHEMA vectors TO vector_reader; + +GRANT INSERT, UPDATE +ON vectors.schema_records, vectors.evidence, vectors.memory TO vector_writer; +GRANT SELECT (record_key, kind, content_hash) +ON vectors.schema_records, vectors.evidence, vectors.memory TO vector_writer; +GRANT USAGE ON ALL SEQUENCES IN SCHEMA vectors TO vector_writer; diff --git a/harness/tht/migrations/vector/004_evidence_generation_gc.sql b/harness/tht/migrations/vector/004_evidence_generation_gc.sql new file mode 100644 index 00000000..94d4a4bb --- /dev/null +++ b/harness/tht/migrations/vector/004_evidence_generation_gc.sql @@ -0,0 +1,3 @@ +-- The writer owns derived-generation reconciliation but not runtime similarity reads. +GRANT SELECT (metadata) ON vectors.evidence TO vector_writer; +GRANT DELETE ON vectors.evidence TO vector_writer; diff --git a/harness/tht/paths.py b/harness/tht/paths.py new file mode 100644 index 00000000..d65249b4 --- /dev/null +++ b/harness/tht/paths.py @@ -0,0 +1,54 @@ +from __future__ import annotations + +from dataclasses import dataclass +from pathlib import Path +from typing import TYPE_CHECKING + +from tht.config import ConfigError + +if TYPE_CHECKING: + from tht.config import Config + + +@dataclass(frozen=True) +class ResolvedPaths: + workspace: Path + sessions: Path + artifacts: Path + indexes: Path + corpus: Path + + +def _resolve_root(path: Path, workspace: Path, name: str) -> Path: + if path.is_absolute(): + # Existing deployments commonly point at an external workspace checkout. Keep + # those paths working while `tht doctor` identifies them for migration. + return path + + resolved = (workspace / path).resolve() + if not resolved.is_relative_to(workspace): + raise ConfigError(f"paths.{name} resolves outside workspace root") + return resolved + + +def resolve_workspace_paths( + config_path: Path, cfg: Config, data_root: Path +) -> ResolvedPaths: + """Resolve portable paths under ``/workspaces/``. + + Relative roots are sandboxed to the logical workspace. Absolute roots are a + compatibility bridge for existing installations and are never rewritten. + """ + canonical_data_root = data_root.resolve() + workspaces_root = canonical_data_root / "workspaces" + workspace = (workspaces_root / config_path.stem).resolve() + if not workspace.is_relative_to(workspaces_root): + raise ConfigError("workspace resolves outside workspaces root") + roots = cfg.roots + return ResolvedPaths( + workspace=workspace, + sessions=_resolve_root(roots.sessions, workspace, "sessions"), + artifacts=_resolve_root(roots.artifacts, workspace, "artifacts"), + indexes=_resolve_root(roots.indexes, workspace, "indexes"), + corpus=_resolve_root(Path("corpus"), workspace, "corpus"), + ) diff --git a/harness/tht/ports/__init__.py b/harness/tht/ports/__init__.py new file mode 100644 index 00000000..1bd4e631 --- /dev/null +++ b/harness/tht/ports/__init__.py @@ -0,0 +1,37 @@ +"""Stable interfaces implemented by Thoth infrastructure adapters.""" + +from tht.ports.dwh import ( + DwhAdapter, + DwhCapabilities, + DwhHealth, + DistinctValues, + UnsupportedCapability, +) +from tht.ports.vector import ( + VectorCapabilities, + VectorHealth, + VectorHit, + VectorRecord, + VectorReadUnavailable, + VectorStore, + VectorStoreError, + VectorWriteRecord, + VectorWriteUnavailable, +) + +__all__ = [ + "DwhAdapter", + "DwhCapabilities", + "DwhHealth", + "DistinctValues", + "UnsupportedCapability", + "VectorCapabilities", + "VectorHealth", + "VectorHit", + "VectorRecord", + "VectorReadUnavailable", + "VectorStore", + "VectorStoreError", + "VectorWriteRecord", + "VectorWriteUnavailable", +] diff --git a/harness/tht/ports/dwh.py b/harness/tht/ports/dwh.py new file mode 100644 index 00000000..10bfebb2 --- /dev/null +++ b/harness/tht/ports/dwh.py @@ -0,0 +1,56 @@ +"""Data-warehouse adapter contract.""" + +from dataclasses import dataclass +from typing import Protocol, runtime_checkable + +from tht.execute import ExecResult, PlanSummary +from tht.mschema.models import PhysicalSchema + + +@dataclass(frozen=True) +class DwhCapabilities: + introspection: bool = True + explain: bool = True + sampling: bool = True + distinct_values: bool = True + + +@dataclass(frozen=True) +class DwhHealth: + ok: bool + detail: str | None = None + database: str | None = None + schema: str | None = None + endpoint: str | None = None + read_only: bool | None = None + writable_tables: tuple[str, ...] = () + can_create: bool = False + error_kind: str | None = None + + +@dataclass(frozen=True) +class DistinctValues: + values: list[object] + truncated: bool + + +class UnsupportedCapability(Exception): + """Raised when an adapter cannot provide an optional DWH operation.""" + + +@runtime_checkable +class DwhAdapter(Protocol): + @property + def capabilities(self) -> DwhCapabilities: ... + + def health(self) -> DwhHealth: ... + + def introspect(self) -> PhysicalSchema: ... + + def run_query(self, sql: str, *, limit: int) -> ExecResult: ... + + def explain(self, sql: str) -> PlanSummary: ... + + def sample_column(self, table: str, column: str, *, limit: int) -> list[object]: ... + + def distinct_values(self, table: str, column: str, *, limit: int) -> DistinctValues: ... diff --git a/harness/tht/ports/evidence.py b/harness/tht/ports/evidence.py new file mode 100644 index 00000000..8cd64f89 --- /dev/null +++ b/harness/tht/ports/evidence.py @@ -0,0 +1,210 @@ +"""Credential-free port for discovering and acquiring Evidence objects.""" + +import re +from collections.abc import Iterable, Mapping, Sequence +from datetime import UTC, datetime +from enum import Enum +from typing import Protocol, Self, runtime_checkable +from urllib.parse import parse_qsl, urlsplit, urlunsplit + +from pydantic import BaseModel, ConfigDict, Field, JsonValue, TypeAdapter, field_validator + + +class FrozenDict(dict): + """A JSON-serializable dict whose mutation operations are disabled.""" + + def _immutable(self, *args, **kwargs): + raise TypeError("frozen JSON metadata cannot be mutated") + + __delitem__ = _immutable + __ior__ = _immutable + __setitem__ = _immutable + clear = _immutable + pop = _immutable + popitem = _immutable + setdefault = _immutable + update = _immutable + + +_CAMEL_BOUNDARY = re.compile(r"(?<=[a-z0-9])(?=[A-Z])") +_SEPARATORS = re.compile(r"[^a-z0-9]+") +_NAMESPACED_VALUE = re.compile(r"^[a-z][a-z0-9_-]*:[A-Za-z0-9._:-]+$") +_CREDENTIAL_KEYS = { + "apikey", + "authorization", + "authtoken", + "bearertoken", + "clientsecret", + "credential", + "credentials", + "password", + "passwd", + "privatekey", + "refreshtoken", + "sessioncookie", + "xapikey", + "accesstoken", +} +_JSON_METADATA = TypeAdapter(dict[str, JsonValue]) + + +def _normalize_key(key: str) -> str: + return _SEPARATORS.sub("", _CAMEL_BOUNDARY.sub("_", key).lower()) + + +def _is_credential_key(key: str) -> bool: + return _normalize_key(key) in _CREDENTIAL_KEYS + + +def _reject_credentials(value, path: str = "metadata") -> None: + if isinstance(value, Mapping): + for key, child in value.items(): + if _is_credential_key(str(key)): + raise ValueError(f"credential-like metadata key is not allowed: {path}.{key}") + _reject_credentials(child, f"{path}.{key}") + elif isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)): + for index, child in enumerate(value): + _reject_credentials(child, f"{path}[{index}]") + + +def freeze_json(value): + """Recursively freeze a Pydantic-validated JSON value without changing its JSON shape.""" + if isinstance(value, Mapping): + return FrozenDict({str(key): freeze_json(child) for key, child in value.items()}) + if isinstance(value, Sequence) and not isinstance(value, (str, bytes, bytearray)): + return tuple(freeze_json(child) for child in value) + return value + + +def validate_safe_metadata(value: dict[str, JsonValue]) -> FrozenDict: + _reject_credentials(value) + return freeze_json(value) + + +def validate_canonical_uri(value: str) -> str: + try: + parsed = urlsplit(value) + _ = parsed.port + except ValueError as error: + raise ValueError("invalid canonical URI") from error + if not parsed.scheme: + raise ValueError("canonical URI must include a scheme") + if parsed.username is not None or parsed.password is not None: + raise ValueError("canonical URI must not contain credentials in userinfo") + for key, _ in parse_qsl(parsed.query, keep_blank_values=True): + if _is_credential_key(key): + raise ValueError("canonical URI must not contain credentials in query parameters") + return value + + +def canonical_provenance_uri(value: str) -> str: + """Return only stable URI identity; transport query/fragment data is never provenance.""" + try: + parsed = urlsplit(value) + _ = parsed.port + except ValueError as error: + raise ValueError("invalid canonical URI") from error + if not parsed.scheme: + raise ValueError("canonical URI must include a scheme") + if parsed.username is not None or parsed.password is not None: + raise ValueError("canonical URI must not contain credentials in userinfo") + return urlunsplit((parsed.scheme, parsed.netloc, parsed.path, "", "")) + + +def normalize_aware_datetime(value: datetime | None) -> datetime | None: + if value is None: + return None + if value.tzinfo is None or value.utcoffset() is None: + raise ValueError("datetime must be timezone-aware") + return value.astimezone(UTC) + + +def validate_namespaced_value(value: str) -> str: + if not _NAMESPACED_VALUE.fullmatch(value): + raise ValueError("value must be namespaced as ':'") + return value + + +class _EvidenceValue(BaseModel): + model_config = ConfigDict( + frozen=True, + extra="forbid", + revalidate_instances="always", + validate_default=True, + ser_json_bytes="base64", + val_json_bytes="base64", + ) + + def model_copy(self, *, update: Mapping[str, object] | None = None, deep: bool = False) -> Self: + """Copy through validation; Pydantic's unchecked update-copy is unsafe for contracts.""" + data = self.model_dump(round_trip=True) + if update: + data.update(update) + return type(self).model_validate(data) + + +class SourceObject(_EvidenceValue): + source_id: str = Field(min_length=1) + uri: str = Field(min_length=1) + fingerprint: str = Field(min_length=1) + modified_at: datetime | None = None + metadata: dict[str, JsonValue] = Field(default_factory=dict) + + _source_id = field_validator("source_id")(validate_namespaced_value) + _fingerprint = field_validator("fingerprint")(validate_namespaced_value) + _safe_uri = field_validator("uri")(validate_canonical_uri) + _aware_modified_at = field_validator("modified_at")(normalize_aware_datetime) + _frozen_metadata = field_validator("metadata")(validate_safe_metadata) + + +class AcquiredDocument(_EvidenceValue): + """Transport result; bytes use explicit base64 encoding in JSON mode.""" + + source: SourceObject + content: bytes + media_type: str | None = None + acquired_at: datetime | None = None + metadata: dict[str, JsonValue] = Field(default_factory=dict) + + _aware_acquired_at = field_validator("acquired_at")(normalize_aware_datetime) + _frozen_metadata = field_validator("metadata")(validate_safe_metadata) + + +class EvidenceSourceErrorCategory(str, Enum): + TRANSIENT = "transient" + PERMANENT = "permanent" + + +class EvidenceSourceError(Exception): + """Classified source failure with credential-free structured diagnostics.""" + + def __init__( + self, + _message: str, + *, + category: EvidenceSourceErrorCategory, + details: dict[str, JsonValue] | None = None, + ) -> None: + super().__init__("evidence source operation failed") + object.__setattr__(self, "category", EvidenceSourceErrorCategory(category)) + object.__setattr__( + self, + "details", + validate_safe_metadata(_JSON_METADATA.validate_python(details or {})), + ) + + def __setattr__(self, name: str, value) -> None: + if name in {"args", "category", "details"} and hasattr(self, name): + raise AttributeError(f"{name} is immutable") + super().__setattr__(name, value) + + @property + def retryable(self) -> bool: + return self.category is EvidenceSourceErrorCategory.TRANSIENT + + +@runtime_checkable +class EvidenceSource(Protocol): + def discover(self) -> Iterable[SourceObject]: ... + + def acquire(self, item: SourceObject) -> AcquiredDocument: ... diff --git a/harness/tht/ports/vector.py b/harness/tht/ports/vector.py new file mode 100644 index 00000000..9fd1cd8d --- /dev/null +++ b/harness/tht/ports/vector.py @@ -0,0 +1,98 @@ +"""Transport-neutral vector-store contract and canonical vector models.""" + +from dataclasses import dataclass +from typing import Protocol, runtime_checkable + +from tht.vectorstore.records import VectorRecord +from tht.vectorstore.store import VectorHit + + +@dataclass(frozen=True) +class VectorCapabilities: + search: bool = True + existing_hashes: bool = False + upsert: bool = False + metadata_filter: bool = False + delete_generation: bool = False + list_evidence_generations: bool = False + + +@dataclass(frozen=True) +class VectorHealth: + ok: bool + detail: str | None = None + read_configured: bool = False + read_reachable: bool | None = None + read_detail: str | None = None + write_configured: bool = False + write_reachable: bool | None = None + write_detail: str | None = None + expected_dimension: int | None = None + observed_dimensions: tuple[int, ...] = () + dimension_compatible: bool | None = None + + +@dataclass(frozen=True) +class VectorWriteRecord: + """A canonical record plus transport-neutral, precomputed vector data.""" + + record: VectorRecord + embedding: list[float] + content_hash: str + + +class VectorStoreError(Exception): + """Base error exposed by vector adapters.""" + + +class VectorWriteUnavailable(VectorStoreError): + """Raised when a deployment has no vector writer credential.""" + + +class VectorReadUnavailable(VectorStoreError): + """Raised when a deployment has no vector reader credential.""" + + +def require_positive_limit(limit: int) -> None: + """Reject coercible values: vector limits are exact positive integers.""" + if type(limit) is not int or limit <= 0: + raise ValueError("Vector search limit must be a positive integer") + + +@runtime_checkable +class VectorStore(Protocol): + @property + def capabilities(self) -> VectorCapabilities: ... + + def health(self) -> VectorHealth: ... + + def search( + self, + collections: list[str], + embedding: list[float], + *, + limit: int, + kinds: list[str] | None = None, + metadata_filter: dict[str, object] | None = None, + ) -> list[VectorHit]: ... + + def existing_hashes(self, collection: str, kinds: list[str]) -> dict[str, str]: ... + + def upsert(self, collection: str, records: list[VectorWriteRecord]) -> int: ... + + def delete_generation(self, collection: str, generation: str, workspace_id: str) -> int: ... + + def list_evidence_generations(self, collection: str, workspace_id: str) -> list[str]: ... + + +__all__ = [ + "VectorCapabilities", + "VectorHealth", + "VectorHit", + "VectorRecord", + "VectorReadUnavailable", + "VectorStore", + "VectorStoreError", + "VectorWriteRecord", + "VectorWriteUnavailable", +] diff --git a/harness/tht/rest/execute.py b/harness/tht/rest/execute.py index 223f460f..f1a272c7 100644 --- a/harness/tht/rest/execute.py +++ b/harness/tht/rest/execute.py @@ -9,7 +9,14 @@ il client mantiene solo l'iniezione del LIMIT (per il rilevamento del troncament import time -from tht.execute import ExecResult, ExecutionError, PlanSummary, _inject_limit, assert_read_only +from tht.execute import ( + ExecResult, + ExecutionError, + PlanSummary, + _inject_limit, + assert_read_only, + require_positive_int, +) from tht.rest.client import RestError from tht.rest.explain import parse_text_plan @@ -17,6 +24,7 @@ from tht.rest.explain import parse_text_plan def run_controlled_rest(client, sql: str, *, limit: int) -> ExecResult: # Guard read-only client-side anche sul path REST (D7): non delegare l'unica verifica # al server. Stesso check strutturale del path diretto. + limit = require_positive_int(limit, name="limit") assert_read_only(sql) final_sql, injected = _inject_limit(sql, limit) start = time.monotonic() diff --git a/harness/tht/search/evidence.py b/harness/tht/search/evidence.py new file mode 100644 index 00000000..077b3355 --- /dev/null +++ b/harness/tht/search/evidence.py @@ -0,0 +1,116 @@ +"""Runtime Evidence lookup bound to the atomically active corpus generation.""" + +import re + +from tht.corpus.store import CorpusStore + + +class CorpusWorkspaceMismatchError(RuntimeError): + """The configured workspace does not own the persisted corpus.""" + + +class ActiveEvidenceSearcher: + """Searcher facade that enforces ACTIVE generation predicates before LIMIT.""" + + def __init__(self, corpus: CorpusStore, delegate, expected_workspace_id: str | None = None): + self.corpus = corpus + self.delegate = delegate + self.expected_workspace_id = expected_workspace_id + + def search(self, embedding, top_n=10, kinds=None, metadata_filter=None): + requested = set(kinds) if kinds is not None else { + "schema_table", "schema_column", "evidence", "memory", "solved_question", + } + include_evidence = "evidence" in requested + other_kinds = sorted(requested - {"evidence"}) + with self.corpus.writer_lock(): + manifest = self.corpus.active_manifest() + persisted_workspace = manifest.metadata.get("workspace_id") if manifest else None + if manifest is not None and ( + not isinstance(persisted_workspace, str) + or re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", persisted_workspace) is None + ): + raise CorpusWorkspaceMismatchError( + "corpus workspace ownership is missing or invalid; use a new corpus root or rebuild" + ) + if manifest is not None and self.expected_workspace_id is not None and ( + persisted_workspace != self.expected_workspace_id + ): + raise CorpusWorkspaceMismatchError( + "corpus belongs to a different workspace; use a new corpus root or rebuild" + ) + if not include_evidence: + kwargs = {"top_n": top_n, "kinds": kinds} + if metadata_filter is not None: + kwargs["metadata_filter"] = metadata_filter + return self.delegate.search(embedding, **kwargs) + hits = [] + if other_kinds: + kwargs = {"top_n": top_n, "kinds": other_kinds} + if metadata_filter is not None: + kwargs["metadata_filter"] = metadata_filter + hits.extend(self.delegate.search(embedding, **kwargs)) + if include_evidence: + workspace_id = manifest.metadata.get("workspace_id") if manifest else None + if manifest is not None and isinstance(workspace_id, str): + by_generation: dict[str, list[str]] = {} + mapping = dict(manifest.metadata.get("document_generations", {})) + for document in manifest.documents: + generation = mapping.get(document.document_id, manifest.vector_generation) + if generation: + by_generation.setdefault(generation, []).append(document.document_id) + for generation, document_ids in sorted(by_generation.items()): + hits.extend(self.delegate.search( + embedding, top_n=top_n, kinds=["evidence"], + metadata_filter={ + "vector_generation": generation, + "document_ids": sorted(document_ids), + "workspace_id": workspace_id, + }, + )) + return sorted(hits, key=lambda hit: (-hit.similarity, hit.id))[:top_n] + + +def active_searcher(cfg, delegate, *, workspace_id: str | None = None): + corpus_root = cfg.paths.artifacts.parent / "corpus" + return ActiveEvidenceSearcher(CorpusStore(corpus_root), delegate, workspace_id) + + +def validate_corpus_workspace(cfg, workspace_id: str) -> None: + """Fail before downstream retrieval setup when configured corpus ownership differs.""" + corpus = CorpusStore(cfg.paths.artifacts.parent / "corpus") + with corpus.writer_lock(): + manifest = corpus.active_manifest() + if manifest is None: + return + persisted = manifest.metadata.get("workspace_id") + if not isinstance(persisted, str) or re.fullmatch( + r"[a-z][a-z0-9_-]{0,63}", persisted + ) is None: + raise CorpusWorkspaceMismatchError( + "corpus workspace ownership is missing or invalid; use a new corpus root or rebuild" + ) + if persisted != workspace_id: + raise CorpusWorkspaceMismatchError( + "corpus belongs to a different workspace; use a new corpus root or rebuild" + ) + + +def resolve_evidence_file( + store: CorpusStore, evidence_id: str, *, materialized_root=None, +) -> str: + with store.writer_lock(): + manifest = store.active_manifest() + if manifest is None: + return "" + for document in manifest.documents: + frontmatter = document.metadata.get("frontmatter", {}) + identifiers = {document.document_id, document.source_id, str(frontmatter.get("id", ""))} + if evidence_id in identifiers: + root = materialized_root or (store.root / "runtime") + filename = document.document_id.removeprefix("doc:") + ".md" + path = store.materialize_document( + document.document_id, root / filename, generation=manifest.manifest_id, + ) + return str(path) if path else "" + return "" diff --git a/harness/tht/session/artifacts.py b/harness/tht/session/artifacts.py index 2758ba06..6f4a65cd 100644 --- a/harness/tht/session/artifacts.py +++ b/harness/tht/session/artifacts.py @@ -5,6 +5,16 @@ from tht.session.models import SchemaLinking def _find_evidence_file(evidence_root: Path, evidence_id: str) -> str: + # New deployments resolve only immutable materialized files from ACTIVE. Keep + # the legacy curated-tree fallback for sessions created before a corpus exists. + corpus_root = evidence_root.parent.parent / "corpus" + if corpus_root.exists(): + from tht.corpus.store import CorpusStore + from tht.search.evidence import resolve_evidence_file + return resolve_evidence_file( + CorpusStore(corpus_root), evidence_id, + materialized_root=evidence_root.parent / ".materialized-evidence", + ) for match in evidence_root.rglob(f"{evidence_id}.md"): return str(match) return "" diff --git a/harness/tht/solved.py b/harness/tht/solved.py index 96de27e5..e62910de 100644 --- a/harness/tht/solved.py +++ b/harness/tht/solved.py @@ -43,25 +43,22 @@ def _solved_hash(record: VectorRecord) -> str: return content_hash(record.content + "\n" + str(record.metadata.get("sql", ""))) -def save_solved_question(record: VectorRecord, *, writer, embedder) -> int: +def save_solved_question(record: VectorRecord, *, store, embedder) -> int: """Upsert one-row della coppia domanda->SQL via writer key (stesso pattern di save_one_memory, spec D11): hash dedup client-side, embedding solo se domanda o SQL sono cambiati. `writer` e' un VectorRestClient (writer key). Ritorna il numero di righe upsertate (0 = invariata).""" - from tht.vectorstore.rest_writer import pack_metadata + from tht.ports.vector import VectorWriteRecord new_hash = _solved_hash(record) - existing = writer.existing_hashes("memory", [SOLVED_KIND]) + existing = store.existing_hashes("memory", [SOLVED_KIND]) if existing.get(record.id) == new_hash: return 0 embedding = embedder.embed_documents([record.content])[0] - return writer.upsert_records("memory", [{ - "record_key": record.id, - "kind": record.kind, - "content_hash": new_hash, - "metadata": pack_metadata(record), - "embedding": embedding, - }]) + return store.upsert( + "memory", + [VectorWriteRecord(record=record, embedding=embedding, content_hash=new_hash)], + ) class SolvedIndexError(Exception): diff --git a/harness/tht/vectorstore/reader.py b/harness/tht/vectorstore/reader.py index 50e852cf..b0cad559 100644 --- a/harness/tht/vectorstore/reader.py +++ b/harness/tht/vectorstore/reader.py @@ -9,8 +9,10 @@ Entrambe mappano i `kind` sulle tabelle per-dominio dello schema `vectors`. from sqlalchemy import Engine +from tht.adapters.vector.legacy_direct import LegacyDirectVectorStore +from tht.adapters.vector.thoth_http import ThothHttpVectorStore from tht.vectorstore.rest_client import VectorRestClient -from tht.vectorstore.store import VectorHit, VectorStore, hit_from_metadata +from tht.vectorstore.store import VectorHit # kind Thoth → tabella dello schema `vectors`. KIND_TO_TABLE = { @@ -30,30 +32,19 @@ def tables_for_kinds(kinds: list[str] | None) -> list[str]: return sorted({KIND_TO_TABLE[k] for k in kinds if k in KIND_TO_TABLE}) -def _merge(hits: list[VectorHit], top_n: int) -> list[VectorHit]: - return sorted(hits, key=lambda h: h.similarity, reverse=True)[:top_n] - - class RestSearcher: """Similarity search via REST: una chiamata `search_similar` per tabella, poi fusione.""" def __init__(self, client: VectorRestClient): self.client = client + self._store = ThothHttpVectorStore(reader=client, writer=None) def search( self, query_vec: list[float], top_n: int = 10, kinds: list[str] | None = None ) -> list[VectorHit]: - hits: list[VectorHit] = [] - for table in tables_for_kinds(kinds): - for row in self.client.search_similar(table, query_vec, top_n, kinds=kinds): - hits.append(hit_from_metadata(row.get("similarity", 0.0), row.get("metadata"))) - # Il filtro per kind avviene server-side (RPC con `kinds`); il post-filter resta - # come difesa per il fallback legacy (server pre-migrazione: 404 -> query senza - # filtro) e per parita' col path diretto (#25). - if kinds: - allowed = set(kinds) - hits = [h for h in hits if h.kind in allowed] - return _merge(hits, top_n) + return self._store.search( + tables_for_kinds(kinds), query_vec, limit=top_n, kinds=kinds + ) class DirectSearcher: @@ -63,13 +54,11 @@ class DirectSearcher: self.engine = engine self.schema = schema self.dim = dim + self._store = LegacyDirectVectorStore(engine, schema=schema, dim=dim) def search( self, query_vec: list[float], top_n: int = 10, kinds: list[str] | None = None ) -> list[VectorHit]: - hits: list[VectorHit] = [] - for table in tables_for_kinds(kinds): - store = VectorStore(self.engine, schema=self.schema, table=table, dim=self.dim) - # passa kinds: dentro schema_records filtra schema_table vs schema_column (#25). - hits.extend(store.search(query_vec, top_n=top_n, kinds=kinds)) - return _merge(hits, top_n) + return self._store.search( + tables_for_kinds(kinds), query_vec, limit=top_n, kinds=kinds + ) diff --git a/harness/tht/vectorstore/rest_client.py b/harness/tht/vectorstore/rest_client.py index 45c006b0..dd7e1ac9 100644 --- a/harness/tht/vectorstore/rest_client.py +++ b/harness/tht/vectorstore/rest_client.py @@ -6,6 +6,7 @@ Errori in italiano e azionabili, stile `rest/client.py`. """ import requests +import re from tht.config import RestConfig @@ -60,6 +61,7 @@ class VectorRestClient: def search_similar( self, table_name: str, query_embedding: list[float], limit_count: int, kinds: list[str] | None = None, + metadata_filter: dict | None = None, ) -> list[dict]: """Ricerca per similarità coseno su `vectors.`: ritorna le righe `{id, similarity, metadata}` ordinate per similarity decrescente. Con `kinds` @@ -72,6 +74,13 @@ class VectorRestClient: "limit_count": limit_count, "table_name": table_name, } + if metadata_filter is not None: + # ACTIVE corpus reads must never degrade to an unfiltered legacy RPC: + # filtering after LIMIT is incomplete and could expose stale generations. + return self._call( + "search_similar", + {**args, "kinds": kinds, "metadata_filter": metadata_filter}, + ) or [] if kinds is not None: try: return self._call("search_similar", {**args, "kinds": kinds}) or [] @@ -118,3 +127,46 @@ class VectorRestClient: return int(payload[0]["upserted"]) return len(payload) return len(rows) + + def delete_generation(self, table_name: str, generation: str, workspace_id: str) -> int: + if table_name != "evidence" or re.fullmatch(r"gen:[0-9a-f]{32}", generation) is None: + raise ValueError("generation must be canonical") + if re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace_id) is None: + raise ValueError("workspace namespace must be canonical") + try: + payload = self._call( + "delete_vector_generation", + {"table_name": table_name, "kind": "evidence", "generation": generation, + "workspace_id": workspace_id}, + ) + except VectorRestError as error: + if "HTTP 404" in str(error): + raise VectorRestError( + "delete_vector_generation RPC is unavailable; deploy the cleanup migration" + ) from None + raise + if isinstance(payload, dict): + return int(payload.get("deleted", 0)) + return 0 + + def list_evidence_generations(self, table_name: str, workspace_id: str) -> list[str]: + if re.fullmatch(r"[a-z][a-z0-9_-]{0,63}", workspace_id) is None: + raise ValueError("workspace namespace must be canonical") + try: + rows = self._call( + "list_evidence_generations", + {"table_name": table_name, "kind": "evidence", "workspace_id": workspace_id}, + ) or [] + except VectorRestError as error: + if "HTTP 404" in str(error): + raise VectorRestError( + "list_evidence_generations RPC is unavailable; deploy the cleanup migration" + ) from None + raise + if not isinstance(rows, list) or any( + not isinstance(row, dict) + or re.fullmatch(r"gen:[0-9a-f]{32}", str(row.get("generation", ""))) is None + for row in rows + ): + raise VectorRestError("list_evidence_generations returned malformed data") + return sorted({row["generation"] for row in rows}) diff --git a/harness/workspaces/tht.example.yaml b/harness/workspaces/tht.example.yaml index b55c8664..1a3438fe 100644 --- a/harness/workspaces/tht.example.yaml +++ b/harness/workspaces/tht.example.yaml @@ -1,26 +1,19 @@ # Workspace ThothII (esempio). I segreti vivono SOLO in .env (${THT_*}). -# La struttura rispecchia esattamente tht/config.py: -# database + rest per il DWH; vector_rest/vector_write_rest per il pgvector (doppia key); -# vector_db per il loading diretto (server-only); embeddings + evidence + execution. +# Ogni risorsa dichiara il proprio adapter tramite `type`. language: it # descrizioni tabelle/colonne ed evidence sono in italiano (PSD) -database: - host: ${THT_DB_HOST} - port: ${THT_DB_PORT} # es. 5437 (Postgres diretto Supabase; 5432 = pooler) - database: ${THT_DB_NAME} # es. postgres (lo schema a stella vive in `datawarehouse`) - schema: datawarehouse - user: ${THT_DB_USER} - password: ${THT_DB_PASSWORD} - transport: rest # direct (Postgres) | rest (Supabase/PostgREST) +dwh: + type: thoth_rest # postgres_direct | thoth_rest + database: + database: ${THT_DB_NAME} + schema: datawarehouse + endpoint: + base_url: ${THT_DWH_REST_URL} + api_key: ${THT_DWH_API_KEY} + ssl_ca: ${THT_SSL_CA} -# Accesso al DWH via REST (richiesto se database.transport = rest). -rest: - base_url: ${THT_DWH_REST_URL} # es. https://supabase-aritmolab.policlinicosandonato.it/dwh/ - api_key: ${THT_DWH_API_KEY} # header X-API-Key, ruolo dwh_reader (read-only) - ssl_ca: ${THT_SSL_CA} # path al certificato CA (per server con CA interna) - -paths: +roots: artifacts: artifacts indexes: indexes sessions: sessions @@ -50,30 +43,23 @@ embeddings: dim: 768 batch_size: 32 -# LOADING del pgvector: connessione diretta, eseguita sul server (profilo server). -# Opzionale su postazione remota (lì la lettura passa da vector_rest). -vector_db: - host: ${THT_VEC_HOST} # Postgres locale del server - port: ${THT_VEC_PORT} # es. 5437 - database: postgres - schema: vectors - user: ${THT_VEC_USER} - password: ${THT_VEC_PASSWORD} - -# LETTURA (similarity search) del pgvector via REST remota: rpc search_similar. -vector_rest: - base_url: ${THT_VEC_REST_URL} # es. https://host/vector/v1/ - api_key: ${THT_VEC_API_KEY} # header X-API-Key, ruolo vector_reader (read-only) - ssl_ca: ${THT_SSL_CA} - -# SCRITTURA controllata del pgvector via REST remota: upsert/hash via RPC allowlist, -# niente delete/clear. Usa una API key SEPARATA dalla lettura (ruolo vector_writer). -# OPZIONALE: assente o key vuota = scrittura non abilitata (solo lettura). -# Abilita tht memory save-one / vector index-schema da postazione remota. -vector_write_rest: - base_url: ${THT_VEC_REST_URL} - api_key: ${THT_VEC_WRITE_API_KEY} - ssl_ca: ${THT_SSL_CA} +vectors: + type: thoth_vector_http # pgvector_direct | thoth_vector_http + reader: + base_url: ${THT_VEC_REST_URL} + api_key: ${THT_VEC_API_KEY} + ssl_ca: ${THT_SSL_CA} + writer: # opzionale: credenziale separata dalla lettura + base_url: ${THT_VEC_REST_URL} + api_key: ${THT_VEC_WRITE_API_KEY} + ssl_ca: ${THT_SSL_CA} + direct: # opzionale: loading server-side diretto + host: ${THT_VEC_HOST} + port: ${THT_VEC_PORT} + database: postgres + schema: vectors + user: ${THT_VEC_USER} + password: ${THT_VEC_PASSWORD} vector: max_chunk_chars: 4000 diff --git a/mkdocs.yml b/mkdocs.yml index a6f00504..42bd396a 100644 --- a/mkdocs.yml +++ b/mkdocs.yml @@ -47,6 +47,7 @@ nav: - Home: index.md - ThothII (Documentazione Tecnica): - Panoramica Architettura: architecture/overview.md + - Installazione Docker (4 contesti): installazione-docker-4-contesti.md - Specifiche di Design: - Architettura ThothII: superpowers/specs/2026-06-25-thothii-architecture-design.md - Backend: superpowers/specs/2026-06-27-backend-design.md diff --git a/scripts/bootstrap-local-psd-docker-config.sh b/scripts/bootstrap-local-psd-docker-config.sh new file mode 100644 index 00000000..1077ccfe --- /dev/null +++ b/scripts/bootstrap-local-psd-docker-config.sh @@ -0,0 +1,68 @@ +#!/bin/sh +set -eu + +root=$(CDPATH= cd -- "$(dirname -- "$0")/.." && pwd) +source_env=${1:-"$root/../../harness/.env"} +workspace=${2:-"$root/../../../tht-workspace-psd"} +auth_file=${3:-"$HOME/.pi/agent/auth.json"} + +value() { + awk -F= -v key="$1" '$1 == key { sub(/^[^=]*=/, ""); sub(/[[:space:]].*$/, ""); print; exit }' "$source_env" +} +required() { + result=$(value "$1") + [ -n "$result" ] || { echo "missing $1 in local source configuration" >&2; exit 2; } + printf '%s' "$result" +} + +test -f "$source_env" +test -d "$workspace" +test -f "$auth_file" +ca=$(value THT_SSL_CA) +[ -z "$ca" ] || test -f "$ca" +model_key=$(jq -er '.zai.key' "$auth_file") +test -n "$model_key" + +umask 077 +mkdir -p "$root/deploy/secrets" "$root/deploy/workspaces" +cp "$root/deploy/compose.psd-local.yaml.example" "$root/deploy/compose.psd-local.yaml" +cp "$root/deploy/workspaces/psd.yaml.example" "$root/deploy/workspaces/psd.yaml" +cat >"$root/.env" <"$root/deploy/secrets/thothii.secrets" <>"$root/deploy/compose.psd-local.yaml" <>"$root/deploy/secrets/thothii.secrets" +else + printf '%s\n' 'THT_CA=/etc/ssl/certs/ca-certificates.crt' >>"$root/deploy/secrets/thothii.secrets" +fi +chmod 600 "$root/.env" "$root/deploy/secrets/thothii.secrets" +echo "Local PSD Docker configuration materialized without printing secret values." diff --git a/scripts/local-vector-smoke.sh b/scripts/local-vector-smoke.sh new file mode 100755 index 00000000..071757ac --- /dev/null +++ b/scripts/local-vector-smoke.sh @@ -0,0 +1,436 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." + +mode=${1:-run} +case "$mode" in + run|--live-collision-test|--backup-restore) ;; + *) echo "usage: $0 [--live-collision-test|--backup-restore]" >&2; exit 2 ;; +esac + +keep_resources=${KEEP_SMOKE_RESOURCES:-0} +if [ "${SMOKE_PROJECT+x}" = x ]; then + echo "SMOKE_PROJECT is not accepted; the smoke always generates an owned namespace" >&2 + exit 2 +fi +secret_dir=$(mktemp -d "${TMPDIR:-/tmp}/thothii-vector-smoke.XXXXXX") +suffix=$(basename "$secret_dir" | tr -cd 'a-z0-9') +smoke_project="thothii-vector-smoke-$(date +%s)-$$-$suffix" +smoke_owner="$smoke_project-owner" +marker="local-vector-$smoke_project" +restore_container="${smoke_project}-restore" +restore_volume="${smoke_project}-restore-data" + +bootstrap_password="smoke-bootstrap-$smoke_project" +migrator_password="smoke-migrator-$smoke_project" +reader_password="smoke-reader-$smoke_project" +writer_password="smoke-writer-$smoke_project" +bundle="$secret_dir/thothii.secrets" +write_bundle() { + umask 077 + { + printf 'THT_VECTOR_BOOTSTRAP_PASSWORD=%s\n' "$bootstrap_password" + printf 'THT_VECTOR_MIGRATOR_PASSWORD=%s\n' "$migrator_password" + printf 'THT_VECTOR_READER_PASSWORD=%s\n' "$reader_password" + printf 'THT_VECTOR_WRITER_PASSWORD=%s\n' "$writer_password" + } >"$bundle" + chmod 0600 "$bundle" +} +write_bundle +export THT_SECRETS_FILE="$bundle" +# The rotation helper has an old/new file interface; these are test-only +# scratch files and are never mounted into a Compose service. +printf '%s' "$bootstrap_password" >"$secret_dir/bootstrap" +chmod 0600 "$secret_dir/bootstrap" +export THT_VECTOR_BOOTSTRAP_USER=thoth_bootstrap_smoke +export THOTH_SMOKE_OWNER="$smoke_owner" + +compose() { + docker compose -f compose.yaml -f deploy/compose.local-vector.yaml \ + --project-name "$smoke_project" --profile local-vector "$@" +} + +resource_ids() { + case "$1" in + container) docker ps -aq --filter "label=com.docker.compose.project=$smoke_project" ;; + volume) docker volume ls -q --filter "label=com.docker.compose.project=$smoke_project" ;; + network) docker network ls -q --filter "label=com.docker.compose.project=$smoke_project" ;; + esac +} + +resource_owner() { + case "$1" in + container) docker inspect --format '{{ index .Config.Labels "io.thothii.smoke-owner" }}' "$2" ;; + volume) docker volume inspect --format '{{ index .Labels "io.thothii.smoke-owner" }}' "$2" ;; + network) docker network inspect --format '{{ index .Labels "io.thothii.smoke-owner" }}' "$2" ;; + esac +} + +assert_no_collision() { + for kind in container volume network; do + ids=$(resource_ids "$kind") + if [ -n "$ids" ]; then + echo "refusing existing Compose project resources for generated namespace $smoke_project" >&2 + return 1 + fi + done +} + +verify_owned_resources() { + for kind in container volume network; do + for id in $(resource_ids "$kind"); do + owner=$(resource_owner "$kind" "$id" 2>/dev/null || true) + if [ "$owner" != "$smoke_owner" ]; then + echo "refusing cleanup of resource not owned by this smoke: $kind $id" >&2 + return 1 + fi + done + done +} + +cleanup() { + if [ "$keep_resources" = "1" ]; then + echo "Keeping smoke resources for project $smoke_project (KEEP_SMOKE_RESOURCES=1)." >&2 + else + if verify_owned_resources; then + docker rm -f "$restore_container" >/dev/null 2>&1 || true + docker volume rm "$restore_volume" >/dev/null 2>&1 || true + compose down --volumes >/dev/null 2>&1 || true + fi + fi + rm -rf "$secret_dir" +} +trap cleanup EXIT HUP INT TERM + +if [ "$mode" = "--live-collision-test" ]; then + collision_volume="${smoke_project}-collision" + docker volume create \ + --label "com.docker.compose.project=$smoke_project" \ + --label 'io.thothii.smoke-owner=foreign-owner' \ + "$collision_volume" >/dev/null + if assert_no_collision 2>/dev/null; then + echo "live collision probe was not detected" >&2 + docker volume rm "$collision_volume" >/dev/null + exit 1 + fi + docker volume rm "$collision_volume" >/dev/null + echo "live local-vector project collision refusal passed." + exit 0 +fi + +probe_vector() { + compose exec -T core sh -ec ' + . /app/docker/secret-policy.sh + tmp=$(mktemp -d); trap "rm -rf \"$tmp\"" EXIT + for role in READER WRITER; do + file="$tmp/$role" + read_bundle_secret /run/secrets/thothii.secrets "THT_VECTOR_${role}_PASSWORD" >"$file" + export "THT_VECTOR_${role}_PASSWORD_FILE=$file" + done + exec /opt/venv/bin/python - "$1" "$2" + ' sh "$marker" "$1" <<'PY' +import hashlib +import os +import sys + +from tht.adapters.vector.pgvector import PgVectorStore +from tht.config import DatabaseConfig +from tht.ports.vector import VectorWriteRecord +from tht.vectorstore.records import VectorRecord + +marker = sys.argv[1] +mode = sys.argv[2] +database = "thoth" +host = "vector-db" + +def credential(role: str) -> DatabaseConfig: + return DatabaseConfig( + host=host, + port=5432, + database=database, + schema="vectors", + user=f"thoth_vector_{role}", + password=open(os.environ[f"THT_VECTOR_{role.upper()}_PASSWORD_FILE"]).read(), + ) + +store = PgVectorStore(credential("reader"), credential("writer"), expected_dimension=768) +health = store.health() +assert health.ok, health +assert health.read_reachable is True and health.write_reachable is True, health + +record = VectorRecord( + id=marker, + kind="memory", + ref=marker, + title="Local vector persistence smoke", + content=marker, + metadata={"smoke": True}, +) +embedding = [1.0] + [0.0] * 767 +if mode == "write": + store.upsert( + "memory", + [VectorWriteRecord(record, embedding, hashlib.sha256(marker.encode()).hexdigest())], + ) +hits = store.search(["memory"], embedding, limit=1, kinds=["memory"]) +assert hits and hits[0].id == marker, hits +print(f"role health and persisted search passed for {marker} ({mode})") +PY +} + +assert_no_collision +compose config --quiet +services=$(compose config --services) +printf '%s\n' "$services" | grep -qx vector-db +printf '%s\n' "$services" | grep -qx vector-reconcile +printf '%s\n' "$services" | grep -qx vector-migrate + +compose up --build --wait vector-reconcile vector-migrate core +core_id=$(compose ps -q core) +inspect_env=$(docker inspect --format '{{json .Config.Env}}' "$core_id") +if printf '%s' "$inspect_env" | grep -q "smoke-\(reader\|writer\)-${smoke_project}"; then + echo "docker inspect exposed a direct vector password" >&2 + exit 1 +fi +printf '%s' "$inspect_env" | grep -q 'THT_SECRETS_FILE=/run/secrets/thothii.secrets' +migration_status=$(compose run --rm --no-deps vector-migrate) +printf '%s\n' "$migration_status" | grep -q '"pending": \[\]' +migrator_flags=$(compose run --rm --no-deps --entrypoint sh vector-reconcile -ec ' + . /opt/thoth/secret-policy.sh + export PGPASSWORD=$(read_bundle_secret /run/secrets/thothii.secrets THT_VECTOR_BOOTSTRAP_PASSWORD) + psql -At --host vector-db --username "$THT_VECTOR_BOOTSTRAP_USER" --dbname thoth \ + --command "SELECT (NOT rolcreaterole) AND (NOT rolcreatedb) AND (NOT rolsuper) FROM pg_roles WHERE rolname = '\''thoth_vector_migrator'\''" +') +test "$migrator_flags" = t +probe_vector write + +old_reader_password="$reader_password" +migrator_password="rotated-migrator-$smoke_project" +reader_password="rotated-reader-$smoke_project" +writer_password="rotated-writer-$smoke_project" +write_bundle + +compose run --rm vector-reconcile +rotation_status=$(compose run --rm --no-deps vector-migrate) +printf '%s\n' "$rotation_status" | grep -q '"pending": \[\]' +if compose run --rm --no-deps --entrypoint psql \ + -e PGPASSWORD="$old_reader_password" vector-reconcile \ + --host vector-db --username thoth_vector_reader --dbname thoth --command 'SELECT 1' \ + >/dev/null 2>&1; then + echo "old reader credential still works after rotation" >&2 + exit 1 +fi + +compose up --force-recreate --no-deps --wait core +probe_vector read + +old_bootstrap_password="$bootstrap_password" +printf '%s' "wrong-bootstrap-${smoke_project}" >"$secret_dir/bootstrap-wrong" +printf '%s' "next-bootstrap-'quoted-${smoke_project}" >"$secret_dir/bootstrap-next" +cp "$secret_dir/bootstrap" "$secret_dir/bootstrap-before-negative" +printf 'invalid bootstrap password\n' >"$secret_dir/bootstrap-whitespace" +chmod 0600 "$secret_dir/bootstrap-wrong" "$secret_dir/bootstrap-next" \ + "$secret_dir/bootstrap-before-negative" "$secret_dir/bootstrap-whitespace" +if COMPOSE_PROJECT_NAME="$smoke_project" \ + ./scripts/vector-rotate-bootstrap-password.sh \ + "$secret_dir/bootstrap" "$secret_dir/bootstrap-whitespace" \ + >/dev/null 2>&1; then + echo "bootstrap rotation accepted whitespace in a secret" >&2 + exit 1 +fi +cmp "$secret_dir/bootstrap" "$secret_dir/bootstrap-before-negative" +compose run --rm --no-deps --entrypoint psql \ + -e PGPASSWORD="$old_bootstrap_password" vector-reconcile \ + --host vector-db --username "$THT_VECTOR_BOOTSTRAP_USER" --dbname thoth \ + --command 'SELECT 1' >/dev/null + +if COMPOSE_PROJECT_NAME="$smoke_project" \ + ./scripts/vector-rotate-bootstrap-password.sh \ + "$secret_dir/bootstrap-wrong" "$secret_dir/bootstrap-next" \ + >/dev/null 2>&1; then + echo "bootstrap rotation accepted the wrong old secret" >&2 + exit 1 +fi +cmp "$secret_dir/bootstrap" "$secret_dir/bootstrap-before-negative" + +COMPOSE_PROJECT_NAME="$smoke_project" \ + ./scripts/vector-rotate-bootstrap-password.sh \ + "$secret_dir/bootstrap" "$secret_dir/bootstrap-next" +new_bootstrap_password=$(cat "$secret_dir/bootstrap") +bootstrap_password="$new_bootstrap_password" +write_bundle +test "$new_bootstrap_password" != "$old_bootstrap_password" +if compose run --rm --no-deps --entrypoint psql \ + -e PGPASSWORD="$old_bootstrap_password" vector-reconcile \ + --host vector-db --username "$THT_VECTOR_BOOTSTRAP_USER" --dbname thoth --command 'SELECT 1' \ + >/dev/null 2>&1; then + echo "old bootstrap credential still works after rotation" >&2 + exit 1 +fi +compose run --rm --no-deps --entrypoint psql \ + -e PGPASSWORD="$new_bootstrap_password" vector-reconcile \ + --host vector-db --username "$THT_VECTOR_BOOTSTRAP_USER" --dbname thoth --command 'SELECT 1' \ + >/dev/null +compose run --rm vector-reconcile +bootstrap_rotation_status=$(compose run --rm --no-deps vector-migrate) +printf '%s\n' "$bootstrap_rotation_status" | grep -q '"pending": \[\]' +compose up --force-recreate --no-deps --wait core +probe_vector read + +compose restart vector-db core +compose up --wait vector-db core +probe_vector read + +if [ "$mode" = "--backup-restore" ]; then + image=$(compose images -q vector-db) + network="${smoke_project}_default" + docker volume create \ + --label "com.docker.compose.project=$smoke_project" \ + --label "io.thothii.smoke-owner=$smoke_owner" "$restore_volume" >/dev/null + docker run -d --name "$restore_container" \ + --label "com.docker.compose.project=$smoke_project" \ + --label "io.thothii.smoke-owner=$smoke_owner" \ + --network "$network" --network-alias vector-db-restore \ + --mount "type=volume,source=$restore_volume,target=/var/lib/postgresql/data" \ + --mount "type=bind,source=$bundle,target=/run/secrets/thothii.secrets,readonly" \ + --mount "type=bind,source=$(pwd)/deploy/vector/vector-db-entrypoint.sh,target=/opt/thoth/vector-db-entrypoint.sh,readonly" \ + --mount "type=bind,source=$(pwd)/deploy/vector/secret-policy.sh,target=/opt/thoth/secret-policy.sh,readonly" \ + -e POSTGRES_DB=thoth -e POSTGRES_USER="$THT_VECTOR_BOOTSTRAP_USER" \ + -e THT_SECRETS_FILE=/run/secrets/thothii.secrets \ + --entrypoint /opt/thoth/vector-db-entrypoint.sh "$image" >/dev/null + attempts=0 + until docker exec "$restore_container" pg_isready \ + -U "$THT_VECTOR_BOOTSTRAP_USER" -d thoth >/dev/null 2>&1; do + attempts=$((attempts + 1)) + [ "$attempts" -lt 30 ] || { echo "restore database did not become ready" >&2; exit 1; } + sleep 1 + done + docker exec -e PGPASSWORD="$new_bootstrap_password" "$restore_container" psql -X \ + -U "$THT_VECTOR_BOOTSTRAP_USER" -d thoth -v ON_ERROR_STOP=1 --command \ + "CREATE SCHEMA vectors; CREATE EXTENSION vector WITH SCHEMA vectors; + CREATE TABLE vectors.memory ( + id bigserial PRIMARY KEY, record_key text UNIQUE NOT NULL, kind text NOT NULL, + content_hash text NOT NULL, metadata jsonb NOT NULL, + embedding vectors.vector(768) NOT NULL, indexed_at timestamptz NOT NULL DEFAULT now()); + INSERT INTO vectors.memory (record_key, kind, content_hash, metadata, embedding) + VALUES ('restore-sentinel', 'memory', 'sentinel-original', '{}', + ('[' || '1,' || repeat('0,', 766) || '0]')::vectors.vector);" >/dev/null + + docker run --rm --network "$network" \ + --mount "type=bind,source=$(pwd),target=/repo,readonly" \ + --mount "type=bind,source=$secret_dir,target=/scratch" "$image" \ + /repo/scripts/vector-backup.sh --host vector-db --database thoth \ + --user "$THT_VECTOR_BOOTSTRAP_USER" --password-file /scratch/bootstrap \ + --output /scratch/vector.dump + + compose exec -T vector-db sh -ec ' + . /opt/thoth/secret-policy.sh + export PGPASSWORD=$(read_bundle_secret /run/secrets/thothii.secrets THT_VECTOR_BOOTSTRAP_PASSWORD) + psql -X -U "$POSTGRES_USER" -d thoth -v ON_ERROR_STOP=1 --command \ + "UPDATE vectors.memory SET content_hash = '\''mutated-after-backup'\'' WHERE record_key = '\''$1'\''"' \ + sh "$marker" >/dev/null + + if docker run --rm --network "$network" \ + --mount "type=bind,source=$(pwd),target=/repo,readonly" \ + --mount "type=bind,source=$secret_dir,target=/scratch" "$image" \ + /repo/scripts/vector-restore.sh \ + --active-host vector-db --active-database thoth --active-user "$THT_VECTOR_BOOTSTRAP_USER" \ + --active-password-file /scratch/bootstrap \ + --target-host vector-db-restore --target-database thoth \ + --target-user "$THT_VECTOR_BOOTSTRAP_USER" --target-password-file /scratch/bootstrap \ + --input /scratch/vector.dump --force-nonempty >/dev/null 2>&1; then + echo "forced restore unexpectedly succeeded without archived ACL roles" >&2 + exit 1 + fi + sentinel=$(docker exec -e PGPASSWORD="$new_bootstrap_password" "$restore_container" psql \ + -XAt -U "$THT_VECTOR_BOOTSTRAP_USER" -d thoth --command \ + "SELECT content_hash FROM vectors.memory WHERE record_key='restore-sentinel'") + test "$sentinel" = sentinel-original + docker exec -e PGPASSWORD="$new_bootstrap_password" "$restore_container" psql -X \ + -U "$THT_VECTOR_BOOTSTRAP_USER" -d thoth -v ON_ERROR_STOP=1 --command \ + "DROP TABLE vectors.memory; CREATE ROLE vector_reader NOLOGIN; CREATE ROLE vector_writer NOLOGIN;" \ + >/dev/null + + docker run --rm --network "$network" \ + --mount "type=bind,source=$(pwd),target=/repo,readonly" \ + --mount "type=bind,source=$secret_dir,target=/scratch" "$image" \ + /repo/scripts/vector-restore.sh \ + --active-host vector-db --active-database thoth --active-user "$THT_VECTOR_BOOTSTRAP_USER" \ + --active-password-file /scratch/bootstrap \ + --target-host vector-db-restore --target-database thoth \ + --target-user "$THT_VECTOR_BOOTSTRAP_USER" --target-password-file /scratch/bootstrap \ + --input /scratch/vector.dump + + docker run --rm --network "$network" \ + --mount "type=bind,source=$(pwd)/deploy/vector/reconcile-roles.sh,target=/opt/thoth/reconcile-roles.sh,readonly" \ + --mount "type=bind,source=$(pwd)/deploy/vector/secret-policy.sh,target=/opt/thoth/secret-policy.sh,readonly" \ + --mount "type=bind,source=$bundle,target=/run/secrets/thothii.secrets,readonly" \ + -e PGHOST=vector-db-restore -e PGDATABASE=thoth \ + -e PGUSER="$THT_VECTOR_BOOTSTRAP_USER" \ + -e THT_SECRETS_FILE=/run/secrets/thothii.secrets \ + -e THT_VECTOR_MIGRATOR_USER=thoth_vector_migrator \ + -e THT_VECTOR_READER_USER=thoth_vector_reader \ + -e THT_VECTOR_WRITER_USER=thoth_vector_writer \ + --entrypoint /opt/thoth/reconcile-roles.sh "$image" >/dev/null + + compose exec -T core sh -ec ' + . /app/docker/secret-policy.sh + tmp=$(mktemp -d); trap "rm -rf \"$tmp\"" EXIT + for role in READER WRITER; do + file="$tmp/$role" + read_bundle_secret /run/secrets/thothii.secrets "THT_VECTOR_${role}_PASSWORD" >"$file" + export "THT_VECTOR_${role}_PASSWORD_FILE=$file" + done + exec /opt/venv/bin/python - "$1" + ' sh "$marker" <<'PY' +import hashlib +import os +import sys + +from tht.adapters.vector.pgvector import PgVectorStore +from tht.config import DatabaseConfig +from tht.ports.vector import VectorWriteRecord +from tht.vectorstore.records import VectorRecord + +def config(role): + return DatabaseConfig( + host="vector-db-restore", port=5432, database="thoth", schema="vectors", + user=f"thoth_vector_{role}", + password=open(os.environ[f"THT_VECTOR_{role.upper()}_PASSWORD_FILE"]).read(), + ) + +store = PgVectorStore(config("reader"), config("writer"), expected_dimension=768) +assert store.health().ok, store.health() +embedding = [1.0] + [0.0] * 767 +marker = sys.argv[1] +assert store.search(["memory"], embedding, limit=1, kinds=["memory"])[0].id == marker +write_id = marker + "-restore-write" +record = VectorRecord( + id=write_id, kind="memory", ref=write_id, title="restore writer", + content=write_id, metadata={}, +) +store.upsert("memory", [VectorWriteRecord(record, embedding, hashlib.sha256(write_id.encode()).hexdigest())]) +assert store.existing_hashes("memory", ["memory"])[write_id] +PY + + restored=$(docker exec -e PGPASSWORD="$new_bootstrap_password" "$restore_container" psql \ + -XAt -U "$THT_VECTOR_BOOTSTRAP_USER" -d thoth --command \ + "SELECT content_hash <> 'mutated-after-backup' FROM vectors.memory WHERE record_key = '$marker'") + test "$restored" = t + expected_migrations=$(find harness/tht/migrations/vector -type f -name '[0-9][0-9][0-9]_*.sql' \ + -exec basename {} \; | sed 's/_.*//' | sort | paste -sd, -) + applied_migrations=$(docker exec -e PGPASSWORD="$new_bootstrap_password" "$restore_container" psql \ + -XAt -U "$THT_VECTOR_BOOTSTRAP_USER" -d thoth --command \ + "SELECT string_agg(version, ',' ORDER BY version) FROM public.tht_vector_migrations") + test "$applied_migrations" = "$expected_migrations" + dimensions=$(docker exec -e PGPASSWORD="$new_bootstrap_password" "$restore_container" psql \ + -XAt -U "$THT_VECTOR_BOOTSTRAP_USER" -d thoth --command \ + "SELECT count(*) = 3 FROM pg_attribute a JOIN pg_class c ON c.oid=a.attrelid + JOIN pg_namespace n ON n.oid=c.relnamespace + WHERE n.nspname='vectors' AND a.attname='embedding' AND format_type(a.atttypid,a.atttypmod)='vectors.vector(768)'") + test "$dimensions" = t + echo "Transactional rollback and disposable-volume restore adapter parity passed." +fi + +echo "Local pgvector runtime/bootstrap rotation, least-privilege roles, and persistence passed." diff --git a/scripts/preprocess-smoke.sh b/scripts/preprocess-smoke.sh new file mode 100755 index 00000000..c9a6486d --- /dev/null +++ b/scripts/preprocess-smoke.sh @@ -0,0 +1,126 @@ +#!/bin/sh +set -eu +cd "$(dirname "$0")/.." + +if [ "${1:-}" = "--cleanup-failure" ] && [ -z "${PREPROCESS_SMOKE_CHILD:-}" ]; then + child_project="thoth-preprocess-failure-$$" + child_tmp=$(mktemp -d "${TMPDIR:-/tmp}/thoth-preprocess-failure.XXXXXX") + set +e + PREPROCESS_SMOKE_CHILD=1 PREPROCESS_SMOKE_INJECT_FAILURE=1 \ + PREPROCESS_SMOKE_PROJECT="$child_project" PREPROCESS_SMOKE_TMP="$child_tmp" "$0" + child_status=$? + set -e + test "$child_status" -eq 97 + test ! -e "$child_tmp" + test -z "$(docker ps -aq --filter "label=com.docker.compose.project=$child_project")" + test -z "$(docker volume ls -q --filter "label=com.docker.compose.project=$child_project")" + test -z "$(docker network ls -q --filter "label=com.docker.compose.project=$child_project")" + echo "injected preprocessing failure preserved status and cleaned every owned resource." + exit 0 +fi + +tmp=${PREPROCESS_SMOKE_TMP:-$(mktemp -d "${TMPDIR:-/tmp}/thoth-preprocess.XXXXXX")} +project=${PREPROCESS_SMOKE_PROJECT:-thoth-preprocess-$$} +compose="" +cleanup() { + original_status=$? + cleanup_failed=0 + set +e + if [ -n "$compose" ]; then + $compose down --volumes >/dev/null + test "$?" -eq 0 || cleanup_failed=1 + fi + test -z "$(docker ps -aq --filter "label=com.docker.compose.project=$project")" || cleanup_failed=1 + test -z "$(docker volume ls -q --filter "label=com.docker.compose.project=$project")" || cleanup_failed=1 + test -z "$(docker network ls -q --filter "label=com.docker.compose.project=$project")" || cleanup_failed=1 + rm -rf "$tmp" + test ! -e "$tmp" || cleanup_failed=1 + if [ "$original_status" -ne 0 ]; then + exit "$original_status" + fi + if [ "$cleanup_failed" -ne 0 ]; then + exit 1 + fi +} +trap cleanup EXIT +trap 'exit 129' HUP +trap 'exit 130' INT +trap 'exit 143' TERM +mkdir -p "$tmp/source/evidence" +printf '%s\n' '# Evidence' 'generation one' >"$tmp/source/evidence/a.md" +bundle="$tmp/thothii.secrets" +{ + printf 'THT_VECTOR_BOOTSTRAP_PASSWORD=smoke-bootstrap-%s\n' "$project" + printf 'THT_VECTOR_MIGRATOR_PASSWORD=smoke-migrator-%s\n' "$project" + printf 'THT_VECTOR_READER_PASSWORD=smoke-reader-%s\n' "$project" + printf 'THT_VECTOR_WRITER_PASSWORD=smoke-writer-%s\n' "$project" +} >"$bundle" +chmod 0600 "$bundle" +export THT_SECRETS_FILE="$bundle" +export THT_OLLAMA_URL=http://mock-embeddings:8081 + +cat >"$tmp/smoke.yaml" <>"$tmp/source/evidence/a.md" +third=$($compose run --rm preprocess-evidence) +after_third=$(generation_count) +dwh=$($compose run --rm preprocess-dwh) +python3 - "$first" "$second" "$third" "$before" "$after_first" "$after_second" "$after_third" <<'PY' +import json, sys +a, b, c = map(json.loads, sys.argv[1:4]) +before, first_count, second_count, third_count = map(int, sys.argv[4:]) +assert len(a["changed"]) == 1 and not a["unchanged"] +assert len(b["unchanged"]) == 1 and not b["changed"] +assert len(c["changed"]) == 1 and c["generation"] != a["generation"] +assert a["published"] and not b["published"] and c["published"] +assert first_count == before + 1 +assert second_count == first_count +assert third_count == second_count + 1 +PY +python3 - "$dwh" <<'PY' +import json, sys +assert json.loads(sys.argv[1])["status"] == "succeeded" +PY +active=$($compose run --rm --no-deps --entrypoint sh preprocess-evidence -c \ + 'cat /data/workspaces/preprocess-evidence/corpus/ACTIVE') +python3 - "$third" "$active" <<'PY' +import json, sys +assert json.loads(sys.argv[1])["generation"] == sys.argv[2].strip() +PY +echo "real Compose preprocessing unchanged rerun, mutation, DWH job, and ACTIVE publish passed." diff --git a/scripts/test-backend-url-policy.sh b/scripts/test-backend-url-policy.sh new file mode 100755 index 00000000..c82dd570 --- /dev/null +++ b/scripts/test-backend-url-policy.sh @@ -0,0 +1,23 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." + +policy=frontend/src/api/backend-url-policy.json +corpus=frontend/src/api/backend-url-cases.json + +jq -c '.[]' "$corpus" | while IFS= read -r case_json; do + value=$(printf '%s' "$case_json" | jq -r '.value') + valid=$(printf '%s' "$case_json" | jq -r '.valid') + if BACKEND_URL_POLICY_FILE="$policy" ./docker/validate-backend-url.sh "$value"; then + actual=true + else + actual=false + fi + if [ "$actual" != "$valid" ]; then + echo "shell policy mismatch for BACKEND_BASE_URL=$value: expected $valid" >&2 + exit 1 + fi +done + +echo "shell canonical URL corpus: ok" diff --git a/scripts/test-container-deployment.sh b/scripts/test-container-deployment.sh new file mode 100755 index 00000000..fb0965c1 --- /dev/null +++ b/scripts/test-container-deployment.sh @@ -0,0 +1,112 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." + +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM + +bundle="$tmp/thothii.secrets" +cat >"$bundle" <<'EOF' +# disposable deployment-contract bundle +THT_MODEL_API_KEY=test-model +THT_VECTOR_BOOTSTRAP_PASSWORD=contract-bootstrap +THT_VECTOR_MIGRATOR_PASSWORD=contract-migrator +THT_VECTOR_READER_PASSWORD=contract-reader +THT_VECTOR_WRITER_PASSWORD=contract-writer +EOF +chmod 0600 "$bundle" +export THT_SECRETS_FILE="$bundle" + +docker compose config >"$tmp/base.yaml" +grep -q '^ core:' "$tmp/base.yaml" +grep -q '^ frontend:' "$tmp/base.yaml" +grep -q 'host_ip: 127.0.0.1' "$tmp/base.yaml" +grep -q 'AUTH_MODE: none' "$tmp/base.yaml" +grep -q 'THOTH_PUBLIC_EXPOSURE: "false"' "$tmp/base.yaml" +grep -q 'THT_SECRETS_FILE: /run/secrets/thothii.secrets' "$tmp/base.yaml" +grep -q 'target: /home/thoth/.pi/agent/models.json' "$tmp/base.yaml" +grep -q 'source: .*/deploy/pi/models.json' "$tmp/base.yaml" +grep -q 'target: /home/thoth/.pi/agent/settings.json' "$tmp/base.yaml" +if grep -q 'THT_[A-Z0-9_]*_SECRET_FILE:' "$tmp/base.yaml"; then + echo "base Compose must not require legacy secret-file variables" >&2 + exit 1 +fi + +docker compose -f compose.yaml -f deploy/compose.local-vector.yaml \ + --profile local-vector config >"$tmp/local-vector.yaml" +grep -q 'target: thothii.secrets' "$tmp/local-vector.yaml" +if grep -Eq 'vector_(bootstrap|migrator|reader|writer)_password|THT_[A-Z0-9_]+_SECRET_FILE' "$tmp/local-vector.yaml"; then + echo "rendered local-vector config contains legacy per-secret references" >&2 + exit 1 +fi +if grep -q 'contract-' "$tmp/local-vector.yaml"; then + echo "rendered local-vector config leaked a bundle secret value" >&2 + exit 1 +fi + +docker compose -f compose.yaml -f deploy/compose.local.yaml \ + config >"$tmp/local.yaml" +if grep -q 'env_file:' "$tmp/local.yaml"; then + echo "local Compose must use the root .env interpolation file" >&2 + exit 1 +fi + +printf '%s\n' 'THT_MODEL_API_KEY=test-model' >"$tmp/thothii.secrets" +chmod 0600 "$tmp/thothii.secrets" +THT_SECRETS_FILE="$tmp/thothii.secrets" \ +THT_DB_NAME=test THT_DWH_REST_URL=https://dwh.example.test \ +THT_VEC_REST_URL=https://vector.example.test THT_OLLAMA_URL=https://embed.example.test \ + docker compose -f compose.yaml -f deploy/compose.production.yaml \ + config >"$tmp/production.yaml" +grep -q 'AUTH_MODE: upstream' "$tmp/production.yaml" +grep -q 'THOTH_PUBLIC_EXPOSURE: "true"' "$tmp/production.yaml" +grep -q 'THT_SECRETS_FILE: /run/secrets/thothii.secrets' "$tmp/production.yaml" +grep -q 'target: thothii.secrets' "$tmp/production.yaml" +if grep -q 'test-model' "$tmp/production.yaml"; then + echo "rendered production config leaked the model API key" >&2 + exit 1 +fi + +if PI_PROVIDER_API_KEY='must-not-leak' ./docker/core-entrypoint.sh doctor 2>"$tmp/legacy-model.err"; then + echo "legacy generic model credential was accepted" >&2 + exit 1 +fi +grep -q 'PI_PROVIDER_API_KEY is unsupported' "$tmp/legacy-model.err" +if grep -q 'must-not-leak' "$tmp/legacy-model.err"; then + echo "legacy model credential leaked through entrypoint diagnostics" >&2 + exit 1 +fi +printf 'THT_VECTOR_READER_PASSWORD=one\nTHT_VECTOR_READER_PASSWORD=two\n' >"$tmp/invalid-bundle" +chmod 0600 "$tmp/invalid-bundle" +if THT_SECRETS_FILE="$tmp/invalid-bundle" ./docker/core-entrypoint.sh doctor \ + >"$tmp/invalid-bundle.out" 2>"$tmp/invalid-bundle.err"; then + echo "entrypoint accepted an invalid secret bundle" >&2 + exit 1 +fi +grep -q 'THT_SECRETS_FILE points to an invalid secret bundle' "$tmp/invalid-bundle.err" +if grep -q 'THT_VECTOR_READER_PASSWORD' "$tmp/invalid-bundle.err"; then + echo "invalid bundle diagnostics leaked key material" >&2 + exit 1 +fi +before_tmp=$(find "${TMPDIR:-/tmp}" -maxdepth 1 -type d -name 'thothii-secrets.*' -print | sort) +THT_SECRETS_FILE="$bundle" ./docker/core-entrypoint.sh doctor >/dev/null 2>&1 || true +after_tmp=$(find "${TMPDIR:-/tmp}" -maxdepth 1 -type d -name 'thothii-secrets.*' -print | sort) +test "$before_tmp" = "$after_tmp" +if grep -Eq 'THT_VECTOR_(BOOTSTRAP|MIGRATOR|READER|WRITER)_PASSWORD_FILE|target: vector_(bootstrap|migrator|reader|writer)_password|dwh_api_key|model_api_key|THT_[A-Z0-9_]+_SECRET_FILE' "$tmp/production.yaml"; then + echo "production external config contains local direct vector secrets" >&2 + exit 1 +fi + +if awk '/^FROM / && $2 !~ /@sha256:/ { found=1 } END { exit !found }' \ + docker/core.Dockerfile docker/frontend.Dockerfile; then + echo "every Dockerfile base must include an immutable digest" >&2 + exit 1 +fi + +grep -qx 'deploy/\*' .dockerignore +grep -qx '!deploy/vector/' .dockerignore +grep -qx 'deploy/vector/\*' .dockerignore +grep -qx '!deploy/vector/secret-policy.sh' .dockerignore + +echo "container deployment security contract passed." diff --git a/scripts/test-default-compose.sh b/scripts/test-default-compose.sh new file mode 100755 index 00000000..b9fcd803 --- /dev/null +++ b/scripts/test-default-compose.sh @@ -0,0 +1,43 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." + +test -f .env.example +test -f deploy/secrets/thothii.secrets.example +grep -q '^docker compose up --build -d$' docs/installazione-docker-4-contesti.md +if grep -q 'cp deploy/env.example deploy/.env\|THT_[A-Z0-9_]*_SECRET_FILE=' docs/installazione-docker-4-contesti.md; then + echo "installation guide still presents the legacy per-file secret setup" >&2 + exit 1 +fi + +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM + +mkdir -p "$tmp/deploy/secrets" "$tmp/deploy/workspaces" +cp compose.yaml "$tmp/compose.yaml" +cp .env.example "$tmp/.env" +cp deploy/secrets/thothii.secrets.example "$tmp/deploy/secrets/thothii.secrets" +printf '%s\n' 'THT_MODEL_API_KEY=example-secret' >>"$tmp/deploy/secrets/thothii.secrets" +chmod 0600 "$tmp/deploy/secrets/thothii.secrets" + +services=$(docker compose --project-directory "$tmp" config --services) +[ "$services" = "core +frontend" ] || { + echo "default Compose services must be core and frontend (got: $services)" >&2 + exit 1 +} + +rendered=$(docker compose --project-directory "$tmp" config) +printf '%s\n' "$rendered" | grep -q 'target: thothii.secrets' +if printf '%s\n' "$rendered" | grep -Eq 'dwh_api_key|vector_reader_api_key|vector_writer_api_key|model_api_key|thoth_ca'; then + echo "default Compose must not declare legacy per-secret mounts" >&2 + exit 1 +fi +if printf '%s\n' "$rendered" | grep -Eq 'THT_[A-Z0-9_]+_SECRET_FILE:'; then + echo "default Compose must not require legacy secret-file variables" >&2 + exit 1 +fi +printf '%s\n' "$rendered" | grep -q 'THT_SECRETS_FILE: /run/secrets/thothii.secrets' + +echo "default Compose contract passed." diff --git a/scripts/test-docker-smoke.sh b/scripts/test-docker-smoke.sh new file mode 100755 index 00000000..0ae82c78 --- /dev/null +++ b/scripts/test-docker-smoke.sh @@ -0,0 +1,86 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." + +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM +log="$tmp/docker.log" +marker_file="$tmp/marker" + +mkdir -p "$tmp/bin" +cat >"$tmp/bin/docker" <<'EOF' +#!/bin/sh +set -eu +printf '%s\n' "$*" >>"$FAKE_DOCKER_LOG" + +case " $* " in + *" port frontend 8080 "*) printf '%s\n' '127.0.0.1:49152' ;; + *" exec -T core sh -c "*"printf"*) + for last do :; done + printf '%s\n' "$last" >"$FAKE_MARKER_FILE" + ;; + *" exec -T core sh -c "*"cat /data/.compose-smoke-marker"*) + cat "$FAKE_MARKER_FILE" + ;; +esac +EOF +cat >"$tmp/bin/curl" <<'EOF' +#!/bin/sh +set -eu +header_file="" +for arg do + if [ "${previous:-}" = "--dump-header" ]; then header_file=$arg; fi + previous=$arg +done +if [ -n "$header_file" ]; then + case "$*" in + *"/events"*) printf 'HTTP/1.1 200 OK\r\nContent-Type: text/event-stream\r\nCache-Control: no-cache\r\n\r\n' >"$header_file" ;; + *) printf 'HTTP/1.1 200 OK\r\nContent-Type: application/json; charset=utf-8\r\n\r\n' >"$header_file" ;; + esac +fi +case "$*" in + *"/health"*) printf '%s\n' '{"status":"ok"}' ;; +esac +EOF +chmod +x "$tmp/bin/docker" "$tmp/bin/curl" + +run_smoke() { + PATH="$tmp/bin:$PATH" \ + FAKE_DOCKER_LOG="$log" \ + FAKE_MARKER_FILE="$marker_file" \ + SMOKE_PROJECT="$1" \ + KEEP_SMOKE_RESOURCES="${2:-0}" \ + ./scripts/docker-smoke.sh +} + +run_smoke thothii-smoke-dynamic + +while IFS= read -r invocation; do + case "$invocation" in + "compose --project-name thothii-smoke-dynamic "*) ;; + *) echo "Compose invocation escaped the smoke project: $invocation" >&2; exit 1 ;; + esac +done <"$log" +grep -q ' down --volumes$' "$log" +if grep -q -- '--remove-orphans' "$log"; then + echo "smoke cleanup must not remove operator orphans" >&2 + exit 1 +fi + +: >"$log" +run_smoke thothii-smoke-kept 1 +if grep -q ' down ' "$log"; then + echo "KEEP_SMOKE_RESOURCES=1 unexpectedly cleaned the project" >&2 + exit 1 +fi + +: >"$log" +if PATH="$tmp/bin:$PATH" FAKE_DOCKER_LOG="$log" FAKE_MARKER_FILE="$marker_file" \ + SMOKE_PROJECT=thothii ./scripts/docker-smoke.sh >/dev/null 2>&1; then + echo "reserved operator project was accepted" >&2 + exit 1 +fi +test ! -s "$log" + +echo "docker-smoke dynamic isolation contract passed." diff --git a/scripts/test-external-compose-lifecycle.sh b/scripts/test-external-compose-lifecycle.sh new file mode 100755 index 00000000..5fca2274 --- /dev/null +++ b/scripts/test-external-compose-lifecycle.sh @@ -0,0 +1,23 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." +project="thothii-external-lifecycle-$$" +cleanup() { docker compose --project-name "$project" --profile external down --volumes >/dev/null 2>&1 || true; } +trap cleanup EXIT HUP INT TERM + +unset THT_VECTOR_BOOTSTRAP_PASSWORD_SECRET_FILE THT_VECTOR_MIGRATOR_PASSWORD_SECRET_FILE +unset THT_VECTOR_READER_PASSWORD_SECRET_FILE THT_VECTOR_WRITER_PASSWORD_SECRET_FILE +rendered=$(docker compose --project-name "$project" --profile external config) +if printf '%s' "$rendered" | grep -q 'THT_VECTOR_.*PASSWORD_FILE\|vector_.*password'; then + echo "external config contains local vector secret references" >&2 + exit 1 +fi +docker compose --project-name "$project" --profile external up --build --wait core +core=$(docker compose --project-name "$project" --profile external ps -q core) +inspect=$(docker inspect "$core") +if printf '%s' "$inspect" | grep -q 'THT_VECTOR_.*PASSWORD_FILE\|/run/secrets/vector_.*password'; then + echo "external core inspect contains local vector secret references" >&2 + exit 1 +fi +echo "external core lifecycle without local vector secrets passed." diff --git a/scripts/test-local-vector-smoke-live-collision.sh b/scripts/test-local-vector-smoke-live-collision.sh new file mode 100755 index 00000000..c7e047ad --- /dev/null +++ b/scripts/test-local-vector-smoke-live-collision.sh @@ -0,0 +1,5 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." +./scripts/local-vector-smoke.sh --live-collision-test diff --git a/scripts/test-local-vector-smoke-safety.sh b/scripts/test-local-vector-smoke-safety.sh new file mode 100755 index 00000000..f4bd4c1c --- /dev/null +++ b/scripts/test-local-vector-smoke-safety.sh @@ -0,0 +1,71 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM + +fake="$tmp/docker" +log="$tmp/docker.log" +state="$tmp/state" + +cat >"$fake" <<'SH' +#!/bin/sh +set -eu +printf '%s\n' "$*" >>"$FAKE_DOCKER_LOG" + +if [ "${FAKE_COLLISION:-0}" = 1 ] && [ "$1 $2" = "ps -aq" ]; then + printf '%s\n' collision-container + exit 0 +fi + +if [ "$1 $2" = "ps -aq" ] || [ "$1 $2" = "volume ls" ] || [ "$1 $2" = "network ls" ]; then + if [ "${FAKE_MISMATCH_ON_CLEANUP:-0}" = 1 ] && [ -f "$FAKE_DOCKER_STATE" ]; then + printf '%s\n' foreign-resource + fi + : >"$FAKE_DOCKER_STATE" + exit 0 +fi + +if [ "$1" = inspect ] || [ "$1 $2" = "volume inspect" ] || [ "$1 $2" = "network inspect" ]; then + printf '%s\n' foreign-owner + exit 0 +fi + +case "$*" in + *"config --services"*) printf '%s\n' vector-db vector-reconcile vector-migrate core frontend ;; + *"run --rm --no-deps vector-migrate"*) printf '%s\n' '{"applied":["001","002","003"],"drifted":[],"pending":[]}' ;; +esac +exit 0 +SH +chmod 0755 "$fake" + +if PATH="$tmp:$PATH" FAKE_DOCKER_LOG="$log" FAKE_DOCKER_STATE="$state" \ + SMOKE_PROJECT=operator-owned ./scripts/local-vector-smoke.sh >"$tmp/out" 2>"$tmp/err"; then + echo "smoke accepted caller-controlled SMOKE_PROJECT" >&2 + exit 1 +fi +grep -q 'SMOKE_PROJECT is not accepted' "$tmp/err" +test ! -s "$log" + +: >"$log" +rm -f "$state" +PATH="$tmp:$PATH" FAKE_DOCKER_LOG="$log" FAKE_DOCKER_STATE="$state" \ + FAKE_COLLISION=1 ./scripts/local-vector-smoke.sh >"$tmp/out" 2>"$tmp/err" || true +grep -q 'refusing existing Compose project resources' "$tmp/err" +if grep -q 'compose.*up' "$log"; then + echo "smoke started after detecting a project collision" >&2 + exit 1 +fi + +: >"$log" +rm -f "$state" +PATH="$tmp:$PATH" FAKE_DOCKER_LOG="$log" FAKE_DOCKER_STATE="$state" \ + FAKE_MISMATCH_ON_CLEANUP=1 ./scripts/local-vector-smoke.sh >"$tmp/out" 2>"$tmp/err" || true +grep -q 'refusing cleanup of resource not owned by this smoke' "$tmp/err" +if grep -q 'down --volumes' "$log"; then + echo "smoke removed resources after ownership mismatch" >&2 + exit 1 +fi + +echo "local-vector smoke collision and cleanup ownership contracts passed." diff --git a/scripts/test-preprocess-compose-config.sh b/scripts/test-preprocess-compose-config.sh new file mode 100755 index 00000000..3249247a --- /dev/null +++ b/scripts/test-preprocess-compose-config.sh @@ -0,0 +1,84 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." + +tmp_bundle=$(mktemp) +trap 'rm -f "$tmp_bundle"' EXIT HUP INT TERM +cat >"$tmp_bundle" <<'EOF' +THT_VECTOR_BOOTSTRAP_PASSWORD=test-bootstrap +THT_VECTOR_MIGRATOR_PASSWORD=test-migrator +THT_VECTOR_READER_PASSWORD=test-reader +THT_VECTOR_WRITER_PASSWORD=test-writer +EOF +chmod 0600 "$tmp_bundle" +export THT_SECRETS_FILE="$tmp_bundle" + +local_files="-f compose.yaml -f deploy/compose.local-vector.yaml -f deploy/compose.preprocess.yaml -f deploy/compose.preprocess-local-vector.yaml" +local_json=$(docker compose $local_files --profile local-vector --profile preprocess config --format json) + +printf '%s' "$local_json" | python3 -c ' +import json, sys + +config = json.load(sys.stdin) +services = config["services"] +assert "thothii_secrets" in config.get("secrets", {}), config.get("secrets") +assert "vector_bootstrap_password" not in config.get("secrets", {}) +assert "vector_migrator_password" not in config.get("secrets", {}) +assert "vector_reader_password" not in config.get("secrets", {}) +assert "vector_writer_password" not in config.get("secrets", {}) +for name, service in services.items(): + if name.startswith("vector-") or name.startswith("preprocess-") or name == "core": + assert any(item.get("target") == "thothii.secrets" for item in service.get("secrets", []) if isinstance(item, dict)), (name, service.get("secrets")) + assert "vector_reader_password" not in str(service) + assert "vector_writer_password" not in str(service) +for name in ("preprocess-evidence", "preprocess-dwh"): + dependency = services[name].get("depends_on", {}).get("vector-migrate") + assert dependency is not None, f"{name} does not depend on vector-migrate" + assert dependency["condition"] == "service_completed_successfully", dependency +' + +external_json=$(docker compose \ + -f compose.yaml -f deploy/compose.preprocess.yaml \ + --profile preprocess config --format json) + +printf '%s' "$external_json" | python3 -c ' +import json, sys + +config = json.load(sys.stdin) +services = config["services"] +assert "vector-db" not in services +assert "vector-migrate" not in services +assert "vector-reconcile" not in services +for name in ("preprocess-evidence", "preprocess-dwh"): + service = services[name] + assert "depends_on" not in service + assert all(item.get("target") == "thothii.secrets" for item in service.get("secrets", []) if isinstance(item, dict)), service.get("secrets") + assert "vector_reader_password" not in str(service) + assert "vector_writer_password" not in str(service) +' + +python3 - <<'PY' +import os +from pathlib import Path + +os.environ.update({ + "THT_DB_NAME": "thoth", + "THT_DWH_REST_URL": "http://dwh.invalid", + "THT_DWH_API_KEY": "dwh", + "THT_VECTOR_DATABASE": "thoth", + "THT_VECTOR_READER_USER": "reader", + "THT_VECTOR_WRITER_USER": "writer", + "THT_VECTOR_READER_PASSWORD_FILE": "/tmp/generated-reader", + "THT_VECTOR_WRITER_PASSWORD_FILE": "/tmp/generated-writer", + "THT_DOCS_ROOT": "/data/source", + "THT_OLLAMA_URL": "http://ollama.invalid", +}) +text = Path("deploy/workspaces/local-vector.yaml").read_text() +assert "password_file: ${THT_VECTOR_READER_PASSWORD_FILE}" in text +assert "password_file: ${THT_VECTOR_WRITER_PASSWORD_FILE}" in text +assert "${THT_SECRETS_FILE}" not in text +print("local-vector workspace resolution contract: ok") +PY + +echo "preprocess compose config: ok" diff --git a/scripts/test-vector-backup-restore-safety.sh b/scripts/test-vector-backup-restore-safety.sh new file mode 100755 index 00000000..76845eea --- /dev/null +++ b/scripts/test-vector-backup-restore-safety.sh @@ -0,0 +1,106 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM +fakebin="$tmp/bin" +mkdir "$fakebin" +printf '%s' secret >"$tmp/password" +chmod 0600 "$tmp/password" + +cat >"$fakebin/pg_dump" <<'SH' +#!/bin/sh +set -eu +for arg in "$@"; do case "$arg" in --file=*) output=${arg#--file=} ;; esac; done +printf 'custom dump' >"$output" +if [ -n "${RACE_OUTPUT:-}" ]; then + printf 'concurrent owner' >"$RACE_OUTPUT" +fi +SH +chmod 0755 "$fakebin/pg_dump" + +victim="$tmp/victim" +output="$tmp/vector.dump" +printf 'sentinel' >"$victim" +ln -s "$victim" "$output.partial" +PATH="$fakebin:$PATH" ./scripts/vector-backup.sh --host source --database thoth --user admin \ + --password-file "$tmp/password" --output "$output" >/dev/null +test "$(cat "$victim")" = sentinel +test "$(cat "$output")" = 'custom dump' +test -L "$output.partial" + +race_output="$tmp/raced.dump" +if PATH="$fakebin:$PATH" RACE_OUTPUT="$race_output" ./scripts/vector-backup.sh \ + --host source --database thoth --user admin --password-file "$tmp/password" \ + --output "$race_output" >"$tmp/race.out" 2>"$tmp/race.err"; then + echo "backup replaced a destination created concurrently" >&2 + exit 1 +fi +test "$(cat "$race_output")" = 'concurrent owner' +if find "$tmp" -name '.raced.dump.tmp.*' -print | grep -q .; then + echo "backup left its owned temporary archive after publication failure" >&2 + exit 1 +fi + +cat >"$fakebin/psql" <<'SH' +#!/bin/sh +set -eu +case "$*" in + *pg_control_system*) + echo same-cluster ;; + *) echo 0 ;; +esac +SH +cat >"$fakebin/pg_restore" <<'SH' +#!/bin/sh +printf '%s\n' "$*" >"$RESTORE_LOG" +SH +chmod 0755 "$fakebin/psql" "$fakebin/pg_restore" +printf 'archive' >"$tmp/input" +if PATH="$fakebin:$PATH" RESTORE_LOG="$tmp/restore.log" ./scripts/vector-restore.sh \ + --active-host source --active-database active --active-user admin \ + --active-password-file "$tmp/password" --target-host target --target-database restore \ + --target-user admin --target-password-file "$tmp/password" --input "$tmp/input" \ + >"$tmp/out" 2>"$tmp/err"; then + echo "restore accepted a target on the active PostgreSQL cluster" >&2 + exit 1 +fi +grep -q 'same PostgreSQL cluster' "$tmp/err" +test ! -e "$tmp/restore.log" + +cat >"$fakebin/psql" <<'SH' +#!/bin/sh +set -eu +case "$*" in + *pg_control_system*) + case "$*" in *--host=source*) echo same-cluster ;; *) echo other-cluster ;; esac ;; + *) echo 0 ;; +esac +SH +chmod 0755 "$fakebin/psql" +PATH="$fakebin:$PATH" RESTORE_LOG="$tmp/restore.log" ./scripts/vector-restore.sh \ + --active-host source --active-database active --active-user admin \ + --active-password-file "$tmp/password" --target-host target --target-database restore \ + --target-user admin --target-password-file "$tmp/password" --input "$tmp/input" >/dev/null +grep -q -- '--single-transaction' "$tmp/restore.log" +grep -q -- '--exit-on-error' "$tmp/restore.log" + +# The live restore smoke must follow the packaged migration set instead of a stale +# hard-coded count when a new migration is added. +if grep -Eq 'vector_(bootstrap|migrator|reader|writer)_password' \ + deploy/compose.local-vector.yaml deploy/compose.preprocess-local-vector.yaml; then + echo "local-vector Compose still declares legacy per-password secrets" >&2 + exit 1 +fi +grep -Fq 'thothii_secrets' deploy/compose.local-vector.yaml +grep -Fq 'thothii_secrets' deploy/compose.preprocess-local-vector.yaml +if grep -Fq 'SELECT count(*) = 3 FROM public.tht_vector_migrations' \ + scripts/local-vector-smoke.sh; then + echo "local vector smoke hard-codes the pre-004 migration count" >&2 + exit 1 +fi +grep -Fq 'expected_migrations=' scripts/local-vector-smoke.sh +grep -Fq 'applied_migrations=' scripts/local-vector-smoke.sh + +echo "vector backup/restore filesystem, identity, and transaction contracts passed." diff --git a/scripts/test-vector-bootstrap-rotation.sh b/scripts/test-vector-bootstrap-rotation.sh new file mode 100755 index 00000000..76db768a --- /dev/null +++ b/scripts/test-vector-bootstrap-rotation.sh @@ -0,0 +1,56 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM + +fake="$tmp/docker" +log="$tmp/docker.log" +cat >"$fake" <<'SH' +#!/bin/sh +set -eu +printf '%s:%s\n' "${THT_VECTOR_BOOTSTRAP_USER:-unset}" "$*" >>"$FAKE_DOCKER_LOG" +exit "${FAKE_DOCKER_EXIT:-0}" +SH +chmod 0755 "$fake" + +printf '%s' old-password >"$tmp/old" +printf '%s' "new-'quoted-\$-password" >"$tmp/new" +cp "$tmp/old" "$tmp/original" + +printf 'invalid password\n' >"$tmp/whitespace" +chmod 0600 "$tmp/old" "$tmp/new" "$tmp/original" "$tmp/whitespace" +: >"$log" +if PATH="$tmp:$PATH" FAKE_DOCKER_LOG="$log" THT_VECTOR_BOOTSTRAP_USER=custom_admin \ + ./scripts/vector-rotate-bootstrap-password.sh "$tmp/old" "$tmp/whitespace" \ + >"$tmp/out" 2>"$tmp/err"; then + echo "rotation accepted a whitespace-containing secret" >&2 + exit 1 +fi +cmp "$tmp/old" "$tmp/original" +test ! -s "$log" +if find "$tmp" -name 'old.rotate.*' -print | grep -q .; then + echo "rotation staged a deployment file before secret validation" >&2 + exit 1 +fi + +if PATH="$tmp:$PATH" FAKE_DOCKER_LOG="$log" FAKE_DOCKER_EXIT=1 \ + ./scripts/vector-rotate-bootstrap-password.sh "$tmp/old" "$tmp/new" \ + >"$tmp/out" 2>"$tmp/err"; then + echo "rotation unexpectedly succeeded when database verification failed" >&2 + exit 1 +fi +cmp "$tmp/old" "$tmp/original" + +: >"$log" +PATH="$tmp:$PATH" FAKE_DOCKER_LOG="$log" THT_VECTOR_BOOTSTRAP_USER=custom_admin \ + ./scripts/vector-rotate-bootstrap-password.sh "$tmp/old" "$tmp/new" \ + >"$tmp/out" 2>"$tmp/err" +cmp "$tmp/old" "$tmp/new" +grep -q '/run/secrets/bootstrap-old:ro' "$log" +grep -q '/run/secrets/bootstrap-new:ro' "$log" +grep -q '^custom_admin:' "$log" +grep -q 'atomically replaced only after verified database login' "$tmp/out" + +echo "bootstrap rotation ordering and no-config-change failure contracts passed." diff --git a/scripts/test-vector-migration-image.sh b/scripts/test-vector-migration-image.sh new file mode 100755 index 00000000..cfeffb97 --- /dev/null +++ b/scripts/test-vector-migration-image.sh @@ -0,0 +1,42 @@ +#!/bin/sh +set -eu + +image=${1:?usage: test-vector-migration-image.sh IMAGE [PLATFORM]} +platform=${2:-${PLATFORM:-linux/arm64}} +slug=$$ +network="thoth-vector-migration-$slug" +database="thoth-vector-db-$slug" + +cleanup() { + docker rm --force "$database" >/dev/null 2>&1 || true + docker network rm "$network" >/dev/null 2>&1 || true +} +trap cleanup EXIT INT TERM + +docker network create "$network" >/dev/null +docker run --detach --rm --platform "$platform" --name "$database" --network "$network" \ + -e POSTGRES_DB=thoth -e POSTGRES_USER=thoth_admin -e POSTGRES_PASSWORD=test-only \ + pgvector/pgvector:pg16 >/dev/null + +attempt=0 +until docker exec "$database" pg_isready -U thoth_admin -d thoth >/dev/null 2>&1; do + attempt=$((attempt + 1)) + if [ "$attempt" -ge 30 ]; then + echo "pgvector test database did not become ready" >&2 + exit 1 + fi + sleep 1 +done + +database_url="postgresql+psycopg2://thoth_admin:test-only@$database:5432/thoth" +applied=$(docker run --rm --platform "$platform" --network "$network" \ + --entrypoint /opt/venv/bin/tht -e THT_VECTOR_ADMIN_URL="$database_url" \ + "$image" vector migrate --json) +status=$(docker run --rm --platform "$platform" --network "$network" \ + --entrypoint /opt/venv/bin/tht -e THT_VECTOR_ADMIN_URL="$database_url" \ + "$image" vector migrate --status --json) + +expected='{"applied": ["001", "002", "003"], "drifted": [], "pending": []}' +test "$applied" = "$expected" +test "$status" = "$expected" +echo "core image vector migration discovery/status smoke passed" diff --git a/scripts/test-vector-secret-policy.sh b/scripts/test-vector-secret-policy.sh new file mode 100755 index 00000000..fc43fc49 --- /dev/null +++ b/scripts/test-vector-secret-policy.sh @@ -0,0 +1,58 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." +tmp=$(mktemp -d) +trap 'rm -rf "$tmp"' EXIT HUP INT TERM + +. ./deploy/vector/secret-policy.sh + +: >"$tmp/empty" +printf 'has newline\n' >"$tmp/newline" +printf 'has space' >"$tmp/space" +printf 'safe-quoted-\047-dollar-$' >"$tmp/valid" +printf 'docker-secret' >"$tmp/docker" +printf 'owner-readonly' >"$tmp/readonly" +printf 'too-open' >"$tmp/open" +printf '# comment\n\nTHT_VECTOR_READER_PASSWORD=reader\nTHT_VECTOR_WRITER_PASSWORD=writer\n' >"$tmp/bundle" +printf 'THT_VECTOR_READER_PASSWORD=reader\nTHT_VECTOR_WRITER_PASSWORD=writer\nTHT_DWH_API_KEY=one\nTHT_DWH_API_KEY=two\n' >"$tmp/duplicate-bundle" +printf 'THT_VECTOR_READER_PASSWORD=reader\r\nTHT_VECTOR_WRITER_PASSWORD=writer\r\n' >"$tmp/crlf-bundle" +awk 'BEGIN { printf "THT_VECTOR_READER_PASSWORD="; for (i = 1; i <= 16385; i++) printf "x"; print "" }' >"$tmp/long-line-bundle" +awk 'BEGIN { for (i = 1; i <= 70000; i++) print "# filler" }' >"$tmp/large-bundle" +chmod 0600 "$tmp/valid" +chmod 0444 "$tmp/docker" +chmod 0400 "$tmp/readonly" +chmod 0640 "$tmp/open" +chmod 0600 "$tmp/bundle" "$tmp/duplicate-bundle" "$tmp/crlf-bundle" "$tmp/long-line-bundle" "$tmp/large-bundle" + +for invalid in empty newline space; do + if validate_secret_file "$tmp/$invalid" "$invalid" >/dev/null 2>&1; then + echo "secret policy accepted $invalid" >&2 + exit 1 + fi +done +validate_secret_file "$tmp/valid" valid +validate_secret_file "$tmp/readonly" readonly +if validate_secret_file "$tmp/docker" docker >/dev/null 2>&1; then + echo "secret policy accepted world-readable host secret" >&2 + exit 1 +fi +if validate_secret_file "$tmp/open" open >/dev/null 2>&1; then + echo "secret policy accepted group-readable host secret" >&2 + exit 1 +fi +test "$(read_secret_file "$tmp/valid" valid)" = "safe-quoted-'-dollar-$" +test "$(read_bundle_secret "$tmp/bundle" THT_VECTOR_READER_PASSWORD)" = reader +test "$(read_bundle_secret "$tmp/crlf-bundle" THT_VECTOR_READER_PASSWORD)" = reader +if read_bundle_secret "$tmp/duplicate-bundle" THT_VECTOR_READER_PASSWORD >/dev/null 2>&1; then + echo "secret policy accepted a duplicate unrelated bundle key" >&2 + exit 1 +fi +for invalid_bundle in long-line-bundle large-bundle; do + if read_bundle_secret "$tmp/$invalid_bundle" THT_VECTOR_READER_PASSWORD >/dev/null 2>&1; then + echo "secret policy accepted oversized $invalid_bundle" >&2 + exit 1 + fi +done + +echo "shared vector secret policy contracts passed." diff --git a/scripts/vector-backup.sh b/scripts/vector-backup.sh new file mode 100755 index 00000000..61b6c586 --- /dev/null +++ b/scripts/vector-backup.sh @@ -0,0 +1,53 @@ +#!/bin/sh +set -eu + +root=$(CDPATH= cd -- "$(dirname "$0")/.." && pwd) +. "$root/deploy/vector/secret-policy.sh" + +usage() { + echo "usage: $0 --host HOST --database DB --user USER --password-file FILE --output FILE [--port PORT]" >&2 + exit 2 +} + +host= database= user= password_file= output= port=5432 +while [ "$#" -gt 0 ]; do + case "$1" in + --host) host=${2-}; shift 2 ;; + --port) port=${2-}; shift 2 ;; + --database) database=${2-}; shift 2 ;; + --user) user=${2-}; shift 2 ;; + --password-file) password_file=${2-}; shift 2 ;; + --output) output=${2-}; shift 2 ;; + *) usage ;; + esac +done +[ -n "$host" ] && [ -n "$database" ] && [ -n "$user" ] || usage +[ -n "$password_file" ] && [ -n "$output" ] || usage +validate_secret_file "$password_file" "backup password file" +[ ! -e "$output" ] || { echo "refusing to overwrite existing backup: $output" >&2; exit 2; } +output_dir=$(dirname "$output") +output_name=$(basename "$output") +[ -d "$output_dir" ] || { echo "backup destination directory does not exist" >&2; exit 2; } + +password=$(read_secret_file "$password_file" "backup password file") + +umask 077 +passfile=$(mktemp "${TMPDIR:-/tmp}/thoth-vector-pgpass.XXXXXX") +temporary_output=$(mktemp "$output_dir/.${output_name}.tmp.XXXXXX") +cleanup() { rm -f "$passfile" "$temporary_output"; } +trap cleanup EXIT HUP INT TERM +escaped=$(printf '%s' "$password" | sed 's/\\/\\\\/g; s/:/\\:/g') +printf '%s:%s:%s:%s:%s\n' "$host" "$port" "$database" "$user" "$escaped" >"$passfile" +chmod 0600 "$passfile" + +PGPASSFILE=$passfile pg_dump \ + --host="$host" --port="$port" --username="$user" --dbname="$database" \ + --format=custom --compress=9 \ + --table=vectors.schema_records --table=vectors.evidence --table=vectors.memory \ + --table=public.tht_vector_migrations --file="$temporary_output" +if ! ln "$temporary_output" "$output"; then + echo "refusing to replace backup destination created concurrently: $output" >&2 + exit 2 +fi +rm -f "$temporary_output" +echo "Vector backup written: $output" diff --git a/scripts/vector-restore.sh b/scripts/vector-restore.sh new file mode 100755 index 00000000..0e910544 --- /dev/null +++ b/scripts/vector-restore.sh @@ -0,0 +1,80 @@ +#!/bin/sh +set -eu + +root=$(CDPATH= cd -- "$(dirname "$0")/.." && pwd) +. "$root/deploy/vector/secret-policy.sh" + +usage() { + echo "usage: $0 --active-host HOST --active-database DB --active-user USER --active-password-file FILE --target-host HOST --target-database DB --target-user USER --target-password-file FILE --input FILE [--active-port PORT] [--target-port PORT] [--force-nonempty]" >&2 + exit 2 +} + +active_host= active_database= active_user= active_password_file= active_port=5432 +target_host= target_database= target_user= target_password_file= target_port=5432 +input= force=0 +while [ "$#" -gt 0 ]; do + case "$1" in + --active-host) active_host=${2-}; shift 2 ;; + --active-port) active_port=${2-}; shift 2 ;; + --active-database) active_database=${2-}; shift 2 ;; + --active-user) active_user=${2-}; shift 2 ;; + --active-password-file) active_password_file=${2-}; shift 2 ;; + --target-host) target_host=${2-}; shift 2 ;; + --target-port) target_port=${2-}; shift 2 ;; + --target-database) target_database=${2-}; shift 2 ;; + --target-user) target_user=${2-}; shift 2 ;; + --target-password-file) target_password_file=${2-}; shift 2 ;; + --input) input=${2-}; shift 2 ;; + --force-nonempty) force=1; shift ;; + *) usage ;; + esac +done +for value in "$active_host" "$active_database" "$active_user" "$active_password_file" \ + "$target_host" "$target_database" "$target_user" "$target_password_file" "$input"; do + [ -n "$value" ] || usage +done +[ -r "$input" ] || { echo "backup input is not readable" >&2; exit 2; } +validate_secret_file "$active_password_file" "active source password file" +validate_secret_file "$target_password_file" "target password file" + +umask 077 +active_pass=$(mktemp "${TMPDIR:-/tmp}/thoth-vector-active-pgpass.XXXXXX") +target_pass=$(mktemp "${TMPDIR:-/tmp}/thoth-vector-target-pgpass.XXXXXX") +cleanup() { rm -f "$active_pass" "$target_pass"; } +trap cleanup EXIT HUP INT TERM +make_passfile() { + secret=$(read_secret_file "$5" "database password file") + escaped=$(printf '%s' "$secret" | sed 's/\\/\\\\/g; s/:/\\:/g') + printf '%s:%s:%s:%s:%s\n' "$1" "$2" "$3" "$4" "$escaped" >"$6" + chmod 0600 "$6" +} +make_passfile "$active_host" "$active_port" "$active_database" "$active_user" \ + "$active_password_file" "$active_pass" +make_passfile "$target_host" "$target_port" "$target_database" "$target_user" \ + "$target_password_file" "$target_pass" + +identity_sql="SELECT system_identifier::text FROM pg_control_system()" +active_identity=$(PGPASSFILE=$active_pass psql -XAt --host="$active_host" --port="$active_port" \ + --username="$active_user" --dbname="$active_database" --command="$identity_sql") +target_identity=$(PGPASSFILE=$target_pass psql -XAt --host="$target_host" --port="$target_port" \ + --username="$target_user" --dbname="$target_database" --command="$identity_sql") +[ "$active_identity" != "$target_identity" ] || { + echo "refusing restore: active source and target are on the same PostgreSQL cluster" >&2 + exit 2 +} + +object_count=$(PGPASSFILE=$target_pass psql -XAt --host="$target_host" --port="$target_port" \ + --username="$target_user" --dbname="$target_database" --command=" + SELECT count(*) FROM pg_class c JOIN pg_namespace n ON n.oid=c.relnamespace + WHERE (n.nspname='vectors' OR (n.nspname='public' AND c.relname='tht_vector_migrations')) + AND c.relkind IN ('r','p','S','v','m');") +if [ "$object_count" != 0 ] && [ "$force" != 1 ]; then + echo "refusing restore into non-empty target; use --force-nonempty explicitly" >&2 + exit 2 +fi + +PGPASSFILE=$target_pass pg_restore --exit-on-error --single-transaction \ + --clean --if-exists --no-owner \ + --host="$target_host" --port="$target_port" --username="$target_user" \ + --dbname="$target_database" "$input" +echo "Vector restore completed into explicit target $target_host:$target_port/$target_database" diff --git a/scripts/vector-rotate-bootstrap-password.sh b/scripts/vector-rotate-bootstrap-password.sh new file mode 100755 index 00000000..ae2836d5 --- /dev/null +++ b/scripts/vector-rotate-bootstrap-password.sh @@ -0,0 +1,46 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." +. ./deploy/vector/secret-policy.sh + +if [ "$#" -ne 2 ]; then + echo "usage: $0 OLD_SECRET_FILE NEW_SECRET_FILE" >&2 + exit 2 +fi + +absolute_file() { + directory=$(CDPATH= cd -- "$(dirname -- "$1")" && pwd) + printf '%s/%s\n' "$directory" "$(basename -- "$1")" +} + +old_secret=$(absolute_file "$1") +new_secret=$(absolute_file "$2") +validate_secret_file "$old_secret" old_bootstrap_secret +validate_secret_file "$new_secret" new_bootstrap_secret +if [ "$old_secret" -ef "$new_secret" ]; then + echo "old and new secret files must be distinct" >&2 + exit 2 +fi + +project=${COMPOSE_PROJECT_NAME:-thothii} +replacement=$(mktemp "${old_secret}.rotate.XXXXXX") +trap 'rm -f "$replacement"' EXIT HUP INT TERM +cp "$new_secret" "$replacement" +chmod 0600 "$replacement" + +docker compose -f compose.yaml -f deploy/compose.local-vector.yaml \ + --project-name "$project" --profile local-vector run --rm --no-deps \ + --user 0:0 \ + --entrypoint /opt/venv/bin/python \ + --volume "$old_secret:/run/secrets/bootstrap-old:ro" \ + --volume "$new_secret:/run/secrets/bootstrap-new:ro" \ + --volume "$(pwd)/deploy/vector/rotate-bootstrap-password.py:/opt/thoth/rotate-bootstrap-password.py:ro" \ + core /opt/thoth/rotate-bootstrap-password.py \ + /run/secrets/bootstrap-old /run/secrets/bootstrap-new + +mv -f "$replacement" "$old_secret" +trap - EXIT HUP INT TERM + +echo "Deployment bootstrap secret atomically replaced only after verified database login." +echo "Re-run: docker compose -f compose.yaml -f deploy/compose.local-vector.yaml --project-name $project --profile local-vector up --wait vector-reconcile vector-migrate core" diff --git a/scripts/verify-container-images.sh b/scripts/verify-container-images.sh new file mode 100755 index 00000000..eee3da93 --- /dev/null +++ b/scripts/verify-container-images.sh @@ -0,0 +1,52 @@ +#!/bin/sh +set -eu + +cd "$(dirname "$0")/.." + +platform=${PLATFORM:-linux/arm64} +slug=$(printf '%s' "$platform" | tr '/:' '--') +core_image="thothii-core:verify-$slug" +frontend_image="thothii-frontend:verify-$slug" +inventory_dir=${CONTAINER_INVENTORY_DIR:-.artifacts/container-images/$slug} + +mkdir -p "$inventory_dir" + +docker buildx build --platform "$platform" --load \ + -f docker/core.Dockerfile -t "$core_image" . +docker buildx build --platform "$platform" --load \ + -f docker/frontend.Dockerfile -t "$frontend_image" . + +docker run --rm --platform "$platform" --entrypoint /app/docker/smoke/core-smoke.sh \ + "$core_image" +./scripts/test-vector-migration-image.sh "$core_image" "$platform" +docker run --rm --platform "$platform" -e BACKEND_BASE_URL=/api \ + "$frontend_image" frontend-config-smoke +docker run --rm --platform "$platform" -e BACKEND_BASE_URL= \ + "$frontend_image" frontend-config-smoke +docker run --rm --platform "$platform" --entrypoint frontend-policy-smoke "$frontend_image" + +if docker run --rm --platform "$platform" -e BACKEND_BASE_URL=/backend \ + "$frontend_image" frontend-config-smoke >/dev/null 2>&1; then + echo "frontend accepted an unsupported BACKEND_BASE_URL" >&2 + exit 1 +fi +if docker run --rm --platform "$platform" -e THOTH_PUBLIC_EXPOSURE=true -e AUTH_MODE=none \ + "$core_image" server >/dev/null 2>&1; then + echo "core accepted public exposure without upstream authentication" >&2 + exit 1 +fi + +docker image inspect "$core_image" >"$inventory_dir/core-image-inspect.json" +docker image inspect "$frontend_image" >"$inventory_dir/frontend-image-inspect.json" +docker run --rm --platform "$platform" --entrypoint sh "$core_image" -c \ + 'dpkg-query -W; /opt/venv/bin/pip freeze; /opt/venv/bin/python -c '"'"'import glob,json; rows=set(); +for path in glob.glob("/app/backend/node_modules/**/package.json", recursive=True): + try: + package=json.load(open(path)); rows.add((package.get("name","?"), package.get("version","?"))) + except (OSError, ValueError): pass +print("\n".join(f"{name}=={version}" for name,version in sorted(rows)))'"'"'' \ + >"$inventory_dir/core-packages.txt" +docker run --rm --platform "$platform" --entrypoint sh "$frontend_image" -c 'apk info -vv' \ + >"$inventory_dir/frontend-packages.txt" + +echo "container verification and inventory complete for $platform"