feat: add canonical workspace schema
This commit is contained in:
@@ -0,0 +1,138 @@
|
||||
import { parseAllDocuments, stringify } from "yaml";
|
||||
import { z } from "zod";
|
||||
|
||||
export const DWH_TRANSPORTS = ["postgres_direct", "rest_api", "ssh_tunnel"] as const;
|
||||
export type DwhTransport = (typeof DWH_TRANSPORTS)[number];
|
||||
|
||||
export const VECTOR_TRANSPORTS = ["pgvector_direct", "rest_api", "ssh_tunnel"] as const;
|
||||
export type VectorTransport = (typeof VECTOR_TRANSPORTS)[number];
|
||||
|
||||
export interface CanonicalWorkspace {
|
||||
workspace: {
|
||||
schema_version: 1;
|
||||
id: string;
|
||||
name: string;
|
||||
description?: string;
|
||||
language: "en" | "it";
|
||||
};
|
||||
dwh: {
|
||||
engine: "postgres";
|
||||
database: string;
|
||||
schema: string;
|
||||
supported_transports: DwhTransport[];
|
||||
};
|
||||
semantic_index: {
|
||||
vector_store: {
|
||||
engine: "pgvector";
|
||||
collection: string;
|
||||
dimensions: number;
|
||||
distance: "cosine" | "l2" | "inner_product";
|
||||
supported_transports: VectorTransport[];
|
||||
};
|
||||
embedding: {
|
||||
provider: "ollama_compatible" | "openai_compatible";
|
||||
model: string;
|
||||
dimensions: number;
|
||||
};
|
||||
};
|
||||
llm_policy: {
|
||||
default?: `${string}/${string}`;
|
||||
allowed: `${string}/${string}`[];
|
||||
};
|
||||
}
|
||||
|
||||
const workspaceId = z.string().regex(/^[a-z][a-z0-9-]{2,62}$/, {
|
||||
message: "workspace id must match ^[a-z][a-z0-9-]{2,62}$",
|
||||
});
|
||||
const identifier = z.string().regex(/^[A-Za-z_][A-Za-z0-9_]*$/, {
|
||||
message: "database identifiers must start with a letter or underscore",
|
||||
});
|
||||
const dimensions = z.number().int().positive().max(32_768);
|
||||
const modelReference = z.string().regex(/^[^/\s]+\/[^/\s]+$/, {
|
||||
message: "model must use provider/model syntax",
|
||||
});
|
||||
|
||||
function unique<T>(values: readonly T[], context: z.RefinementCtx, path: PropertyKey[]) {
|
||||
if (new Set(values).size !== values.length) {
|
||||
context.addIssue({ code: "custom", path, message: "supported transports must not repeat" });
|
||||
}
|
||||
}
|
||||
|
||||
const WorkspaceSchema = z.object({
|
||||
workspace: z.object({
|
||||
schema_version: z.literal(1),
|
||||
id: workspaceId,
|
||||
name: z.string().trim().min(1),
|
||||
description: z.string().trim().min(1).optional(),
|
||||
language: z.enum(["en", "it"]),
|
||||
}).strict(),
|
||||
dwh: z.object({
|
||||
engine: z.literal("postgres"),
|
||||
database: identifier,
|
||||
schema: identifier,
|
||||
supported_transports: z.array(z.enum(DWH_TRANSPORTS)).min(1),
|
||||
}).strict(),
|
||||
semantic_index: z.object({
|
||||
vector_store: z.object({
|
||||
engine: z.literal("pgvector"),
|
||||
collection: identifier,
|
||||
dimensions,
|
||||
distance: z.enum(["cosine", "l2", "inner_product"]),
|
||||
supported_transports: z.array(z.enum(VECTOR_TRANSPORTS)).min(1),
|
||||
}).strict(),
|
||||
embedding: z.object({
|
||||
provider: z.enum(["ollama_compatible", "openai_compatible"]),
|
||||
model: z.string().trim().min(1),
|
||||
dimensions,
|
||||
}).strict(),
|
||||
}).strict(),
|
||||
llm_policy: z.object({
|
||||
default: modelReference.optional(),
|
||||
allowed: z.array(modelReference).min(1),
|
||||
}).strict(),
|
||||
}).strict().superRefine((workspace, context) => {
|
||||
unique(workspace.dwh.supported_transports, context, ["dwh", "supported_transports"]);
|
||||
unique(
|
||||
workspace.semantic_index.vector_store.supported_transports,
|
||||
context,
|
||||
["semantic_index", "vector_store", "supported_transports"],
|
||||
);
|
||||
unique(workspace.llm_policy.allowed, context, ["llm_policy", "allowed"]);
|
||||
|
||||
if (workspace.semantic_index.vector_store.dimensions !== workspace.semantic_index.embedding.dimensions) {
|
||||
context.addIssue({
|
||||
code: "custom",
|
||||
path: ["semantic_index", "embedding", "dimensions"],
|
||||
message: "embedding dimensions must match vector store dimensions",
|
||||
});
|
||||
}
|
||||
|
||||
if (workspace.llm_policy.default && !workspace.llm_policy.allowed.includes(workspace.llm_policy.default)) {
|
||||
context.addIssue({
|
||||
code: "custom",
|
||||
path: ["llm_policy", "default"],
|
||||
message: "LLM default must be included in the allowlist",
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
export function parseWorkspaceYaml(source: string): CanonicalWorkspace {
|
||||
const documents = parseAllDocuments(source, { uniqueKeys: true });
|
||||
if (documents.length !== 1) {
|
||||
throw new Error("Workspace YAML must contain exactly one document");
|
||||
}
|
||||
|
||||
const document = documents[0];
|
||||
if (document.errors.length > 0) {
|
||||
throw new Error(`Invalid workspace YAML: ${document.errors.map((error) => error.message).join("; ")}`);
|
||||
}
|
||||
|
||||
return WorkspaceSchema.parse(document.toJSON()) as CanonicalWorkspace;
|
||||
}
|
||||
|
||||
export function serializeWorkspaceYaml(workspace: CanonicalWorkspace): string {
|
||||
const canonical = WorkspaceSchema.parse(workspace) as CanonicalWorkspace;
|
||||
return stringify(canonical, { lineWidth: 0, sortMapEntries: true });
|
||||
}
|
||||
|
||||
export { buildInstallationContract, renderWorkspaceDocs } from "./contracts.js";
|
||||
Reference in New Issue
Block a user