feat: protect sensitive catalog samples

This commit is contained in:
Codex
2026-08-30 12:14:23 +02:00
parent 6278ee9d81
commit 0736983bc5
28 changed files with 1162 additions and 128 deletions
@@ -222,7 +222,7 @@ test("adds only bounded transient source samples to the model request", async ()
tables: [{ name: "patients", sourceComment: null }],
columns: [{
tableName: "patients",
name: "status",
name: "patient_email",
ordinalPosition: 1,
dataType: "text",
isNullable: true,
@@ -243,9 +243,18 @@ test("adds only bounded transient source samples to the model request", async ()
});
const table = (await repository.listTables(database.id))[0]!;
const columns = await repository.listColumns(database.id, table.id);
const column = columns.find((candidate) => candidate.name === "status")!;
const column = columns.find((candidate) => candidate.name === "patient_email")!;
const ward = columns.find((candidate) => candidate.name === "ward")!;
const sampleSecret = "ONLY_IN_TRANSIENT_SAMPLE_7f29c8";
await repository.updateColumnMetadata(
database.id,
table.id,
column.id,
column.version,
column.description,
column.generatedDescription,
true,
);
const sampleSecret = "real.patient@hospital.invalid";
const sourceSampler: DescriptionSourceSampler = {
sample: vi.fn(async () => [{
targetId: column.id,
@@ -265,11 +274,14 @@ test("adds only bounded transient source samples to the model request", async ()
rows: [
{ fields: [{ name: ward.name, value: "row-4" }] },
{ fields: [{ name: ward.name, value: "row-5" }] },
{ fields: [{ name: ward.name, value: "row-6-must-be-omitted" }] },
{ fields: [{ name: ward.name, value: "row-6" }] },
{ fields: [{ name: ward.name, value: "row-7" }] },
{ fields: [{ name: ward.name, value: "row-8" }] },
{ fields: [{ name: ward.name, value: "row-9-must-be-omitted" }] },
],
representativeValues: [{
column: ward.name,
values: ["ward-1", "ward-2", "ward-3-must-be-omitted"],
values: ["ward-1", "ward-2", "ward-3", "ward-4", "ward-5", "ward-6-must-be-omitted"],
}],
}]),
};
@@ -321,7 +333,7 @@ test("adds only bounded transient source samples to the model request", async ()
expect(sourceSampler.sample).toHaveBeenCalledWith(
expect.objectContaining({ id: database.id, binding: database.binding }),
[
{ targetId: column.id, tableName: table.name, columnNames: [column.name] },
{ targetId: column.id, tableName: table.name, columnNames: [] },
{ targetId: ward.id, tableName: table.name, columnNames: [ward.name] },
],
expect.any(AbortSignal),
@@ -338,21 +350,29 @@ test("adds only bounded transient source samples to the model request", async ()
targetContext.sourceSample?.representativeValues.flatMap((entry) => entry.values) ?? []
),
);
expect(sampledRows).toHaveLength(5);
expect(representativeValues).toHaveLength(5);
expect(context.targets[0].sourceSample.rows).toHaveLength(3);
expect(context.targets[1].sourceSample.rows).toHaveLength(2);
expect(sampledRows).toHaveLength(10);
expect(representativeValues).toHaveLength(10);
expect(context.targets[0].sourceSample.rows).toHaveLength(5);
expect(context.targets[1].sourceSample.rows).toHaveLength(5);
expect(context.targets[0].sourceSample.representativeValues).toEqual([{
column: column.name,
values: [sampleSecret, "two", "three"],
values: [
"marta.rossi@example.com",
"luca.bianchi@example.com",
"elena.conti@example.com",
"paolo.romano@example.com",
"giulia.ferrari@example.com",
],
}]);
expect(context.targets[1].sourceSample.representativeValues).toEqual([{
column: ward.name,
values: ["ward-1", "ward-2"],
values: ["ward-1", "ward-2", "ward-3", "ward-4", "ward-5"],
}]);
expect(userMessage).toContain(sampleSecret);
expect(userMessage).not.toContain(sampleSecret);
expect(userMessage).not.toMatch(/synthetic|fake|fittizi/i);
expect(userMessage).toContain("marta.rossi@example.com");
expect(userMessage).not.toMatch(
/row-6-must-be-omitted|ward-3-must-be-omitted/,
/row-9-must-be-omitted|ward-6-must-be-omitted/,
);
const persisted = JSON.stringify({
@@ -366,6 +386,173 @@ test("adds only bounded transient source samples to the model request", async ()
expect(persisted).not.toContain(sampleSecret);
});
test("gives every sensitive column synthetic context without consuming the real sample budget", async () => {
const repository = new MemoryCatalogRepository();
const database = await repository.create({
workspaceId: "psd-clinical",
engine: "postgres",
databaseName: "warehouse",
schema: "datawarehouse",
binding: { transport: "postgres_direct", host: "db.internal", port: 5432, username: "reader" },
});
await repository.applySchemaSync(database.id, database.version, "all", [], {
schemaVersion: 1,
capabilities: { tables: "available", columns: "available", relationships: "available" },
tables: [{ name: "patients", sourceComment: null }],
columns: [{
tableName: "patients",
name: "patient_email",
ordinalPosition: 1,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
sourceComment: null,
}, {
tableName: "patients",
name: "patient_phone",
ordinalPosition: 2,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
sourceComment: null,
}, {
tableName: "patients",
name: "ward",
ordinalPosition: 3,
dataType: "text",
isNullable: true,
defaultExpression: null,
primaryKeyPosition: null,
sourceComment: null,
}],
relationships: [],
});
const table = (await repository.listTables(database.id))[0]!;
const columns = await repository.listColumns(database.id, table.id);
const email = columns.find((column) => column.name === "patient_email")!;
const phone = columns.find((column) => column.name === "patient_phone")!;
const ward = columns.find((column) => column.name === "ward")!;
for (const column of [email, phone]) {
await repository.updateColumnMetadata(
database.id,
table.id,
column.id,
column.version,
column.description,
column.generatedDescription,
true,
);
}
const wardValues = ["ward-a", "ward-b", "ward-c", "ward-d", "ward-e"];
const sourceSampler: DescriptionSourceSampler = {
sample: vi.fn(async (_database, targets) => targets.map((target) => {
if (target.columnNames.length === 0) {
return {
targetId: target.targetId,
tableName: target.tableName,
rows: [],
representativeValues: [],
};
}
const columnName = target.columnNames[0]!;
return {
targetId: target.targetId,
tableName: target.tableName,
rows: wardValues.map((value) => ({ fields: [{ name: columnName, value }] })),
representativeValues: [{ column: columnName, values: wardValues }],
};
})),
};
const completer: ModelCompleter = {
complete: vi.fn(async (request) => {
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
return JSON.stringify({
results: context.targets.map((target: { targetId: string }) => ({
targetId: target.targetId,
outcome: "generated",
description: "Descrizione generata.",
})),
});
}),
};
const models: MetadataGenerationModels = {
catalog: () => ({ models: [{ id: "openai-mini", label: "OpenAI Mini" }], default: "openai-mini" }),
resolve: () => ({
id: "openai-mini",
provider: "openai",
model: "gpt-4.1-mini",
apiKeyEnv: "OPENAI_API_KEY",
apiKey: "test-provider-secret",
}),
};
const worker = new DescriptionGenerationWorker(
repository,
{
read: vi.fn(async () => ({
workspace: { workspace: { language: "it" } },
revision: {},
})),
} as unknown as WorkspaceRegistry,
models,
completer,
new CatalogOperationCoordinator(),
sourceSampler,
);
const run = await worker.start(
database.id,
"openai-mini",
"selected_columns",
[email.id, phone.id, ward.id],
);
await worker.waitForRun(run.id);
expect(sourceSampler.sample).toHaveBeenCalledWith(
expect.objectContaining({ id: database.id }),
[
{ targetId: email.id, tableName: table.name, columnNames: [] },
{ targetId: phone.id, tableName: table.name, columnNames: [] },
{ targetId: ward.id, tableName: table.name, columnNames: [ward.name] },
],
expect.any(AbortSignal),
);
const request = vi.mocked(completer.complete).mock.calls[0]![0] as ModelCompletionRequest;
const context = JSON.parse(request.messages[1]!.content.split("\n").slice(1).join("\n"));
const targets = new Map(
context.targets.map((target: { targetId: string }) => [target.targetId, target]),
);
expect(targets.get(email.id)).toMatchObject({
sourceSample: {
rows: expect.arrayContaining([
{ fields: [{ name: email.name, value: "marta.rossi@example.com" }] },
]),
representativeValues: [{
column: email.name,
values: expect.arrayContaining(["marta.rossi@example.com"]),
}],
},
});
expect(targets.get(phone.id)).toMatchObject({
sourceSample: {
rows: expect.arrayContaining([
{ fields: [{ name: phone.name, value: "+39 02 5550 1001" }] },
]),
representativeValues: [{
column: phone.name,
values: expect.arrayContaining(["+39 02 5550 1001"]),
}],
},
});
expect(targets.get(ward.id)).toMatchObject({
sourceSample: {
rows: wardValues.map((value) => ({ fields: [{ name: ward.name, value }] })),
representativeValues: [{ column: ward.name, values: wardValues }],
},
});
});
test("continues metadata-only with one safe warning when source sampling is unavailable", async () => {
const repository = new MemoryCatalogRepository();
const database = await repository.create({