fix: exclude schema tables from session memories

This commit is contained in:
2026-07-23 15:54:27 +02:00
parent 9b5fca59e8
commit 2e1bb621a0
2 changed files with 106 additions and 4 deletions
+51
View File
@@ -203,6 +203,57 @@ def test_memory_projection_does_not_repeat_detail_as_marker_rationale(tmp_path):
}
def test_memory_projection_excludes_standalone_schema_tables_but_keeps_mappings(tmp_path):
manifest = create_session("q", _db(), tmp_path)
session_dir = tmp_path / manifest.id
decisions = [
_decision(1, "table_promoted", "fact_procedure", "Tabella dei fatti"),
_decision(2, "memory_promoted", "fact_procedure", "seq:1"),
_decision(3, "table_excluded", "dim_time", "Dimensione temporale"),
_decision(4, "memory_promotion_declined", "dim_time", "Dimensione temporale"),
_decision(
5,
"concept_clarified",
"ablazione valida",
"Mappata su `fact_procedure.ablazione_transcatetere = TRUE`.",
),
_decision(6, "memory_promoted", "ablazione valida", "seq:5"),
_decision(
7,
"memory_promotion_declined",
"conteggio annuale",
"`COUNT(DISTINCT patient_id)` raggruppato per `dim_time.year`.",
),
]
(session_dir / "schema_linking.json").write_text(json.dumps({
"question": "q",
"candidates": [{"kind": "table", "name": "fact_procedure"}],
"joins": [],
"excluded": [{"kind": "table", "name": "dim_time"}],
"open_questions": [],
}))
(session_dir / "review_decisions.jsonl").write_text(
"\n".join(d.model_dump_json() for d in decisions) + "\n"
)
by_key = {doc["key"]: doc for doc in build_documents(manifest, session_dir)}
assert json.loads(by_key["memories"]["content"]) == [
{
"status": "approved",
"subject": "ablazione valida",
"detail": "Mappata su `fact_procedure.ablazione_transcatetere = TRUE`.",
"rationale": "",
},
{
"status": "declined",
"subject": "conteggio annuale",
"detail": "`COUNT(DISTINCT patient_id)` raggruppato per `dim_time.year`.",
"rationale": "",
},
]
def test_filesystem_and_snapshot_builders_share_the_same_projection(tmp_path):
manifest, session_dir, question, report, decisions = _rich_session(tmp_path)
snapshot = SessionSnapshot(
+55 -4
View File
@@ -154,6 +154,7 @@ _NEXT_H2 = re.compile(r"^##\s+", re.MULTILINE)
_MEMORY_TYPES = frozenset(
{"concept_clarified", "memory_promoted", "memory_promotion_declined", "memory_rejected"}
)
_TABLE_DECISION_TYPES = frozenset({"table_approved", "table_promoted", "table_excluded"})
_SUMMARY_HIDDEN_TYPES = _MEMORY_TYPES | frozenset(
{
"phase_approved",
@@ -191,7 +192,49 @@ def _referenced_decision(marker, by_seq: dict[int, object]):
return by_seq.get(int(match.group(1))) if match else None
def _memory_projection(decisions: list) -> list[dict[str, str]]:
def _normalized_identifier(value: str) -> str:
return value.strip().strip("`\"'").lower()
def _schema_linking_tables(content: str | None, decisions: list) -> set[str]:
names = {
_normalized_identifier(decision.subject)
for decision in decisions
if decision.type in _TABLE_DECISION_TYPES and decision.subject.strip()
}
if not content:
return names
try:
schema_linking = json.loads(content)
except (TypeError, json.JSONDecodeError):
return names
if not isinstance(schema_linking, dict):
return names
for key in ("candidates", "excluded"):
values = schema_linking.get(key, [])
if not isinstance(values, list):
continue
for value in values:
if not isinstance(value, dict) or value.get("kind") != "table":
continue
name = value.get("name")
if isinstance(name, str) and name.strip():
names.add(_normalized_identifier(name))
return names
def _is_standalone_table_subject(subject: str, schema_tables: set[str]) -> bool:
normalized = _normalized_identifier(subject)
leaf = normalized.rsplit(".", 1)[-1]
schema_leaves = {name.rsplit(".", 1)[-1] for name in schema_tables}
return (
normalized in schema_tables
or leaf in schema_leaves
or re.fullmatch(r"(?:fact|dim)(?:_[a-z0-9_]+)?", leaf) is not None
)
def _memory_projection(decisions: list, schema_tables: set[str]) -> list[dict[str, str]]:
by_seq = {decision.seq: decision for decision in decisions}
approved = [decision for decision in decisions if decision.type == "memory_promoted"]
declined = [
@@ -203,6 +246,11 @@ def _memory_projection(decisions: list) -> list[dict[str, str]]:
for status, markers in (("approved", approved), ("declined", declined)):
for marker in markers:
origin = _referenced_decision(marker, by_seq)
if origin is not None and origin.type != "concept_clarified":
continue
subject = origin.subject if origin is not None else marker.subject
if _is_standalone_table_subject(subject, schema_tables):
continue
detail = origin.detail if origin is not None else marker.detail
if re.fullmatch(r"seq:\d+", detail.strip()):
detail = ""
@@ -215,7 +263,7 @@ def _memory_projection(decisions: list) -> list[dict[str, str]]:
items.append(
{
"status": status,
"subject": (origin.subject if origin is not None else marker.subject),
"subject": subject,
"detail": detail,
"rationale": rationale,
}
@@ -259,14 +307,17 @@ def _document_bundle(manifest: SessionManifest, artifacts: dict[str, str], decis
"format": "markdown", "content": assumptions,
})
memories = _memory_projection(decisions)
schema_linking = artifacts.get("schema_linking")
memories = _memory_projection(
decisions,
_schema_linking_tables(schema_linking, decisions),
)
if memories:
docs.append({
"phase": "F8", "key": "memories", "title": "Memories",
"format": "memories", "content": json.dumps(memories, ensure_ascii=False),
})
schema_linking = artifacts.get("schema_linking")
if schema_linking is not None:
docs.append({
"phase": "F4", "key": "schema_linking", "title": "Schema linking",