diff --git a/harness/tests/test_session_documents.py b/harness/tests/test_session_documents.py index eba9af9a..b69baeec 100644 --- a/harness/tests/test_session_documents.py +++ b/harness/tests/test_session_documents.py @@ -203,6 +203,57 @@ def test_memory_projection_does_not_repeat_detail_as_marker_rationale(tmp_path): } +def test_memory_projection_excludes_standalone_schema_tables_but_keeps_mappings(tmp_path): + manifest = create_session("q", _db(), tmp_path) + session_dir = tmp_path / manifest.id + decisions = [ + _decision(1, "table_promoted", "fact_procedure", "Tabella dei fatti"), + _decision(2, "memory_promoted", "fact_procedure", "seq:1"), + _decision(3, "table_excluded", "dim_time", "Dimensione temporale"), + _decision(4, "memory_promotion_declined", "dim_time", "Dimensione temporale"), + _decision( + 5, + "concept_clarified", + "ablazione valida", + "Mappata su `fact_procedure.ablazione_transcatetere = TRUE`.", + ), + _decision(6, "memory_promoted", "ablazione valida", "seq:5"), + _decision( + 7, + "memory_promotion_declined", + "conteggio annuale", + "`COUNT(DISTINCT patient_id)` raggruppato per `dim_time.year`.", + ), + ] + (session_dir / "schema_linking.json").write_text(json.dumps({ + "question": "q", + "candidates": [{"kind": "table", "name": "fact_procedure"}], + "joins": [], + "excluded": [{"kind": "table", "name": "dim_time"}], + "open_questions": [], + })) + (session_dir / "review_decisions.jsonl").write_text( + "\n".join(d.model_dump_json() for d in decisions) + "\n" + ) + + by_key = {doc["key"]: doc for doc in build_documents(manifest, session_dir)} + + assert json.loads(by_key["memories"]["content"]) == [ + { + "status": "approved", + "subject": "ablazione valida", + "detail": "Mappata su `fact_procedure.ablazione_transcatetere = TRUE`.", + "rationale": "", + }, + { + "status": "declined", + "subject": "conteggio annuale", + "detail": "`COUNT(DISTINCT patient_id)` raggruppato per `dim_time.year`.", + "rationale": "", + }, + ] + + def test_filesystem_and_snapshot_builders_share_the_same_projection(tmp_path): manifest, session_dir, question, report, decisions = _rich_session(tmp_path) snapshot = SessionSnapshot( diff --git a/harness/tht/session/store.py b/harness/tht/session/store.py index bca094f7..c535d0e0 100644 --- a/harness/tht/session/store.py +++ b/harness/tht/session/store.py @@ -154,6 +154,7 @@ _NEXT_H2 = re.compile(r"^##\s+", re.MULTILINE) _MEMORY_TYPES = frozenset( {"concept_clarified", "memory_promoted", "memory_promotion_declined", "memory_rejected"} ) +_TABLE_DECISION_TYPES = frozenset({"table_approved", "table_promoted", "table_excluded"}) _SUMMARY_HIDDEN_TYPES = _MEMORY_TYPES | frozenset( { "phase_approved", @@ -191,7 +192,49 @@ def _referenced_decision(marker, by_seq: dict[int, object]): return by_seq.get(int(match.group(1))) if match else None -def _memory_projection(decisions: list) -> list[dict[str, str]]: +def _normalized_identifier(value: str) -> str: + return value.strip().strip("`\"'").lower() + + +def _schema_linking_tables(content: str | None, decisions: list) -> set[str]: + names = { + _normalized_identifier(decision.subject) + for decision in decisions + if decision.type in _TABLE_DECISION_TYPES and decision.subject.strip() + } + if not content: + return names + try: + schema_linking = json.loads(content) + except (TypeError, json.JSONDecodeError): + return names + if not isinstance(schema_linking, dict): + return names + for key in ("candidates", "excluded"): + values = schema_linking.get(key, []) + if not isinstance(values, list): + continue + for value in values: + if not isinstance(value, dict) or value.get("kind") != "table": + continue + name = value.get("name") + if isinstance(name, str) and name.strip(): + names.add(_normalized_identifier(name)) + return names + + +def _is_standalone_table_subject(subject: str, schema_tables: set[str]) -> bool: + normalized = _normalized_identifier(subject) + leaf = normalized.rsplit(".", 1)[-1] + schema_leaves = {name.rsplit(".", 1)[-1] for name in schema_tables} + return ( + normalized in schema_tables + or leaf in schema_leaves + or re.fullmatch(r"(?:fact|dim)(?:_[a-z0-9_]+)?", leaf) is not None + ) + + +def _memory_projection(decisions: list, schema_tables: set[str]) -> list[dict[str, str]]: by_seq = {decision.seq: decision for decision in decisions} approved = [decision for decision in decisions if decision.type == "memory_promoted"] declined = [ @@ -203,6 +246,11 @@ def _memory_projection(decisions: list) -> list[dict[str, str]]: for status, markers in (("approved", approved), ("declined", declined)): for marker in markers: origin = _referenced_decision(marker, by_seq) + if origin is not None and origin.type != "concept_clarified": + continue + subject = origin.subject if origin is not None else marker.subject + if _is_standalone_table_subject(subject, schema_tables): + continue detail = origin.detail if origin is not None else marker.detail if re.fullmatch(r"seq:\d+", detail.strip()): detail = "" @@ -215,7 +263,7 @@ def _memory_projection(decisions: list) -> list[dict[str, str]]: items.append( { "status": status, - "subject": (origin.subject if origin is not None else marker.subject), + "subject": subject, "detail": detail, "rationale": rationale, } @@ -259,14 +307,17 @@ def _document_bundle(manifest: SessionManifest, artifacts: dict[str, str], decis "format": "markdown", "content": assumptions, }) - memories = _memory_projection(decisions) + schema_linking = artifacts.get("schema_linking") + memories = _memory_projection( + decisions, + _schema_linking_tables(schema_linking, decisions), + ) if memories: docs.append({ "phase": "F8", "key": "memories", "title": "Memories", "format": "memories", "content": json.dumps(memories, ensure_ascii=False), }) - schema_linking = artifacts.get("schema_linking") if schema_linking is not None: docs.append({ "phase": "F4", "key": "schema_linking", "title": "Schema linking",