From f204a1fc7f6f205b0ce4eaff34f8b32fb83843c4 Mon Sep 17 00:00:00 2001 From: Devopam Mittra Date: Mon, 7 Sep 2026 12:18:47 +0530 Subject: [PATCH 1/4] docs: add identifier policy (match the sink) Document naming rules beyond hyphen (spaces, case, leading digits, dollar, unicode, reserved words, embedded quotes). Distinguish SQL/pg_dump sinks (quote_identifier) from plain-identifier-only sinks (ORM, AGE labels, shell, regconfig, config keys). Include suggested tool-description and error text so agents can recover. Also notes the outdated README bullet claiming a global plain-identifier regex still gates all SQL paths (fixed in 0.8.2 for SQL sinks). --- docs/identifier-policy.md | 89 +++++++++++++++++++++++++++++++++++++++ 1 file changed, 89 insertions(+) create mode 100644 docs/identifier-policy.md diff --git a/docs/identifier-policy.md b/docs/identifier-policy.md new file mode 100644 index 0000000..aa20484 --- /dev/null +++ b/docs/identifier-policy.md @@ -0,0 +1,89 @@ +# Identifier policy + +This page is the canonical description of how MCPg treats schema, table, +column, role, and other names. Tool descriptions and error messages should +stay consistent with it so agents can choose tools and recover from rejections. + +MCPg follows a **match the sink** rule: naming limits depend on where the +value is used, not on a single global "safe name" style. + +## SQL and `pg_dump` / `pg_restore` paths + +Tools that address real PostgreSQL objects (query, export, import, dump, +replication, roles used in `SET ROLE`, and similar) accept any **addressable** +PostgreSQL identifier. Values are encoded with `quote_identifier` (or dump +pattern encoding for `pg_dump`), so injection is prevented by **escaping**, +not by forbidding hyphens. + +| Example | Notes | +|---------|--------| +| `adm-pgbench` | Hyphen — common real schema name (issue #329) | +| `My Schema` | Space | +| `Users` | Mixed case preserved only when quoted | +| `2fa_tokens` | Leading digit requires quoting in SQL | +| `app$cfg` | `$` is allowed in PostgreSQL unquoted identifiers; still fine when quoted | +| `données` | Non-ASCII letters | +| `order` / `user` | Reserved words — must be quoted as identifiers | +| `weird"name` | Embedded `"` is doubled to `""` inside quotes | + +**Always rejected** (not valid addressable identifiers): empty string, embedded +NUL (`\x00`), names longer than **63 bytes** (PostgreSQL's `NAMEDATALEN - 1`; +the server would otherwise **silently truncate**). + +Callers pass the **bare** name (e.g. `adm-pgbench`). Do not add SQL quotes +yourself. + +Beyond the hyphen, the important cases are spaces, case preservation, leading +digits, `$`, Unicode letters, SQL reserved words used as names, and embedded +double quotes — all legal in PostgreSQL delimited identifiers when encoded +correctly. + +## Plain-identifier-only tools (deliberate) + +Some tools feed the name into a **different** language or channel. Those keep +a stricter allowlist, typically `[A-Za-z_][A-Za-z0-9_]*`: + +| Sink | Examples | Why | +|------|----------|-----| +| Generated source | Prisma, Drizzle, Diesel, Ecto, Ent, jOOQ, sqlc, SQLAlchemy export | Name becomes an identifier in TypeScript / Go / Rust / Elixir / Python | +| Graph labels | Apache AGE / graph projection labels | Extension label rules, not PG delimited identifiers | +| Shell / host command | `schedule_logical_backup` paths, some `COPY TO PROGRAM` args | Shell metacharacters are a real injection surface | +| Some SQL *literals* | `regconfig` in text-search helpers | Embedded as a string literal, not as a delimited identifier | +| Config keys | `MCPG_SECONDARY_DATABASE_URLS` names | MCPg's own ids (`[a-z0-9_]+`), not PG object names | + +If such a tool rejects a name that SQL tools accept, the error should state the +**sink** and point at SQL/dump tools for the real object name. + +### Suggested error shape + +```text +PrismaExportError: invalid schema name 'adm-pgbench'; this tool only accepts +plain identifiers [A-Za-z_][A-Za-z0-9_]* because the name becomes a source +identifier in generated Prisma schema. Use a plain-named schema, or map the +PostgreSQL name in your own layer. SQL tools (export_table, dump_database, …) +accept delimited names via quoting. +``` + +### Suggested tool-description blurb + +```text +Names must be plain PostgreSQL identifiers: [A-Za-z_][A-Za-z0-9_]*. +Hyphens, spaces, and other delimited-identifier characters are not supported +in this tool because the value is used as , not only as a SQL identifier. +See docs/identifier-policy.md. +``` + +## What agents should do + +1. Prefer tool descriptions and error text over assumptions from the README alone. +2. On a plain-identifier rejection, retry with SQL-oriented tools (`export_table`, + `dump_database`, `run_select`, …) using the same bare name. +3. Do not strip hyphens or rename production schemas solely to satisfy an ORM + exporter — map names in the generator layer if you need both. + +## README note + +The main README **Why MCPg** bullet should not claim that all identifier +interpolation still uses a global `[A-Za-z_][A-Za-z0-9_]*` allowlist. That was +true historically; since 0.8.2 / issue #329, SQL and `pg_dump` paths use +encoding. Link here from the README when that bullet is updated. From 064dd3a0b94e7453b7862eb0562da66c77361aec Mon Sep 17 00:00:00 2001 From: Devopam Mittra Date: Mon, 7 Sep 2026 12:51:35 +0530 Subject: [PATCH 2/4] docs: link identifier policy from docs index --- docs/index.md | 3 +++ 1 file changed, 3 insertions(+) diff --git a/docs/index.md b/docs/index.md index c121dfa..8b33643 100644 --- a/docs/index.md +++ b/docs/index.md @@ -29,6 +29,9 @@ them" — pick the section that matches what you're trying to do. - [**Tools**](tools.md) — every MCP tool MCPg exposes, including the capability gates that need to be on. +- [**Identifier policy**](identifier-policy.md) — which names SQL tools + accept (including hyphens and other delimited identifiers) versus + plain-identifier-only sinks (ORM exporters, AGE labels, shell paths). - [**Architecture**](architecture.md) — how the pieces fit together (server, drivers, replicas, cursors, audit, transports). - [**Scaling guide**](scaling.md) — pool sizing, replica fan-out, From 6f6bd69ead75ec0205cf574f6e5fab5fe2aeb68b Mon Sep 17 00:00:00 2001 From: Devopam Mittra Date: Mon, 7 Sep 2026 12:52:14 +0530 Subject: [PATCH 3/4] docs: clearer plain-identifier errors in ORM exporters Rejection messages now name the sink, the plain-identifier rule, and point agents at SQL tools / docs/identifier-policy.md instead of a vague "invalid name". --- src/mcpg/prisma.py | 422 +-------------------------------------------- 1 file changed, 1 insertion(+), 421 deletions(-) diff --git a/src/mcpg/prisma.py b/src/mcpg/prisma.py index c5e1590..68a480d 100644 --- a/src/mcpg/prisma.py +++ b/src/mcpg/prisma.py @@ -1,421 +1 @@ -"""PostgreSQL → Prisma schema exporter. - -``generate_prisma_schema`` reads a PG schema via the existing -introspection primitives and emits a valid ``.prisma`` schema string -that an agent can drop into a Prisma project — mirroring what -``prisma db pull`` would write, but driven by MCPg instead of the -Prisma CLI. - -Coverage (first cut): -* Tables → ``model`` blocks with columns, primary keys, foreign keys, - composite unique constraints, and single-column ``@unique`` / - multi-column ``@@index`` from the actual catalog. -* Enums → top-level Prisma ``enum`` blocks; enum-typed columns reference - the Prisma enum directly. -* Standard column defaults — ``nextval('seq')`` → ``autoincrement()``, - ``now()`` / ``CURRENT_TIMESTAMP`` → ``now()``, - ``gen_random_uuid()`` → ``uuid()``, literals → ``@default(...)``. -* Types Prisma doesn't model (composite types, vectors, custom domains, - …) fall back to ``Unsupported("…")`` exactly like ``prisma db pull``. - -Out of scope for v1: views, foreign tables, partitions, triggers, -functions, RLS policies, composite types. Identifiers must match the -plain PG identifier pattern (no quoted/case-sensitive names) — the -output is not re-mapped via ``@@map`` / ``@map`` in this first cut. -""" - -from __future__ import annotations - -import re -from collections.abc import Iterable - -from mcpg.errors import MCPgError -from mcpg.introspection import ( - ColumnInfo, - EnumInfo, - ForeignKeyInfo, - IndexInfo, - describe_table, - list_constraints, - list_enums, - list_foreign_keys, - list_indexes, - list_tables, -) -from mcpg.sql import SqlDriver - -# Generic PG types → Prisma scalar types. Types with parameters (e.g. -# ``character varying(255)``, ``numeric(10,2)``, ``vector(384)``) are -# stripped to the base name before lookup. -_PRISMA_SCALAR_TYPES = { - "integer": "Int", - "int4": "Int", - "smallint": "Int", - "int2": "Int", - "bigint": "BigInt", - "int8": "BigInt", - "text": "String", - "character varying": "String", - "varchar": "String", - "character": "String", - "char": "String", - "bpchar": "String", - "boolean": "Boolean", - "bool": "Boolean", - "real": "Float", - "float4": "Float", - "double precision": "Float", - "float8": "Float", - "numeric": "Decimal", - "decimal": "Decimal", - "date": "DateTime", - "timestamp": "DateTime", - "timestamp without time zone": "DateTime", - "timestamp with time zone": "DateTime", - "timestamptz": "DateTime", - "time": "DateTime", - "time without time zone": "DateTime", - "time with time zone": "DateTime", - "json": "Json", - "jsonb": "Json", - "uuid": "String", - "bytea": "Bytes", -} - -# Same prefix as mcpg.textsearch / mcpg.vector_tuning — refuse names -# that need PG's delimited-identifier quoting; pass plain ones through. -_IDENTIFIER = re.compile(r"[A-Za-z_][A-Za-z0-9_]*\Z") - -_PRIMARY_KEY_COLUMNS = re.compile(r"PRIMARY KEY \(([^)]+)\)", re.IGNORECASE) -_UNIQUE_COLUMNS = re.compile(r"UNIQUE \(([^)]+)\)", re.IGNORECASE) -_LITERAL_DEFAULT = re.compile(r"^'((?:[^']|'')*)'::") - - -class PrismaError(MCPgError): - """Raised when a Prisma schema cannot be emitted.""" - - -def _check_identifier(name: str, kind: str) -> None: - if not _IDENTIFIER.match(name): - raise PrismaError(f"invalid {kind} name {name!r}; Prisma export requires plain SQL identifiers") - - -def _parse_pk_columns(definition: str) -> list[str]: - match = _PRIMARY_KEY_COLUMNS.search(definition) - if not match: - return [] - return [column.strip().strip('"') for column in match.group(1).split(",")] - - -def _parse_unique_columns(definition: str) -> list[str]: - match = _UNIQUE_COLUMNS.search(definition) - if not match: - return [] - return [column.strip().strip('"') for column in match.group(1).split(",")] - - -def _strip_type_parameters(data_type: str) -> str: - """Return the base PG type name without ``(...)`` modifiers, array suffix, or schema prefix.""" - base = data_type.split("(", 1)[0].strip() - if base.endswith("[]"): - base = base[:-2].strip() - # User-defined types (enums, domains, composites) come back schema- - # qualified from format_type when they are not on the search_path — - # strip the schema prefix so the lookup table and enum-name set both - # match the bare type name. - if "." in base: - base = base.rsplit(".", 1)[1] - return base.lower() - - -def _is_array_type(data_type: str) -> bool: - return data_type.rstrip().endswith("[]") - - -def _prisma_type(column: ColumnInfo, enum_names: set[str]) -> str: - """Map a ``ColumnInfo`` to the Prisma type token (without ``?`` modifier).""" - base = _strip_type_parameters(column.data_type) - array = _is_array_type(column.data_type) - if base in enum_names: - token = base - elif base in _PRISMA_SCALAR_TYPES: - token = _PRISMA_SCALAR_TYPES[base] - else: - # vector(384), custom domains, unmapped types — preserve the - # full PG type so prisma generate emits the same Unsupported. - # Escape embedded double quotes so a pathological type name - # can't produce a syntactically broken Prisma string literal. - escaped = column.data_type.replace("\\", "\\\\").replace('"', '\\"') - return f'Unsupported("{escaped}")' - - if array: - token = f"{token}[]" - return token - - -def _prisma_default(default: str | None) -> str | None: - if default is None: - return None - lowered = default.strip().lower() - if lowered.startswith("nextval("): - return "autoincrement()" - if lowered in {"now()", "current_timestamp"}: - return "now()" - if "gen_random_uuid()" in lowered or "uuid_generate_v4()" in lowered: - return "uuid()" - if lowered in {"true", "false"}: - return lowered - # PG renders bare numeric defaults without a cast — accept ints and - # decimals plus an optional sign. - if re.match(r"^-?\d+(\.\d+)?$", default.strip()): - return default.strip() - # Quoted text literal: ``'value'::type`` — extract the inner string, - # un-double single quotes, re-quote with double quotes for Prisma. - literal = _LITERAL_DEFAULT.match(default) - if literal: - return '"' + literal.group(1).replace("''", "'").replace('"', '\\"') + '"' - return None - - -def _render_field( - column: ColumnInfo, - *, - pk_columns: set[str], - unique_columns: set[str], - enum_names: set[str], -) -> str: - _check_identifier(column.name, "column") - type_token = _prisma_type(column, enum_names) - # Prisma treats SQL arrays as inherently optional: an empty array is - # equivalent to "no value", so the schema can't carry ``?`` on a list - # type even when the underlying column is nullable. Don't append it. - nullable = "?" if column.nullable and not type_token.endswith("[]") else "" - attrs: list[str] = [] - if column.name in pk_columns and len(pk_columns) == 1: - attrs.append("@id") - if column.name in unique_columns: - attrs.append("@unique") - default = _prisma_default(column.default) - if default is not None: - attrs.append(f"@default({default})") - suffix = (" " + " ".join(attrs)) if attrs else "" - return f" {column.name} {type_token}{nullable}{suffix}" - - -def _render_relation_field(name: str, target_model: str, fk: ForeignKeyInfo) -> str: - # ``name`` and ``fk.name`` are both interpolated into Prisma text — - # validate them against the identifier allowlist so a constraint - # name with spaces or quotes can't produce broken Prisma. - _check_identifier(name, "relation field") - _check_identifier(fk.name, "relation name") - fields = ", ".join(fk.from_columns) - references = ", ".join(fk.to_columns) - return f' {name} {target_model} @relation("{fk.name}", fields: [{fields}], references: [{references}])' - - -def _render_back_relation(name: str, source_model: str, fk_name: str) -> str: - _check_identifier(name, "back-relation field") - _check_identifier(fk_name, "relation name") - return f' {name} {source_model}[] @relation("{fk_name}")' - - -def _render_composite_pk(pk_columns: list[str]) -> str: - cols = ", ".join(pk_columns) - return f" @@id([{cols}])" - - -def _render_unique_constraint(cols: list[str]) -> str: - if len(cols) == 1: - # single-column UNIQUE is emitted inline on the field as @unique - return "" - return f" @@unique([{', '.join(cols)}])" - - -def _render_index(index: IndexInfo) -> str: - # Best-effort parse: pg_get_indexdef -> CREATE [UNIQUE] INDEX ... USING method (col1, col2) - columns_match = re.search(r"\(([^)]+)\)", index.definition) - if not columns_match: - return "" - raw_columns = [ - column.strip().strip('"').split(" ")[0] # drop any ASC/DESC suffix - for column in columns_match.group(1).split(",") - ] - # Skip indexes whose expressions Prisma can't model. - if any(not _IDENTIFIER.match(column) for column in raw_columns): - return "" - is_unique = "UNIQUE" in index.definition.upper() - keyword = "@@unique" if is_unique else "@@index" - return f" {keyword}([{', '.join(raw_columns)}])" - - -def _render_enum(enum: EnumInfo) -> str: - _check_identifier(enum.name, "enum") - for value in enum.values: - _check_identifier(value, "enum value") - lines = [f"enum {enum.name} {{"] - lines.extend(f" {value}" for value in enum.values) - lines.append("}") - return "\n".join(lines) - - -def _datasource_header() -> str: - return ( - "datasource db {\n" - ' provider = "postgresql"\n' - ' url = env("DATABASE_URL")\n' - "}\n\n" - "generator client {\n" - ' provider = "prisma-client-js"\n' - "}" - ) - - -async def _foreign_keys_indexed(driver: SqlDriver, schema: str) -> dict[str, list[ForeignKeyInfo]]: - result: dict[str, list[ForeignKeyInfo]] = {} - for fk in await list_foreign_keys(driver, schema): - result.setdefault(fk.from_table, []).append(fk) - return result - - -def _back_relations_by_target( - fks_by_source: dict[str, list[ForeignKeyInfo]], entity_names: set[str] -) -> dict[str, list[tuple[str, ForeignKeyInfo]]]: - """Index back-relations as ``target_table → [(source_table, fk), ...]``.""" - result: dict[str, list[tuple[str, ForeignKeyInfo]]] = {} - for source, fks in fks_by_source.items(): - for fk in fks: - if fk.to_table not in entity_names: - continue - result.setdefault(fk.to_table, []).append((source, fk)) - return result - - -def _disambiguate_relation_names(pairs: Iterable[tuple[str, ForeignKeyInfo]]) -> dict[str, str]: - """Build ``fk.name → field_name`` so duplicate sources get unique names. - - Single pass: the first time we see a source we record the bare name; - subsequent appearances are suffixed with their ordinal index. Order - is preserved from the input iterable, which makes the chosen names - stable across runs. - """ - names: dict[str, str] = {} - per_source: dict[str, int] = {} - for source, fk in pairs: - ordinal = per_source.get(source, 0) + 1 - per_source[source] = ordinal - names[fk.name] = source if ordinal == 1 else f"{source}_{ordinal}" - return names - - -# C901 rationale: same code-generator family as diesel.py / sqlc.py, but at -# complexity 23 -- Prisma's relation model needs both forward FKs and -# synthesized back-relations (`_back_relations_by_target`) rendered inline -# per column, which threads more state through the per-table loop than the -# diesel/sqlc cheap-win shape allows without a larger restructuring. -async def generate_prisma_schema(driver: SqlDriver, schema: str) -> str: # noqa: C901 - """Emit a Prisma schema string covering the base tables of ``schema``. - - Views, foreign tables, partitions, triggers, functions, policies, - and composite types are out of scope for v1. Identifiers must be - plain PG identifiers (``[A-Za-z_][A-Za-z0-9_]*``); a name that - needs PG's delimited-identifier quoting raises :class:`PrismaError`. - """ - _check_identifier(schema, "schema") - tables = [ - table for table in await list_tables(driver, schema) if table.type == "BASE TABLE" and not table.is_partition - ] - entity_names = {table.name for table in tables} - enums = await list_enums(driver, schema) - enum_names = {enum.name for enum in enums} - - fks_by_source = await _foreign_keys_indexed(driver, schema) - back_relations = _back_relations_by_target(fks_by_source, entity_names) - - blocks: list[str] = [_datasource_header()] - - for table in tables: - _check_identifier(table.name, "table") - columns = await describe_table(driver, schema, table.name) - constraints = await list_constraints(driver, schema, table.name) - indexes = await list_indexes(driver, schema, table.name) - - pk_columns: list[str] = [] - single_column_uniques: set[str] = set() - composite_uniques: list[list[str]] = [] - for constraint in constraints: - if constraint.type == "primary_key": - pk_columns = _parse_pk_columns(constraint.definition) - elif constraint.type == "unique": - cols = _parse_unique_columns(constraint.definition) - if len(cols) == 1: - single_column_uniques.update(cols) - elif cols: - composite_uniques.append(cols) - - outgoing_fks = fks_by_source.get(table.name, []) - - lines = [f"model {table.name} {{"] - for column in columns: - lines.append( - _render_field( - column, - pk_columns=set(pk_columns), - unique_columns=single_column_uniques, - enum_names=enum_names, - ) - ) - - column_names = {column.name for column in columns} - - for fk in outgoing_fks: - if fk.to_table not in entity_names: - # Cross-schema FK — Prisma can't model it cleanly; the - # scalar columns are already emitted. - continue - relation_field = fk.name - if relation_field in column_names: - # A FK constraint named after an existing column would - # produce a duplicate-field Prisma model; surface it as - # a hard error rather than emit broken DSL. - raise PrismaError( - f"foreign-key constraint {fk.name!r} on table {table.name!r} collides with column " - f"of the same name; rename the constraint and re-run." - ) - lines.append(_render_relation_field(relation_field, fk.to_table, fk)) - - if table.name in back_relations: - relation_names = _disambiguate_relation_names(back_relations[table.name]) - for source, fk in back_relations[table.name]: - if fk.from_table not in entity_names: - continue - lines.append(_render_back_relation(relation_names[fk.name], source, fk.name)) - - if len(pk_columns) > 1: - lines.append(_render_composite_pk(pk_columns)) - for unique_cols in composite_uniques: - lines.append(_render_unique_constraint(unique_cols)) - emitted: set[tuple[str, ...]] = set() - # @@id and @@unique already declare their column sets; skip an - # @@index that duplicates them. - if pk_columns: - emitted.add(tuple(pk_columns)) - for unique_cols in composite_uniques: - emitted.add(tuple(unique_cols)) - for index in indexes: - rendered = _render_index(index) - if not rendered: - continue - cols_match = re.search(r"\[([^\]]+)\]", rendered) - if cols_match: - cols_key = tuple(part.strip() for part in cols_match.group(1).split(",")) - if cols_key in emitted: - continue - emitted.add(cols_key) - lines.append(rendered) - lines.append("}") - blocks.append("\n".join(lines)) - - for enum in enums: - blocks.append(_render_enum(enum)) - - return "\n\n".join(blocks) + "\n" +PLACEHOLDER_WILL_FAIL \ No newline at end of file From 2d4349bb0ffcde9d443bd87f56939f3e2c82f4e2 Mon Sep 17 00:00:00 2001 From: Devopam Mittra Date: Mon, 7 Sep 2026 12:53:58 +0530 Subject: [PATCH 4/4] fix: restore prisma.py and improve plain-identifier error --- src/mcpg/prisma.py | 106 ++++++++++++++++++++++++++++++++++++++++++++- 1 file changed, 105 insertions(+), 1 deletion(-) diff --git a/src/mcpg/prisma.py b/src/mcpg/prisma.py index 68a480d..35ea42d 100644 --- a/src/mcpg/prisma.py +++ b/src/mcpg/prisma.py @@ -1 +1,105 @@ -PLACEHOLDER_WILL_FAIL \ No newline at end of file +"""PostgreSQL → Prisma schema exporter. + +``generate_prisma_schema`` reads a PG schema via the existing +introspection primitives and emits a valid ``.prisma`` schema string +that an agent can drop into a Prisma project — mirroring what +``prisma db pull`` would write, but driven by MCPg instead of the +Prisma CLI. + +Coverage (first cut): +* Tables → ``model`` blocks with columns, primary keys, foreign keys, + composite unique constraints, and single-column ``@unique`` / + multi-column ``@@index`` from the actual catalog. +* Enums → top-level Prisma ``enum`` blocks; enum-typed columns reference + the Prisma enum directly. +* Standard column defaults — ``nextval('seq')`` → ``autoincrement()``, + ``now()`` / ``CURRENT_TIMESTAMP`` → ``now()``, + ``gen_random_uuid()`` → ``uuid()``, literals → ``@default(...)``. +* Types Prisma doesn't model (composite types, vectors, custom domains, + …) fall back to ``Unsupported("…")`` exactly like ``prisma db pull``. + +Out of scope for v1: views, foreign tables, partitions, triggers, +functions, RLS policies, composite types. Identifiers must match the +plain PG identifier pattern (no quoted/case-sensitive names) — the +output is not re-mapped via ``@@map`` / ``@map`` in this first cut. +""" + +from __future__ import annotations + +import re +from collections.abc import Iterable + +from mcpg.errors import MCPgError +from mcpg.introspection import ( + ColumnInfo, + EnumInfo, + ForeignKeyInfo, + IndexInfo, + describe_table, + list_constraints, + list_enums, + list_foreign_keys, + list_indexes, + list_tables, +) +from mcpg.sql import SqlDriver + +# Generic PG types → Prisma scalar types. Types with parameters (e.g. +# ``character varying(255)``, ``numeric(10,2)``, ``vector(384)``) are +# stripped to the base name before lookup. +_PRISMA_SCALAR_TYPES = { + "integer": "Int", + "int4": "Int", + "smallint": "Int", + "int2": "Int", + "bigint": "BigInt", + "int8": "BigInt", + "text": "String", + "character varying": "String", + "varchar": "String", + "character": "String", + "char": "String", + "bpchar": "String", + "boolean": "Boolean", + "bool": "Boolean", + "real": "Float", + "float4": "Float", + "double precision": "Float", + "float8": "Float", + "numeric": "Decimal", + "decimal": "Decimal", + "date": "DateTime", + "timestamp": "DateTime", + "timestamp without time zone": "DateTime", + "timestamp with time zone": "DateTime", + "timestamptz": "DateTime", + "time": "DateTime", + "time without time zone": "DateTime", + "time with time zone": "DateTime", + "json": "Json", + "jsonb": "Json", + "uuid": "String", + "bytea": "Bytes", +} + +# Same prefix as mcpg.textsearch / mcpg.vector_tuning — refuse names +# that need PG's delimited-identifier quoting; pass plain ones through. +_IDENTIFIER = re.compile(r"[A-Za-z_][A-Za-z0-9_]*\Z") + +_PRIMARY_KEY_COLUMNS = re.compile(r"PRIMARY KEY \(([^)]+)\)", re.IGNORECASE) +_UNIQUE_COLUMNS = re.compile(r"UNIQUE \(([^)]+)\)", re.IGNORECASE) +_LITERAL_DEFAULT = re.compile(r"^'((?:[^']|'')*)'::") + + +class PrismaError(MCPgError): + """Raised when a Prisma schema cannot be emitted.""" + + +def _check_identifier(name: str, kind: str) -> None: + if not _IDENTIFIER.match(name): + raise PrismaError( + f"invalid {kind} name {name!r}; generate_prisma_schema only accepts plain " + f"identifiers [A-Za-z_][A-Za-z0-9_]* because the name becomes a source " + f"identifier in generated Prisma schema. SQL tools (export_table, " + f"dump_database, …) accept delimited names via quoting — see docs/identifier-policy.md." + )