From 85da324ad88c19e112a94a1aa484112c9bce4092 Mon Sep 17 00:00:00 2001 From: yohey-w Date: Sat, 22 Aug 2026 19:19:20 +0900 Subject: [PATCH 1/8] =?UTF-8?q?fix(parsing):=20=E5=88=97=E5=88=B6=E7=B4=84?= =?UTF-8?q?=E3=81=A7=E6=9B=B8=E3=81=8B=E3=82=8C=E3=81=9F=E5=A4=96=E9=83=A8?= =?UTF-8?q?=E3=82=AD=E3=83=BC=E3=82=92=E6=8A=BD=E5=87=BA=E3=81=99=E3=82=8B?= =?UTF-8?q?=E2=80=94=E2=80=94FK45=E6=9C=AC=E3=82=920=E6=9C=AC=E3=81=A8?= =?UTF-8?q?=E5=A0=B1=E5=91=8A=E3=81=97=E3=81=A6=E3=81=84=E3=81=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit SQLは外部キーを2通りで書ける: 表制約: FOREIGN KEY (a) REFERENCES parent (id) 列制約: a uuid not null references parent (id) ← FOREIGN KEY 語が無い 従来は表制約しか拾わず、列制約だけで書かれたスキーマを 「参照整合性が無い」と誤って報告していた(実例: FK45本を0本と要約)。 列制約は PostgreSQL / MySQL / SQLite いずれでも一般的な書き方。 あわせて2件: - SQLコメントを除去してから解析(列定義の直前にコメントが挟まると その列のFKだけが静かに落ちていた) - 区切りを幅ゼロの後読みに(default gen_random_uuid(), の閉じ括弧に 次の列が飲まれるのを防ぐ) --- codd/parsing/schemas.py | 77 +++++++++++++- tests/test_sql_column_level_foreign_keys.py | 107 ++++++++++++++++++++ 2 files changed, 179 insertions(+), 5 deletions(-) create mode 100644 tests/test_sql_column_level_foreign_keys.py diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index f817b0fc..77bf353d 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -113,13 +113,58 @@ def _sql_first_object_name(content_bytes: bytes, node: Any) -> str: return _normalize_ws(_node_text(content_bytes, child)) return "" +# SQL は外部キーを2通りで書ける。両方を拾わないと FK 数を過少に報告する。 +# 表制約: FOREIGN KEY (a) REFERENCES parent (id) +# 列制約: a uuid not null references parent (id) ← FOREIGN KEY 語が無い +# 列制約は PostgreSQL / MySQL / SQLite いずれでも一般的な書き方であり、 +# これを取りこぼすと「参照整合性が無い」という誤った所見が出る。 +_TABLE_LEVEL_FK = re.compile( + r"(?:CONSTRAINT\s+(?P\w+)\s+)?FOREIGN\s+KEY\s*\((?P[^)]+)\)" + r"\s+REFERENCES\s+(?P[^\s(]+)\s*\((?P[^)]+)\)", + re.IGNORECASE, +) + +# 列制約。定義の区切り(行頭・"(" ・ ",")直後の列名を捕らえ、同じ定義内の REFERENCES と対にする。 +# 区切りは "(" と "," の直後(幅ゼロの後読み)。行頭を含めると "create table x (" の +# "create" を列名として拾い、区切りを消費すると "default gen_random_uuid()," のような +# 直前の閉じ括弧に飲まれて次の列を取りこぼす。 +# 参照列の指定は省略できる(親の主キーに解決される)ので任意扱いにする。 +_COLUMN_LEVEL_FK = re.compile( + r"(?<=[(,])\s*(?:CONSTRAINT\s+(?P\w+)\s+)?" + r'(?P"[^"]+"|`[^`]+`|\[[^\]]+\]|\w+)' + r"(?![\w\s]*\bFOREIGN\s+KEY\b)" + r"[^,;\n]*?\bREFERENCES\s+(?P\"[^\"]+\"|`[^`]+`|\[[^\]]+\]|[\w.]+)" + r"\s*(?:\(\s*(?P[^)]+)\))?", + re.IGNORECASE | re.MULTILINE, +) + +_FK_COLUMN_RESERVED = { + "foreign", + "key", + "constraint", + "primary", + "unique", + "check", + "references", +} + +_SQL_LINE_COMMENT = re.compile(r"--[^\n]*") +_SQL_BLOCK_COMMENT = re.compile(r"/\*.*?\*/", re.DOTALL) + +def _strip_sql_comments(statement_text: str) -> str: + """コメントを取り除く。 + + 列定義の直前にコメント行が挟まると、区切り("," や "(")と列名の間に + 別の行が入り、列制約の外部キーを取りこぼす。行数は変えず中身だけ空にする。 + """ + without_block = _SQL_BLOCK_COMMENT.sub("", statement_text) + return _SQL_LINE_COMMENT.sub("", without_block) + def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, Any]]: matches: list[dict[str, Any]] = [] - pattern = re.compile( - r"(?:CONSTRAINT\s+(?P\w+)\s+)?FOREIGN\s+KEY\s*\((?P[^)]+)\)\s+REFERENCES\s+(?P[^\s(]+)\s*\((?P[^)]+)\)", - re.IGNORECASE, - ) - for match in pattern.finditer(statement_text): + statement_text = _strip_sql_comments(statement_text) + + for match in _TABLE_LEVEL_FK.finditer(statement_text): matches.append( { "name": match.group("name") or "", @@ -129,8 +174,30 @@ def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, "references_columns": _split_csv(match.group("ref_columns")), } ) + + for match in _COLUMN_LEVEL_FK.finditer(statement_text): + column = _strip_identifier_quotes(match.group("column")) + if not column or column.lower() in _FK_COLUMN_RESERVED: + continue + ref_columns_raw = match.group("ref_columns") + matches.append( + { + "name": match.group("name") or "", + "table": table_name, + "columns": [column], + "references_table": _strip_identifier_quotes(match.group("ref_table")), + "references_columns": _split_csv(ref_columns_raw) if ref_columns_raw else [], + } + ) + return matches +def _strip_identifier_quotes(identifier: str) -> str: + value = (identifier or "").strip() + if len(value) >= 2 and value[0] in '"`[' and value[-1] in '"`]': + return value[1:-1].strip() + return value + def _regex_create_index(statement_text: str) -> dict[str, Any] | None: match = re.search( r"CREATE\s+(?:UNIQUE\s+)?INDEX\s+(?P[^\s]+)\s+ON\s+(?P[^\s(]+)\s*\((?P[^)]+)\)", diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py new file mode 100644 index 00000000..aa75e83d --- /dev/null +++ b/tests/test_sql_column_level_foreign_keys.py @@ -0,0 +1,107 @@ +"""列制約で書かれた外部キーの抽出。 + +SQL は外部キーを2通りで書ける: + 表制約: FOREIGN KEY (a) REFERENCES parent (id) + 列制約: a uuid not null references parent (id) ← FOREIGN KEY 語が無い + +列制約は PostgreSQL / MySQL / SQLite いずれでも一般的な書き方であり、 +これを取りこぼすと「参照整合性が無い」という誤った所見が出る +(実例: 45本の FK を持つスキーマを「FK数0」と報告した)。 +""" + +import pytest + +from codd.parsing.schemas import _regex_foreign_keys + + +@pytest.mark.parametrize( + "label, statement, expected", + [ + ( + "列制約・参照列あり", + "create table child (id uuid primary key, parent_id uuid not null references parent (id));", + 1, + ), + ( + "列制約・参照列を省略(親の主キーに解決される)", + "create table child (parent_id uuid references parent);", + 1, + ), + ( + "表制約(従来から拾えていた形)", + "create table child (id uuid, foreign key (parent_id) references parent (id));", + 1, + ), + ( + "1行にまとめた定義でも拾う", + "create table t (id uuid primary key, a_id uuid references a (id), b_id uuid references b (id));", + 2, + ), + ( + "引用符つき識別子", + 'create table t ("parent_id" uuid references "parent" ("id"));', + 1, + ), + ( + "スキーマ修飾された親テーブル", + "create table t (tenant_id uuid not null references public.tenant (id));", + 1, + ), + ( + "外部キーが無ければ0", + "create table t (id uuid primary key, name text not null);", + 0, + ), + ], +) +def test_foreign_key_count(label: str, statement: str, expected: int) -> None: + assert len(_regex_foreign_keys(statement, "t")) == expected, label + + +def test_table_level_and_column_level_are_both_counted() -> None: + statement = """create table t ( + id uuid primary key, + a_id uuid not null references a (id), + b_id uuid references b (id) on delete cascade, + foreign key (c_id) references c (id) + );""" + assert len(_regex_foreign_keys(statement, "t")) == 3 + + +def test_column_level_extraction_records_the_reference() -> None: + result = _regex_foreign_keys( + "create table child (parent_id uuid not null references public.parent (id));", + "child", + ) + assert result == [ + { + "name": "", + "table": "child", + "columns": ["parent_id"], + "references_table": "public.parent", + "references_columns": ["id"], + } + ] + + +def test_comment_before_a_column_does_not_hide_its_foreign_key() -> None: + """列定義の直前にコメントが挟まっても取りこぼさない。 + + "," と列名の間に別行が入ると、コメントを除去しない実装では + その列の外部キーだけが静かに欠落する。 + """ + statement = """create table facility ( + id uuid primary key default gen_random_uuid(), + tenant_id uuid not null references tenant (id), + -- ER図の "||" 側をNOT NULL FKとして表現する + item_set_id uuid not null references item_set (id), + /* ブロックコメントでも同じ */ + report_definition_id uuid not null references report_definition (id) + );""" + columns = [fk["columns"][0] for fk in _regex_foreign_keys(statement, "facility")] + assert columns == ["tenant_id", "item_set_id", "report_definition_id"] + + +def test_references_inside_a_comment_is_not_counted() -> None: + statement = "create table t (id uuid primary key); -- references nothing" + assert _regex_foreign_keys(statement, "t") == [] From e3ffeb9e931db3cc32d7de3c8ba7129772a6b1e8 Mon Sep 17 00:00:00 2001 From: yohey-w Date: Mon, 14 Sep 2026 20:34:05 +0900 Subject: [PATCH 2/8] =?UTF-8?q?fix(parsing):=20=E6=96=87=E5=AD=97=E5=88=97?= =?UTF-8?q?=E3=81=AE=E4=B8=AD=E3=81=AE=20"--"=20=E3=82=92=E3=82=B3?= =?UTF-8?q?=E3=83=A1=E3=83=B3=E3=83=88=E3=81=A8=E8=AA=A4=E3=82=89=E3=81=AA?= =?UTF-8?q?=E3=81=84=E2=80=94=E2=80=94=E8=A1=A8=E5=88=B6=E7=B4=84=E3=81=AE?= =?UTF-8?q?FK=E3=81=BE=E3=81=A7=E6=B6=88=E3=81=97=E3=81=A6=E3=81=84?= =?UTF-8?q?=E3=81=9F?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 列制約のFK抽出(#42)で入れたコメント除去が、文字列リテラルを跨いでいた。 `default '--'` の "--" を行コメントの開始と誤り、そこから行末までを捨てるため、 同じ行にある `FOREIGN KEY (...) REFERENCES ...` まで消えていた。 これは本PR以前は拾えていた表制約の検出を壊す回帰だった。 create table t (sep text default '--', foreign key (p) references parent (id)); 旧: FK=1 → 本PRの初版: FK=0 → 本コミット: FK=1 コメント除去を、引用符(' " `)の中に入らない走査に置き換えた。 エスケープは SQL 標準の二重化('' や "")を終端と見ない。 あわせて列定義の切り出しを正規表現から深さ付きの走査に替えた。 `numeric(10,2)` の "," や `default 'a,b'` の "," は列の区切りではないのに、 区切りとみなしてカンマの右隣を列名として拾い、列名が "2" や "b" になっていた。 副次的に、複数行にまたがる列定義の外部キーも拾えるようになった。 回帰テストを先にRED(旧実装で5本失敗)にしてから修正した。 `ALTER TABLE ... ADD COLUMN` の列制約は未対応のまま(本PR以前も0本・回帰ではない)。 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_011xpq9A9SFvMUky6i6NLFz2 --- codd/parsing/schemas.py | 139 +++++++++++++++++--- tests/test_sql_column_level_foreign_keys.py | 65 +++++++++ 2 files changed, 184 insertions(+), 20 deletions(-) diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index 77bf353d..77507158 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -124,18 +124,21 @@ def _sql_first_object_name(content_bytes: bytes, node: Any) -> str: re.IGNORECASE, ) -# 列制約。定義の区切り(行頭・"(" ・ ",")直後の列名を捕らえ、同じ定義内の REFERENCES と対にする。 -# 区切りは "(" と "," の直後(幅ゼロの後読み)。行頭を含めると "create table x (" の -# "create" を列名として拾い、区切りを消費すると "default gen_random_uuid()," のような -# 直前の閉じ括弧に飲まれて次の列を取りこぼす。 +# 列制約。ひとつの列定義の中から「列名」と「参照先」を取り出す。 +# 定義の切り出しは正規表現ではなく _sql_table_definitions が行う—— +# numeric(10,2) の "," や default 'a,b' の "," は区切りではないので、 +# 正規表現でカンマを区切りとみなすと列名として "2" や "b" を拾ってしまう。 # 参照列の指定は省略できる(親の主キーに解決される)ので任意扱いにする。 _COLUMN_LEVEL_FK = re.compile( - r"(?<=[(,])\s*(?:CONSTRAINT\s+(?P\w+)\s+)?" - r'(?P"[^"]+"|`[^`]+`|\[[^\]]+\]|\w+)' - r"(?![\w\s]*\bFOREIGN\s+KEY\b)" - r"[^,;\n]*?\bREFERENCES\s+(?P\"[^\"]+\"|`[^`]+`|\[[^\]]+\]|[\w.]+)" + r"^\s*(?:CONSTRAINT\s+(?P\w+)\s+)?" + r'(?P"[^"]+"|`[^`]+`|\[[^\]]+\]|\w+)(?!\w)', + re.IGNORECASE, +) + +_REFERENCES_CLAUSE = re.compile( + r"\bREFERENCES\s+(?P\"[^\"]+\"|`[^`]+`|\[[^\]]+\]|[\w.]+)" r"\s*(?:\(\s*(?P[^)]+)\))?", - re.IGNORECASE | re.MULTILINE, + re.IGNORECASE, ) _FK_COLUMN_RESERVED = { @@ -146,19 +149,107 @@ def _sql_first_object_name(content_bytes: bytes, node: Any) -> str: "unique", "check", "references", + "exclude", + "like", } -_SQL_LINE_COMMENT = re.compile(r"--[^\n]*") -_SQL_BLOCK_COMMENT = re.compile(r"/\*.*?\*/", re.DOTALL) +# 引用符の開き文字 -> 閉じ文字。SQL のエスケープは引用符の二重化('' や "")。 +_SQL_QUOTES = {"\'": "\'", '"': '"', "`": "`"} + +def _skip_sql_quoted(text: str, index: int) -> int: + """text[index] の引用符から、その閉じ引用符の次の位置までを返す。""" + closer = _SQL_QUOTES[text[index]] + cursor = index + 1 + while cursor < len(text): + if text[cursor] == closer: + if cursor + 1 < len(text) and text[cursor + 1] == closer: + cursor += 2 # 二重化されたエスケープ。まだ閉じていない + continue + return cursor + 1 + cursor += 1 + return len(text) # 閉じていない引用符。末尾まで文字列として扱う def _strip_sql_comments(statement_text: str) -> str: - """コメントを取り除く。 + """コメントを取り除く。文字列リテラルと引用符つき識別子の中は触らない。 列定義の直前にコメント行が挟まると、区切り("," や "(")と列名の間に - 別の行が入り、列制約の外部キーを取りこぼす。行数は変えず中身だけ空にする。 + 別の行が入り、列制約の外部キーを取りこぼす。 + + ただし "--" や "/*" は文字列の中にも現れる(例: default \'--\')。 + そこをコメントの開始と誤ると、そこから行末までが消え、 + 本来拾えていた表制約の外部キーまで検出できなくなる。 """ - without_block = _SQL_BLOCK_COMMENT.sub("", statement_text) - return _SQL_LINE_COMMENT.sub("", without_block) + out: list[str] = [] + cursor = 0 + length = len(statement_text) + while cursor < length: + char = statement_text[cursor] + if char in _SQL_QUOTES: + end = _skip_sql_quoted(statement_text, cursor) + out.append(statement_text[cursor:end]) + cursor = end + continue + if statement_text.startswith("--", cursor): + newline = statement_text.find("\n", cursor) + cursor = length if newline == -1 else newline # 改行は残す + continue + if statement_text.startswith("/*", cursor): + end = statement_text.find("*/", cursor + 2) + cursor = length if end == -1 else end + 2 + continue + out.append(char) + cursor += 1 + return "".join(out) + +def _sql_table_definitions(statement_text: str) -> list[str]: + """CREATE TABLE の括弧内を、深さ0のカンマで定義単位に割る。 + + numeric(10,2) の "," は型パラメータの区切り、default \'a,b\' の "," は + ただの文字。どちらも列定義の区切りではない。ここを取り違えると + カンマの右隣("2" や "b")を列名として拾う。 + """ + start = -1 + depth = 0 + cursor = 0 + length = len(statement_text) + definitions: list[str] = [] + current: list[str] = [] + while cursor < length: + char = statement_text[cursor] + if char in _SQL_QUOTES: + end = _skip_sql_quoted(statement_text, cursor) + if start != -1: + current.append(statement_text[cursor:end]) + cursor = end + continue + if char == "(": + depth += 1 + if depth == 1 and start == -1: + start = cursor # 列定義リストの開き括弧。中身はまだ取らない + cursor += 1 + continue + elif char == ")": + depth -= 1 + if depth == 0 and start != -1: + definitions.append("".join(current)) + return [d.strip() for d in definitions if d.strip()] + elif char == "," and depth == 1: + definitions.append("".join(current)) + current = [] + cursor += 1 + continue + if start != -1: + current.append(char) + cursor += 1 + if start != -1 and current: + definitions.append("".join(current)) + return [d.strip() for d in definitions if d.strip()] + +_HAS_FOREIGN_KEY = re.compile(r"\bFOREIGN\s+KEY\b", re.IGNORECASE) + +# TODO: `ALTER TABLE t ADD COLUMN p uuid REFERENCES parent (id)` の列制約は +# まだ拾えない(列定義リストの括弧が無いため)。本PR以前も 0 本で、回帰ではない。 + def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, Any]]: matches: list[dict[str, Any]] = [] @@ -175,17 +266,25 @@ def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, } ) - for match in _COLUMN_LEVEL_FK.finditer(statement_text): - column = _strip_identifier_quotes(match.group("column")) + for definition in _sql_table_definitions(statement_text): + if _HAS_FOREIGN_KEY.search(definition): + continue # 表制約。上のループで拾い済み + reference = _REFERENCES_CLAUSE.search(definition) + if reference is None: + continue + head = _COLUMN_LEVEL_FK.match(definition) + if head is None: + continue + column = _strip_identifier_quotes(head.group("column")) if not column or column.lower() in _FK_COLUMN_RESERVED: continue - ref_columns_raw = match.group("ref_columns") + ref_columns_raw = reference.group("ref_columns") matches.append( { - "name": match.group("name") or "", + "name": head.group("name") or "", "table": table_name, "columns": [column], - "references_table": _strip_identifier_quotes(match.group("ref_table")), + "references_table": _strip_identifier_quotes(reference.group("ref_table")), "references_columns": _split_csv(ref_columns_raw) if ref_columns_raw else [], } ) diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py index aa75e83d..c10b87fa 100644 --- a/tests/test_sql_column_level_foreign_keys.py +++ b/tests/test_sql_column_level_foreign_keys.py @@ -105,3 +105,68 @@ def test_comment_before_a_column_does_not_hide_its_foreign_key() -> None: def test_references_inside_a_comment_is_not_counted() -> None: statement = "create table t (id uuid primary key); -- references nothing" assert _regex_foreign_keys(statement, "t") == [] + + +def test_double_dash_inside_a_string_literal_is_not_a_comment() -> None: + """文字列リテラルの中の "--" を行コメントと誤認すると、後続が丸ごと消える。 + + コメント除去を文字列リテラルを跨いで行うと、本PR以前は拾えていた + 表制約の外部キーまで検出できなくなる(回帰)。 + """ + statement = ( + "create table t (sep text default '--', " + "foreign key (parent_id) references parent (id));" + ) + assert len(_regex_foreign_keys(statement, "t")) == 1 + + column_level = ( + "create table t (sep text default '--', parent_id uuid references parent (id));" + ) + assert [fk["columns"][0] for fk in _regex_foreign_keys(column_level, "t")] == [ + "parent_id" + ] + + +def test_block_comment_opener_inside_a_string_literal_is_not_a_comment() -> None: + statement = ( + "create table t (glob text default '/*', " + "foreign key (parent_id) references parent (id));" + ) + assert len(_regex_foreign_keys(statement, "t")) == 1 + + +def test_doubled_quote_inside_a_string_literal_does_not_end_it() -> None: + """SQL のエスケープは引用符の二重化。'' を終端と誤ると以降の解釈がずれる。""" + statement = ( + "create table t (note text default 'it''s -- fine', " + "parent_id uuid references parent (id));" + ) + assert [fk["columns"][0] for fk in _regex_foreign_keys(statement, "t")] == [ + "parent_id" + ] + + +def test_double_dash_inside_a_quoted_identifier_is_not_a_comment() -> None: + statement = ( + 'create table t ("odd--name" text, ' + "foreign key (parent_id) references parent (id));" + ) + assert len(_regex_foreign_keys(statement, "t")) == 1 + + +def test_column_name_is_not_taken_from_inside_a_type_parameter_list() -> None: + """`numeric(10,2)` のカンマを列区切りと誤ると、列名が "2" になる。 + + 外部キーの本数は合っていても、どの列が親を参照しているかが誤る。 + """ + result = _regex_foreign_keys( + "create table t (amount numeric(10,2) references currency (code));", "t" + ) + assert [fk["columns"][0] for fk in result] == ["amount"] + + +def test_column_name_is_not_taken_from_inside_a_string_default() -> None: + result = _regex_foreign_keys( + "create table t (tag text default 'a,b' references taglist (code));", "t" + ) + assert [fk["columns"][0] for fk in result] == ["tag"] From b58dffca338de20caac7bf2b6f15c1358d346cc8 Mon Sep 17 00:00:00 2001 From: yohey-w Date: Mon, 14 Sep 2026 20:50:30 +0900 Subject: [PATCH 3/8] =?UTF-8?q?fix(parsing):=20=E6=94=B9=E8=A1=8C=E3=82=92?= =?UTF-8?q?=E6=BD=B0=E3=81=99=E5=89=8D=E3=81=AB=E3=82=B3=E3=83=A1=E3=83=B3?= =?UTF-8?q?=E3=83=88=E3=82=92=E8=90=BD=E3=81=A8=E3=81=99=E2=80=94=E2=80=94?= =?UTF-8?q?=E6=BD=B0=E3=81=97=E3=81=9F=E5=BE=8C=E3=81=A7=E3=81=AF=E6=96=87?= =?UTF-8?q?=E3=81=AE=E6=AE=8B=E3=82=8A=E5=85=A8=E9=83=A8=E3=81=8C=E6=B6=88?= =?UTF-8?q?=E3=81=88=E3=82=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 公開の抽出経路は tree-sitter のノード文字列を _normalize_ws で1行に潰してから _regex_foreign_keys へ渡す。潰したあとにコメントを除去すると、行コメント "--" の終端である改行が既に消えているため、コメント以降=文の残り全部が削られる。 実コメント1行が、その下にある既存の表制約FKまで道連れにしていた。 create table facility ( id uuid primary key, -- ER図の "||" 側を… tenant_id uuid not null references tenant (id), foreign key (report_definition_id) references report_definition (id) ); 修正前: FK=0(コメント行以降が全部消える) → 本コミット: FK=3 コメント除去を _normalize_ws より前に移した(tree経路・正規表現フォールバック経路の両方)。 あわせて、文字列リテラルの中の REFERENCES を外部キーと読み違えないようにした。 `default 'references parent (id)'` が偽のFKとして数えられていた。 参照句を探すときだけシングルクォートの中身を空白で潰す(長さは変えないので 取り出す位置はずれない)。引用符つき識別子は列名・テーブル名なので残す。 2件とも先にRED(e3ffeb9 で失敗)にしてから修正。全 7753 passed / 2 xfailed。 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_011xpq9A9SFvMUky6i6NLFz2 --- codd/parsing/schemas.py | 31 +++++++++++++++++---- tests/test_sql_column_level_foreign_keys.py | 28 +++++++++++++++++++ 2 files changed, 54 insertions(+), 5 deletions(-) diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index 77507158..2f092d15 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -245,6 +245,24 @@ def _sql_table_definitions(statement_text: str) -> list[str]: definitions.append("".join(current)) return [d.strip() for d in definitions if d.strip()] +def _blank_string_literals(definition: str) -> str: + """文字列リテラルの中身を空白で埋める(長さは変えない)。 + + `default \'references parent(id)\'` のような値を外部キーと読み違えないため。 + 引用符つき識別子(" と `)は列名・テーブル名なので残す。 + """ + out: list[str] = [] + cursor = 0 + while cursor < len(definition): + if definition[cursor] == "\'": + end = _skip_sql_quoted(definition, cursor) + out.append(" " * (end - cursor)) + cursor = end + continue + out.append(definition[cursor]) + cursor += 1 + return "".join(out) + _HAS_FOREIGN_KEY = re.compile(r"\bFOREIGN\s+KEY\b", re.IGNORECASE) # TODO: `ALTER TABLE t ADD COLUMN p uuid REFERENCES parent (id)` の列制約は @@ -269,7 +287,7 @@ def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, for definition in _sql_table_definitions(statement_text): if _HAS_FOREIGN_KEY.search(definition): continue # 表制約。上のループで拾い済み - reference = _REFERENCES_CLAUSE.search(definition) + reference = _REFERENCES_CLAUSE.search(_blank_string_literals(definition)) if reference is None: continue head = _COLUMN_LEVEL_FK.match(definition) @@ -334,6 +352,9 @@ def _extract_sql_schema_from_tree(root: Any, content: str, file_path: str) -> Sq for node in _iter_named_nodes(root): statement_text = _normalize_ws(_node_text(content_bytes, node)) + # 外部キーだけは改行を潰す前にコメントを落とす。潰したあとでは + # 行コメント "--" の終端(改行)が消え、文以降の既存FKまで削れる。 + fk_source = _normalize_ws(_strip_sql_comments(_node_text(content_bytes, node))) if node.type == "create_table": table_name = _sql_first_object_name(content_bytes, node) if not table_name: @@ -359,12 +380,12 @@ def _extract_sql_schema_from_tree(root: Any, content: str, file_path: str) -> Sq if constraint_text: constraints.append(constraint_text) schema.tables.append({"name": table_name, "columns": columns, "constraints": constraints}) - for foreign_key in _regex_foreign_keys(statement_text, table_name): + for foreign_key in _regex_foreign_keys(fk_source, table_name): _append_foreign_key(schema, foreign_key, seen_foreign_keys) elif node.type == "alter_table": table_name = _sql_first_object_name(content_bytes, node) if table_name: - for foreign_key in _regex_foreign_keys(statement_text, table_name): + for foreign_key in _regex_foreign_keys(fk_source, table_name): _append_foreign_key(schema, foreign_key, seen_foreign_keys) elif node.type == "create_index": index = _regex_create_index(statement_text) @@ -414,12 +435,12 @@ def _extract_sql_schema(content: str, file_path: str) -> SqlSchemaInfo: } ) schema.tables.append({"name": table_name, "columns": columns, "constraints": constraints}) - schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(table_match.group(0)), table_name)) + schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(_strip_sql_comments(table_match.group(0))), table_name)) for statement in re.findall(r"ALTER\s+TABLE\s+.*?;", content, re.IGNORECASE | re.DOTALL): match = re.search(r"ALTER\s+TABLE\s+([^\s;]+)", statement, re.IGNORECASE) if match: - schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(statement), match.group(1))) + schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(_strip_sql_comments(statement)), match.group(1))) for index_match in re.finditer(r"CREATE\s+(?:UNIQUE\s+)?INDEX\s+.*?;", content, re.IGNORECASE | re.DOTALL): index = _regex_create_index(_normalize_ws(index_match.group(0))) diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py index c10b87fa..ffe7fe58 100644 --- a/tests/test_sql_column_level_foreign_keys.py +++ b/tests/test_sql_column_level_foreign_keys.py @@ -170,3 +170,31 @@ def test_column_name_is_not_taken_from_inside_a_string_default() -> None: "create table t (tag text default 'a,b' references taglist (code));", "t" ) assert [fk["columns"][0] for fk in result] == ["tag"] + + +def test_public_path_keeps_foreign_keys_after_a_real_comment_line() -> None: + """公開の抽出経路は改行を空白に潰す。潰したあとにコメントを除去すると、 + 行コメント "--" の終端が失われ、文の残り全部(既存の表制約FKを含む)が消える。 + """ + from codd.parsing.schemas import _extract_sql_schema + + content = """create table facility ( + id uuid primary key, + -- ER図の "||" 側をNOT NULL FKとして表現する + tenant_id uuid not null references tenant (id), + item_set_id uuid not null references item_set (id), + foreign key (report_definition_id) references report_definition (id) + );""" + schema = _extract_sql_schema(content, "schema.sql") + columns = sorted(fk["columns"][0] for fk in schema.foreign_keys) + assert columns == ["item_set_id", "report_definition_id", "tenant_id"] + + +def test_references_inside_a_string_default_is_not_a_foreign_key() -> None: + statement = ( + "create table t (note text default 'references parent (id)', " + "parent_id uuid references parent (id));" + ) + assert [fk["columns"][0] for fk in _regex_foreign_keys(statement, "t")] == [ + "parent_id" + ] From a49bc9cdf79c822cc3b79af882d138282d83e0ae Mon Sep 17 00:00:00 2001 From: yohey-w Date: Mon, 14 Sep 2026 21:07:24 +0900 Subject: [PATCH 4/8] =?UTF-8?q?fix(parsing):=20=E6=96=B9=E8=A8=80=E3=81=94?= =?UTF-8?q?=E3=81=A8=E3=81=AE=E5=BC=95=E7=94=A8=E3=82=92=E5=AD=97=E5=8F=A5?= =?UTF-8?q?=E3=81=A8=E3=81=97=E3=81=A6=E9=A3=9B=E3=81=B0=E3=81=99=E2=80=94?= =?UTF-8?q?=E2=80=94=E5=BC=95=E7=94=A8=E3=81=AE=E5=8F=96=E3=82=8A=E9=81=95?= =?UTF-8?q?=E3=81=88=E3=81=AF=E5=BE=8C=E7=B6=9A=E3=81=AEFK=E3=82=92?= =?UTF-8?q?=E6=B6=88=E3=81=99?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit コメント除去は「引用の中に入らない」ことで成り立っている。引用の種類を 取りこぼすと、値の中の "--" を行コメントの開始と誤り、そこから行末までを 捨てて、本PR以前は拾えていた表制約の外部キーまで消してしまう。 新たに字句として扱うもの(いずれも修正前は FK=0・修正後 FK=1): PostgreSQL のドル引用 default $$--$$ MySQL のバックスラッシュ default 'it\'s -- fine' T-SQL の角括弧識別子 [odd--name] あわせて、表制約の検索を文字列リテラルを空白で潰した版に対して行うようにした。 `default 'FOREIGN KEY (fake) REFERENCES fake (id)'` が偽の外部キーとして 数えられていた(これは本PR以前からの誤検出で、回帰ではない)。 4件とも先にRED(b58dffc で失敗)にしてから修正。全 7757 passed / 2 xfailed。 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_011xpq9A9SFvMUky6i6NLFz2 --- codd/parsing/schemas.py | 40 +++++++++++++++++---- tests/test_sql_column_level_foreign_keys.py | 38 ++++++++++++++++++++ 2 files changed, 72 insertions(+), 6 deletions(-) diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index 2f092d15..dd4253e3 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -154,13 +154,38 @@ def _sql_first_object_name(content_bytes: bytes, node: Any) -> str: } # 引用符の開き文字 -> 閉じ文字。SQL のエスケープは引用符の二重化('' や "")。 -_SQL_QUOTES = {"\'": "\'", '"': '"', "`": "`"} +# 角括弧は T-SQL の識別子(閉じは "]"、エスケープは "]]")。 +_SQL_QUOTES = {"\'": "\'", '"': '"', "`": "`", "[": "]"} + +# PostgreSQL のドル引用($$ ... $$ / $tag$ ... $tag$)。 +_DOLLAR_QUOTE = re.compile(r"\$(\w*)\$") + +def _dollar_quote_tag(text: str, index: int) -> str | None: + match = _DOLLAR_QUOTE.match(text, index) + return match.group(0) if match else None def _skip_sql_quoted(text: str, index: int) -> int: - """text[index] の引用符から、その閉じ引用符の次の位置までを返す。""" - closer = _SQL_QUOTES[text[index]] + """text[index] の引用符から、その閉じ引用符の次の位置までを返す。 + + 引用の中身は「コメントでも区切りでもない」ただの文字として飛ばす。 + ここを甘くすると、値の中の "--" を行コメントの開始と誤り、 + そこから行末までを捨てて既存の外部キーまで消してしまう。 + """ + tag = _dollar_quote_tag(text, index) + if tag is not None: # $$ ... $$ / $tag$ ... $tag$ + end = text.find(tag, index + len(tag)) + return len(text) if end == -1 else end + len(tag) + + quote = text[index] + closer = _SQL_QUOTES[quote] cursor = index + 1 while cursor < len(text): + # MySQL はバックスラッシュでエスケープする('it\\'s')。 + # PostgreSQL の標準文字列ではバックスラッシュはただの文字だが、 + # 誤って飛ばしても失うのは1文字で、引用の終端は取り違えない。 + if quote == "\'" and text[cursor] == "\\" and cursor + 1 < len(text): + cursor += 2 + continue if text[cursor] == closer: if cursor + 1 < len(text) and text[cursor + 1] == closer: cursor += 2 # 二重化されたエスケープ。まだ閉じていない @@ -184,7 +209,7 @@ def _strip_sql_comments(statement_text: str) -> str: length = len(statement_text) while cursor < length: char = statement_text[cursor] - if char in _SQL_QUOTES: + if char in _SQL_QUOTES or _dollar_quote_tag(statement_text, cursor): end = _skip_sql_quoted(statement_text, cursor) out.append(statement_text[cursor:end]) cursor = end @@ -216,7 +241,7 @@ def _sql_table_definitions(statement_text: str) -> list[str]: current: list[str] = [] while cursor < length: char = statement_text[cursor] - if char in _SQL_QUOTES: + if char in _SQL_QUOTES or _dollar_quote_tag(statement_text, cursor): end = _skip_sql_quoted(statement_text, cursor) if start != -1: current.append(statement_text[cursor:end]) @@ -273,7 +298,10 @@ def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, matches: list[dict[str, Any]] = [] statement_text = _strip_sql_comments(statement_text) - for match in _TABLE_LEVEL_FK.finditer(statement_text): + # 表制約は文字列の中身を空白で潰した版に対して探す。 + # `default 'FOREIGN KEY (fake) REFERENCES fake (id)'` のような値を + # 外部キーとして数えないため。長さは変わらないので位置はずれない。 + for match in _TABLE_LEVEL_FK.finditer(_blank_string_literals(statement_text)): matches.append( { "name": match.group("name") or "", diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py index ffe7fe58..6a2e1f52 100644 --- a/tests/test_sql_column_level_foreign_keys.py +++ b/tests/test_sql_column_level_foreign_keys.py @@ -198,3 +198,41 @@ def test_references_inside_a_string_default_is_not_a_foreign_key() -> None: assert [fk["columns"][0] for fk in _regex_foreign_keys(statement, "t")] == [ "parent_id" ] + + +def test_table_level_foreign_key_written_inside_a_string_is_not_counted() -> None: + statement = ( + "create table t (note text default 'FOREIGN KEY (fake) REFERENCES fake (id)', " + "p uuid references parent (id));" + ) + assert [fk["columns"][0] for fk in _regex_foreign_keys(statement, "t")] == ["p"] + + +@pytest.mark.parametrize( + "label, statement", + [ + ( + "PostgreSQL のドル引用", + "create table t (s text default $$--$$, " + "foreign key (p) references parent (id));", + ), + ( + "MySQL のバックスラッシュエスケープ", + "create table t (s text default 'it\\'s -- fine', " + "foreign key (p) references parent (id));", + ), + ( + "T-SQL の角括弧識別子", + "create table t ([odd--name] text, " + "foreign key (p) references parent (id));", + ), + ], +) +def test_dialect_quoting_does_not_swallow_a_following_foreign_key( + label: str, statement: str +) -> None: + """方言ごとの引用の中の "--" を行コメントと誤ると、後続の表制約FKが消える。 + + いずれも本PR以前は拾えていたケースなので、取りこぼしは回帰になる。 + """ + assert len(_regex_foreign_keys(statement, "t")) == 1, label From 916a659779963c52139247735f11871e84928ea6 Mon Sep 17 00:00:00 2001 From: yohey-w Date: Mon, 14 Sep 2026 21:24:21 +0900 Subject: [PATCH 5/8] =?UTF-8?q?fix(parsing):=20=E3=83=90=E3=83=83=E3=82=AF?= =?UTF-8?q?=E3=82=B9=E3=83=A9=E3=83=83=E3=82=B7=E3=83=A5=E3=81=AF=E9=96=89?= =?UTF-8?q?=E3=81=98=E3=82=8B=E8=AA=AD=E3=81=BF=E6=96=B9=E3=82=92=E6=8E=A1?= =?UTF-8?q?=E3=82=8B=E2=80=94=E2=80=94=E9=96=89=E3=81=98=E3=81=AA=E3=81=84?= =?UTF-8?q?=E8=AA=AD=E3=81=BF=E3=81=AF=E6=AE=8B=E3=82=8A=E5=85=A8=E9=83=A8?= =?UTF-8?q?=E3=82=92=E9=A3=B2=E3=82=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 方言でバックスラッシュの意味が違う。 MySQL: 'it\'s' → \' はエスケープで、文字列はまだ閉じない PostgreSQL: 'C:\' → \ はただの文字で、次の ' で閉じる 前コミットは常にエスケープと読んだため、PostgreSQL の標準文字列 `default '\'` で文字列が閉じなくなり、以降を全部リテラルとみなして 後続の表制約FKを落としていた(main=1 / 前コミット=0)。 見分けはつかないので、まずエスケープありで読み、それで閉じないときだけ エスケープなしで読み直す。閉じない=残り全部を失う、なので閉じる読み方を採る。 両方言とも後続の外部キーを保てる。 MySQL の `default 1--2`("--" の直後に空白が無ければ減算)は取りこぼす。 空白を必須にすると PostgreSQL / SQLite の `--コメント`(空白なし・実務で 最も多い書き方)を取りこぼすため、意図的な選択としてコードにも注記した。 RED(a49bc9c で失敗)→ GREEN。全 7758 passed / 2 xfailed。 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_011xpq9A9SFvMUky6i6NLFz2 --- codd/parsing/schemas.py | 25 ++++++++++++++++----- tests/test_sql_column_level_foreign_keys.py | 13 +++++++++++ 2 files changed, 33 insertions(+), 5 deletions(-) diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index dd4253e3..40ea1472 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -177,13 +177,24 @@ def _skip_sql_quoted(text: str, index: int) -> int: return len(text) if end == -1 else end + len(tag) quote = text[index] - closer = _SQL_QUOTES[quote] + if quote == "\'": + # バックスラッシュを方言差の分かれ道として扱う。 + # MySQL: 'it\\'s' → \\' はエスケープで、文字列はまだ閉じない + # PostgreSQL: 'C:\\' → \\ はただの文字で、次の ' で閉じる + # 見分けはつかないので、まずエスケープありで読み、それだと + # 閉じないときだけエスケープなしで読み直す。 + # 「閉じない」=残り全部を文字列とみなす=後続のFKを全部失う、なので + # 閉じる読み方があるならそちらを採る。 + escaped = _scan_quoted(text, index, quote, quote, backslash_escapes=True) + if escaped < len(text): + return escaped + return _scan_quoted(text, index, quote, quote, backslash_escapes=False) + return _scan_quoted(text, index, quote, _SQL_QUOTES[quote], backslash_escapes=False) + +def _scan_quoted(text: str, index: int, quote: str, closer: str, *, backslash_escapes: bool) -> int: cursor = index + 1 while cursor < len(text): - # MySQL はバックスラッシュでエスケープする('it\\'s')。 - # PostgreSQL の標準文字列ではバックスラッシュはただの文字だが、 - # 誤って飛ばしても失うのは1文字で、引用の終端は取り違えない。 - if quote == "\'" and text[cursor] == "\\" and cursor + 1 < len(text): + if backslash_escapes and text[cursor] == "\\" and cursor + 1 < len(text): cursor += 2 continue if text[cursor] == closer: @@ -214,6 +225,10 @@ def _strip_sql_comments(statement_text: str) -> str: out.append(statement_text[cursor:end]) cursor = end continue + # 注: MySQL は "--" の直後に空白が要るので `default 1--2` は減算だが、 + # PostgreSQL / SQLite は空白なしの `--コメント` もコメント。 + # 空白を必須にすると後者(実務で最も多い書き方)を取りこぼすので、 + # ここでは空白を要求しない。`1--2` は取りこぼす(既知・方言の分かれ道)。 if statement_text.startswith("--", cursor): newline = statement_text.find("\n", cursor) cursor = length if newline == -1 else newline # 改行は残す diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py index 6a2e1f52..0bf37776 100644 --- a/tests/test_sql_column_level_foreign_keys.py +++ b/tests/test_sql_column_level_foreign_keys.py @@ -236,3 +236,16 @@ def test_dialect_quoting_does_not_swallow_a_following_foreign_key( いずれも本PR以前は拾えていたケースなので、取りこぼしは回帰になる。 """ assert len(_regex_foreign_keys(statement, "t")) == 1, label + + +def test_trailing_backslash_in_a_standard_string_still_closes_it() -> None: + """バックスラッシュは方言の分かれ道。 + + MySQL では `\\'` がエスケープ、PostgreSQL の標準文字列ではただの文字。 + エスケープと読んだ結果、文字列が閉じなくなる(=残り全部を飲み込む)なら、 + 閉じる読み方を採る。どちらの方言でも後続の外部キーを失わない。 + """ + postgres = "create table t (s text default '\\', foreign key (p) references parent (id));" + mysql = "create table t (s text default 'it\\'s -- fine', foreign key (p) references parent (id));" + assert len(_regex_foreign_keys(postgres, "t")) == 1 + assert len(_regex_foreign_keys(mysql, "t")) == 1 From 31abfacf0c2aea52f82890ca6ccfd411cfdcc31b Mon Sep 17 00:00:00 2001 From: yohey-w Date: Mon, 14 Sep 2026 21:32:42 +0900 Subject: [PATCH 6/8] =?UTF-8?q?fix(parsing):=20`1--2`=20=E3=81=AF=E6=B8=9B?= =?UTF-8?q?=E7=AE=97=E2=80=94=E2=80=94=E4=B8=A1=E5=81=B4=E3=81=8C=E5=BC=8F?= =?UTF-8?q?=E3=81=AB=E8=A6=8B=E3=81=88=E3=82=8B=E3=81=A8=E3=81=8D=E3=81=A0?= =?UTF-8?q?=E3=81=91=E6=BC=94=E7=AE=97=E5=AD=90=E3=81=A8=E8=AA=AD=E3=82=80?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit MySQL は "--" の直後に空白を要求するので `default 1--2` は「1 から -2 を引く」。 PostgreSQL / SQLite は空白なしの `--コメント` もコメントなので、 「空白があればコメント」という規則では後者(実務で最も多い書き方)を落とす。 そこで両側で切る。直前が値の終わり(英数字・"_"・")")で、直後が数値か 開き括弧のときだけ演算子と読む。`--コメント` は直後が文字なのでコメントのまま。 create table t (n int default 1--2, foreign key (p) references parent (id)); main=1 → 修正前=0(コメント扱いで後続が消える) → 本コミット=1 公開経路(_extract_sql_schema)でも main と同じ1本に戻ることを確認した。 前コミットで「取りこぼす」と注記した既知の分かれ道を、注記ごと解消した。 RED(916a659 で失敗)→ GREEN。全 7759 passed / 2 xfailed。 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_011xpq9A9SFvMUky6i6NLFz2 --- codd/parsing/schemas.py | 22 ++++++++++++++++----- tests/test_sql_column_level_foreign_keys.py | 20 +++++++++++++++++++ 2 files changed, 37 insertions(+), 5 deletions(-) diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index 40ea1472..9e3a0a2c 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -205,6 +205,20 @@ def _scan_quoted(text: str, index: int, quote: str, closer: str, *, backslash_es cursor += 1 return len(text) # 閉じていない引用符。末尾まで文字列として扱う +def _is_double_minus_operator(text: str, index: int) -> bool: + """`1--2`(1 から -2 を引く)の "--" か。 + + MySQL は "--" の直後に空白を要求するのでこれは演算子。PostgreSQL / SQLite は + 空白なしの `--コメント` もコメントなので、空白の有無だけでは切れない。 + そこで「両側が式に見えるとき」だけ演算子と読む——直前が値の終わりで、 + 直後が数値か開き括弧のときに限る。`--コメント`(直後が文字)はコメントのまま。 + """ + if index == 0: + return False + before = text[index - 1] + after = text[index + 2] if index + 2 < len(text) else "" + return (before.isalnum() or before in "_)") and (after.isdigit() or after == "(") + def _strip_sql_comments(statement_text: str) -> str: """コメントを取り除く。文字列リテラルと引用符つき識別子の中は触らない。 @@ -225,11 +239,9 @@ def _strip_sql_comments(statement_text: str) -> str: out.append(statement_text[cursor:end]) cursor = end continue - # 注: MySQL は "--" の直後に空白が要るので `default 1--2` は減算だが、 - # PostgreSQL / SQLite は空白なしの `--コメント` もコメント。 - # 空白を必須にすると後者(実務で最も多い書き方)を取りこぼすので、 - # ここでは空白を要求しない。`1--2` は取りこぼす(既知・方言の分かれ道)。 - if statement_text.startswith("--", cursor): + if statement_text.startswith("--", cursor) and not _is_double_minus_operator( + statement_text, cursor + ): newline = statement_text.find("\n", cursor) cursor = length if newline == -1 else newline # 改行は残す continue diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py index 0bf37776..68457e28 100644 --- a/tests/test_sql_column_level_foreign_keys.py +++ b/tests/test_sql_column_level_foreign_keys.py @@ -249,3 +249,23 @@ def test_trailing_backslash_in_a_standard_string_still_closes_it() -> None: mysql = "create table t (s text default 'it\\'s -- fine', foreign key (p) references parent (id));" assert len(_regex_foreign_keys(postgres, "t")) == 1 assert len(_regex_foreign_keys(mysql, "t")) == 1 + + +def test_double_minus_between_operands_is_subtraction_not_a_comment() -> None: + """`default 1--2` は「1 から -2 を引く」。コメントと誤ると後続FKが消える。 + + 直後が文字なら(`--コメント`)コメントのまま。空白の有無では切らない—— + PostgreSQL / SQLite は空白なしの `--コメント` もコメントだから。 + """ + from codd.parsing.schemas import _extract_sql_schema + + subtraction = ( + "create table t (n int default 1--2, foreign key (p) references parent (id));" + ) + assert len(_regex_foreign_keys(subtraction, "t")) == 1 + assert len(_extract_sql_schema(subtraction, "schema.sql").foreign_keys) == 1 + + comment = ( + "create table t (a int,\n--コメント\n foreign key (p) references parent (id));" + ) + assert len(_extract_sql_schema(comment, "schema.sql").foreign_keys) == 1 From 14fb2c6f25a66c93dfee9a61533b7d026c5f9b49 Mon Sep 17 00:00:00 2001 From: yohey-w Date: Mon, 14 Sep 2026 21:49:42 +0900 Subject: [PATCH 7/8] =?UTF-8?q?fix(parsing):=20=E8=A1=A8=E5=88=B6=E7=B4=84?= =?UTF-8?q?=E3=81=AF=E3=82=B3=E3=83=A1=E3=83=B3=E3=83=88=E9=99=A4=E5=8E=BB?= =?UTF-8?q?=E3=82=92=E9=80=9A=E3=81=95=E3=81=AA=E3=81=84=E2=80=94=E2=80=94?= =?UTF-8?q?=E5=AD=97=E5=8F=A5=E3=81=AE=E8=AA=AD=E3=81=BF=E9=81=95=E3=81=84?= =?UTF-8?q?=E3=81=A7=E5=BE=93=E6=9D=A5=E3=81=AE=E6=A4=9C=E5=87=BA=E3=82=92?= =?UTF-8?q?=E6=B8=9B=E3=82=89=E3=81=95=E3=81=AA=E3=81=84?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit コメント除去は列制約(本PRで新しく拾えるようにした側)のためだけに要る仕組み。 それを表制約の経路にも通していたため、方言の字句をひとつ読み違えるたびに 本PR以前から拾えていた表制約FKまで道連れになっていた。実際、レビューのたびに 同じ形の指摘が出た('--' / ドル引用 / バックスラッシュ / 角括弧 / 1--2 / a--b)。 create table t (a int, b int, c int as (a--b), foreign key (p) references parent (id)); main=1 → 修正前=0 → 本コミット=1 a-/*gap*/-b(公開経路で二重にコメント除去していた) main=1 → 修正前=0 → 本コミット=1 経路を分けた。表制約は元の文(文字列リテラルだけ空白化)から探し、 列制約はコメント除去済みの文から探す。これで字句判定を誤っても、 失うのは新しく増えた分だけで、従来の検出結果は構造的に減らない。 公開経路が事前に除去した版は引数で渡し、二重除去もやめた。 RED(31abfac で失敗)→ GREEN。全 7761 passed / 2 xfailed。 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_011xpq9A9SFvMUky6i6NLFz2 --- codd/parsing/schemas.py | 30 ++++++++++++++++----- tests/test_sql_column_level_foreign_keys.py | 24 +++++++++++++++++ 2 files changed, 47 insertions(+), 7 deletions(-) diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index 9e3a0a2c..9e0e4045 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -321,9 +321,23 @@ def _blank_string_literals(definition: str) -> str: # まだ拾えない(列定義リストの括弧が無いため)。本PR以前も 0 本で、回帰ではない。 -def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, Any]]: +def _regex_foreign_keys( + statement_text: str, + table_name: str, + comment_free_text: str | None = None, +) -> list[dict[str, Any]]: + """外部キーを2通りの書き方から拾う。 + + 表制約はコメントを除去しない元の文から探す。コメント除去は列制約の + ためだけに要る仕組みで、そこで方言の字句("--" が減算か・引用の閉じ方)を + 読み違えると、本PR以前から拾えていた表制約まで消えてしまう。 + 経路を分けておけば、字句判定を誤っても失うのは新しく増えた分だけで、 + 従来の検出結果は減らない。 + + comment_free_text は、呼び出し側が改行を空白に潰す前にコメントを + 除去した版(行コメントの終端は改行なので、潰した後では取れない)。 + """ matches: list[dict[str, Any]] = [] - statement_text = _strip_sql_comments(statement_text) # 表制約は文字列の中身を空白で潰した版に対して探す。 # `default 'FOREIGN KEY (fake) REFERENCES fake (id)'` のような値を @@ -339,7 +353,9 @@ def _regex_foreign_keys(statement_text: str, table_name: str) -> list[dict[str, } ) - for definition in _sql_table_definitions(statement_text): + if comment_free_text is None: + comment_free_text = _strip_sql_comments(statement_text) + for definition in _sql_table_definitions(comment_free_text): if _HAS_FOREIGN_KEY.search(definition): continue # 表制約。上のループで拾い済み reference = _REFERENCES_CLAUSE.search(_blank_string_literals(definition)) @@ -435,12 +451,12 @@ def _extract_sql_schema_from_tree(root: Any, content: str, file_path: str) -> Sq if constraint_text: constraints.append(constraint_text) schema.tables.append({"name": table_name, "columns": columns, "constraints": constraints}) - for foreign_key in _regex_foreign_keys(fk_source, table_name): + for foreign_key in _regex_foreign_keys(statement_text, table_name, fk_source): _append_foreign_key(schema, foreign_key, seen_foreign_keys) elif node.type == "alter_table": table_name = _sql_first_object_name(content_bytes, node) if table_name: - for foreign_key in _regex_foreign_keys(fk_source, table_name): + for foreign_key in _regex_foreign_keys(statement_text, table_name, fk_source): _append_foreign_key(schema, foreign_key, seen_foreign_keys) elif node.type == "create_index": index = _regex_create_index(statement_text) @@ -490,12 +506,12 @@ def _extract_sql_schema(content: str, file_path: str) -> SqlSchemaInfo: } ) schema.tables.append({"name": table_name, "columns": columns, "constraints": constraints}) - schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(_strip_sql_comments(table_match.group(0))), table_name)) + schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(table_match.group(0)), table_name, _normalize_ws(_strip_sql_comments(table_match.group(0))))) for statement in re.findall(r"ALTER\s+TABLE\s+.*?;", content, re.IGNORECASE | re.DOTALL): match = re.search(r"ALTER\s+TABLE\s+([^\s;]+)", statement, re.IGNORECASE) if match: - schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(_strip_sql_comments(statement)), match.group(1))) + schema.foreign_keys.extend(_regex_foreign_keys(_normalize_ws(statement), match.group(1), _normalize_ws(_strip_sql_comments(statement)))) for index_match in re.finditer(r"CREATE\s+(?:UNIQUE\s+)?INDEX\s+.*?;", content, re.IGNORECASE | re.DOTALL): index = _regex_create_index(_normalize_ws(index_match.group(0))) diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py index 68457e28..83d874c3 100644 --- a/tests/test_sql_column_level_foreign_keys.py +++ b/tests/test_sql_column_level_foreign_keys.py @@ -269,3 +269,27 @@ def test_double_minus_between_operands_is_subtraction_not_a_comment() -> None: "create table t (a int,\n--コメント\n foreign key (p) references parent (id));" ) assert len(_extract_sql_schema(comment, "schema.sql").foreign_keys) == 1 + + +@pytest.mark.parametrize( + "label, statement", + [ + ("生成列の減算 a--b", "create table t (a int, b int, c int as (a--b), " + "foreign key (p) references parent (id));"), + ("ブロックコメントを挟んだ減算 a-/*gap*/-b", + "create table t (a int, b int, c int as (a-/*gap*/-b), " + "foreign key (p) references parent (id));"), + ], +) +def test_comment_lexing_never_costs_a_table_level_foreign_key( + label: str, statement: str +) -> None: + """表制約はコメント除去を通さない経路で拾う。 + + コメント除去は列制約のためだけに要る仕組み。方言の字句を読み違えても、 + 本PR以前から拾えていた表制約まで道連れにしてはいけない。 + """ + from codd.parsing.schemas import _extract_sql_schema + + assert len(_regex_foreign_keys(statement, "t")) == 1, label + assert len(_extract_sql_schema(statement, "schema.sql").foreign_keys) == 1, label From a5550c3b58868a46ab4583b2fa490910224a21a1 Mon Sep 17 00:00:00 2001 From: yohey-w Date: Mon, 14 Sep 2026 22:06:48 +0900 Subject: [PATCH 8/8] =?UTF-8?q?fix(parsing):=20=E8=A1=A8=E5=88=B6=E7=B4=84?= =?UTF-8?q?=E3=81=AF=E5=85=83=E3=81=AE=E6=96=87=E3=82=92=E3=81=9D=E3=81=AE?= =?UTF-8?q?=E3=81=BE=E3=81=BE=E6=8E=A2=E3=81=99=E2=80=94=E2=80=94=E6=89=8B?= =?UTF-8?q?=E3=82=92=E5=85=A5=E3=82=8C=E3=81=AA=E3=81=84=E3=81=93=E3=81=A8?= =?UTF-8?q?=E3=81=8C=E9=80=80=E8=A1=8C=E3=81=97=E3=81=AA=E3=81=84=E4=BF=9D?= =?UTF-8?q?=E8=A8=BC=E3=81=AB=E3=81=AA=E3=82=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 前コミットは表制約を「文字列リテラルを空白で潰した版」から探していた。 偽FK(`default 'FOREIGN KEY (fk) REFERENCES fake (id)'`)を消せる代わりに、 SQLite が受け付ける単一引用符の識別子まで巻き添えで消していた。 CREATE TABLE child(parent_id INTEGER, FOREIGN KEY(parent_id) REFERENCES 'parent'(id)); main=1 → 前コミット=0 → 本コミット=1 偽FKは本PR以前からある誤検出で、本PRの主題(列制約の外部キーを拾う)ではない。 表制約の入力に手を入れないと決めれば、その経路の出力は main と同一になり、 退行しないことが手続きとして保証される。偽FKは既知の未対応としてテストに残した。 45ケースのDDLコーパスで公開経路を main と差分比較: 落とした FK は0件、 新たに拾えた FK は13件。全 7762 passed / 2 xfailed。 Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_011xpq9A9SFvMUky6i6NLFz2 --- codd/parsing/schemas.py | 11 +++++---- tests/test_sql_column_level_foreign_keys.py | 25 +++++++++++++++++++-- 2 files changed, 30 insertions(+), 6 deletions(-) diff --git a/codd/parsing/schemas.py b/codd/parsing/schemas.py index 9e0e4045..792b2438 100644 --- a/codd/parsing/schemas.py +++ b/codd/parsing/schemas.py @@ -339,10 +339,13 @@ def _regex_foreign_keys( """ matches: list[dict[str, Any]] = [] - # 表制約は文字列の中身を空白で潰した版に対して探す。 - # `default 'FOREIGN KEY (fake) REFERENCES fake (id)'` のような値を - # 外部キーとして数えないため。長さは変わらないので位置はずれない。 - for match in _TABLE_LEVEL_FK.finditer(_blank_string_literals(statement_text)): + # 表制約は元の文をそのまま探す=本PR以前とまったく同じ手続き。 + # 文字列の中身を空白で潰せば `default 'FOREIGN KEY (fk) REFERENCES fake (id)'` + # のような偽の外部キーは消せるが、SQLite が受け付ける + # `REFERENCES 'parent'(id)`(単一引用符の識別子)まで巻き添えで消える。 + # 偽FKは本PR以前からある誤検出で、本PRの主題ではない。 + # ここを触らないと決めておけば、表制約の検出結果は構造的に減らない。 + for match in _TABLE_LEVEL_FK.finditer(statement_text): matches.append( { "name": match.group("name") or "", diff --git a/tests/test_sql_column_level_foreign_keys.py b/tests/test_sql_column_level_foreign_keys.py index 83d874c3..fce7c404 100644 --- a/tests/test_sql_column_level_foreign_keys.py +++ b/tests/test_sql_column_level_foreign_keys.py @@ -200,12 +200,33 @@ def test_references_inside_a_string_default_is_not_a_foreign_key() -> None: ] -def test_table_level_foreign_key_written_inside_a_string_is_not_counted() -> None: +def test_single_quoted_identifier_after_references_is_still_a_foreign_key() -> None: + """SQLite は識別子を単一引用符でも書ける(`REFERENCES \'parent\'(id)`)。 + + 文字列リテラルを空白で潰してから表制約を探すと、これを巻き添えで消す。 + 表制約は元の文をそのまま探す=本PR以前と同じ手続きにしてある。 + """ + statement = ( + "CREATE TABLE child(parent_id INTEGER, " + "FOREIGN KEY(parent_id) REFERENCES 'parent'(id));" + ) + assert len(_regex_foreign_keys(statement, "child")) == 1 + + +def test_table_level_foreign_key_written_inside_a_string_is_still_counted() -> None: + """既知の未対応(本PR以前からの誤検出・回帰ではない)。 + + 文字列の中に表制約の形が書いてあると外部キーとして数える。 + 潰すと上の SQLite の識別子まで消えるので、本PRでは触らない。 + """ statement = ( "create table t (note text default 'FOREIGN KEY (fake) REFERENCES fake (id)', " "p uuid references parent (id));" ) - assert [fk["columns"][0] for fk in _regex_foreign_keys(statement, "t")] == ["p"] + assert sorted(fk["columns"][0] for fk in _regex_foreign_keys(statement, "t")) == [ + "fake", + "p", + ] @pytest.mark.parametrize(