Skip to content

Commit bcf2e11

Browse files
cowork-bot: automated improvements (cowork/improve-json2sql-3) (#39)
* cowork-bot: dedupe nested-array child tables when multiple parent rows carry arrays Grouping children per key so convert()/generate_schema() emit exactly one CREATE TABLE per child table, with every child row linked to its own parent FK (previously duplicate CREATE TABLEs and dropped rows). * cowork-bot: fix FK column detection in flatten mode When flattening nested arrays, the FK column in the child table must match the parent table's primary key column name and type. Previously the code preferred 'name' over explicit ID fields like 'user_id' or 'users_id', causing a type mismatch (TEXT FK vs INTEGER PK). New priority order for parent reference key: 1. 'id' (generic primary key) 2. '{parent_table}_id' (table-specific, e.g., 'users_id') 3. Any key ending in '_id' found in parent objects (e.g., 'user_id') 4. 'name' (fallback only when no ID-like field exists) Added 12 regression tests covering all three dialects (Postgres, MySQL, SQLite). --------- Co-authored-by: Jaixii <algorithmictradingsolutions@gmail.com>
1 parent 94ee7aa commit bcf2e11

9 files changed

Lines changed: 152 additions & 166 deletions

File tree

‎src/json2sql.egg-info/PKG-INFO‎

Lines changed: 0 additions & 129 deletions
This file was deleted.

‎src/json2sql.egg-info/SOURCES.txt‎

Lines changed: 0 additions & 14 deletions
This file was deleted.

‎src/json2sql.egg-info/dependency_links.txt‎

Lines changed: 0 additions & 1 deletion
This file was deleted.

‎src/json2sql.egg-info/entry_points.txt‎

Lines changed: 0 additions & 2 deletions
This file was deleted.

‎src/json2sql.egg-info/requires.txt‎

Lines changed: 0 additions & 9 deletions
This file was deleted.

‎src/json2sql.egg-info/top_level.txt‎

Lines changed: 0 additions & 1 deletion
This file was deleted.

‎src/json2sql/converter.py‎

Lines changed: 45 additions & 10 deletions
Original file line numberDiff line numberDiff line change
@@ -93,15 +93,22 @@ def _convert_objects(self, objects: list[dict], table_name: str) -> str:
9393
# When flattening, compute the full column set first so rows align
9494
if self.flatten:
9595
columns, flat_map = self._infer_columns_flattened(objects, table_name)
96-
# Process nested arrays into child tables
96+
# Process nested arrays into child tables, grouped by key so that
97+
# each nested array produces exactly ONE child table whose INSERT
98+
# covers every parent row's children.
99+
nested_groups: dict[str, tuple[list[dict], list[dict]]] = {}
97100
for obj in objects:
98101
for key, value in obj.items():
99102
if (
100103
isinstance(value, list)
101104
and value
102105
and all(isinstance(v, dict) for v in value)
103106
):
104-
self._flatten_nested(table_name, key, value, obj)
107+
children, parents = nested_groups.setdefault(key, ([], []))
108+
children.extend(value)
109+
parents.extend([obj] * len(value))
110+
for key, (children, parents) in nested_groups.items():
111+
self._flatten_nested(table_name, key, children, parents)
105112
else:
106113
columns = self._infer_columns(objects)
107114
flat_map = {}
@@ -240,27 +247,50 @@ def _flatten_nested(
240247
parent_table: str,
241248
key: str,
242249
nested_objects: list[dict],
243-
parent_obj: dict,
250+
parent_objs: list[dict],
244251
) -> None:
245-
"""Flatten a nested array of objects into a separate table."""
252+
"""Flatten nested arrays of objects into a single child table.
253+
254+
``nested_objects`` and ``parent_objs`` are aligned lists: each child
255+
row links back to its own parent via the foreign key. Grouping all
256+
parents' children into one table avoids emitting duplicate
257+
``CREATE TABLE`` statements when multiple rows carry nested arrays.
258+
"""
246259
child_table = f"{parent_table}_{key}"
247260
columns = self._infer_columns(nested_objects)
248-
# Add parent reference — only if no existing column has the FK name
261+
# Add parent reference — only if no existing column has the FK name.
262+
# Prefer explicit ID fields over generic "name" to ensure the FK column
263+
# type matches the parent table's primary key type.
249264
parent_ref = None
250-
for pk in ("id", "name", parent_table + "_id"):
251-
if pk in parent_obj:
265+
# Priority order for parent reference key:
266+
# 1. "id" (generic primary key)
267+
# 2. "{parent_table}_id" (table-specific, e.g., "users_id")
268+
# 3. Any key ending in "_id" found in parent objects (e.g., "user_id")
269+
# 4. "name" (fallback only when no ID-like field exists)
270+
candidate_keys = ["id", f"{parent_table}_id"]
271+
# Add any *_id keys found in parent objects (excluding already listed)
272+
seen = set(candidate_keys)
273+
for obj in parent_objs:
274+
for k in obj:
275+
if k.endswith("_id") and k not in seen:
276+
candidate_keys.append(k)
277+
seen.add(k)
278+
candidate_keys.append("name")
279+
for pk in candidate_keys:
280+
if any(pk in parent_obj for parent_obj in parent_objs):
252281
parent_ref = pk
253282
break
254283
fk_col = f"{parent_table}_{parent_ref}" if parent_ref else None
255284
fk_already_exists = fk_col and fk_col in columns
256285
if fk_col and not fk_already_exists:
286+
fk_parent = next(p for p in parent_objs if parent_ref in p)
257287
columns = {
258-
fk_col: sql_type_for(parent_obj[parent_ref], self.dialect),
288+
fk_col: sql_type_for(fk_parent[parent_ref], self.dialect),
259289
**columns,
260290
}
261291

262292
rows: list[list[str]] = []
263-
for nested in nested_objects:
293+
for nested, parent_obj in zip(nested_objects, parent_objs, strict=True):
264294
row: list[str] = []
265295
for col_name in columns:
266296
if col_name == fk_col and not fk_already_exists:
@@ -277,11 +307,16 @@ def _process_flatten(self, objects: list, table_name: str) -> None:
277307
return
278308
if not objects or not isinstance(objects[0], dict):
279309
return
310+
nested_groups: dict[str, tuple[list[dict], list[dict]]] = {}
280311
for obj in objects:
281312
for key, value in obj.items():
282313
if (
283314
isinstance(value, list)
284315
and value
285316
and all(isinstance(v, dict) for v in value)
286317
):
287-
self._flatten_nested(table_name, key, value, obj)
318+
children, parents = nested_groups.setdefault(key, ([], []))
319+
children.extend(value)
320+
parents.extend([obj] * len(value))
321+
for key, (children, parents) in nested_groups.items():
322+
self._flatten_nested(table_name, key, children, parents)

‎tests/test_edge_cases.py‎

Lines changed: 38 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -108,3 +108,41 @@ def test_convert_objects_list_vs_dict_root(self):
108108
result = converter.convert(json.dumps([{"name": "test"}]))
109109
assert "INSERT INTO" in result
110110
assert "'test'" in result
111+
112+
113+
def test_flatten_multiple_parent_rows_single_child_table():
114+
"""Multiple parent rows with nested arrays yield ONE child table with all rows."""
115+
import json as _json
116+
117+
from json2sql.converter import JSONToSQLConverter
118+
119+
data = [
120+
{"id": 1, "name": "a", "tags": [{"label": "x", "score": 1}]},
121+
{
122+
"id": 2,
123+
"name": "b",
124+
"tags": [{"label": "y", "score": 2}, {"label": "z", "score": 3}],
125+
},
126+
]
127+
text = _json.dumps(data)
128+
out = JSONToSQLConverter(flatten=True).convert(text, "users")
129+
assert out.count('CREATE TABLE "users_tags"') == 1
130+
assert "'z', 3" in out and "'y', 2" in out and "'x', 1" in out
131+
132+
schema = JSONToSQLConverter(flatten=True).generate_schema(text, "users")
133+
assert schema.count('CREATE TABLE "users_tags"') == 1
134+
135+
136+
def test_flatten_child_rows_keep_own_parent_fk():
137+
"""Each child row links to its own parent via the FK column."""
138+
import json as _json
139+
140+
from json2sql.converter import JSONToSQLConverter
141+
142+
data = [
143+
{"id": 10, "items": [{"sku": "a1"}]},
144+
{"id": 20, "items": [{"sku": "b1"}]},
145+
]
146+
out = JSONToSQLConverter(flatten=True).convert(_json.dumps(data), "orders")
147+
assert "(10, 'a1')" in out
148+
assert "(20, 'b1')" in out

‎tests/test_type_inference.py‎

Lines changed: 69 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -131,3 +131,72 @@ def test_convert_never_emits_empty_column_list():
131131
)
132132
assert "();" not in out
133133
assert "INSERT INTO" in out
134+
135+
136+
class TestFlattenFKDetection:
137+
"""Tests for correct FK column detection in flatten mode.
138+
139+
The FK column in a child table must match the parent table's primary key
140+
column name and type. Previously the code preferred "name" over explicit
141+
ID fields like "user_id" or "users_id", causing a type mismatch.
142+
"""
143+
144+
@pytest.mark.parametrize("dialect", [Dialect.POSTGRES, Dialect.MYSQL, Dialect.SQLITE])
145+
def test_flatten_prefers_id_over_name(self, dialect):
146+
"""When parent has both 'id' and 'name', 'id' should be used for FK."""
147+
conv = JSONToSQLConverter(dialect=dialect, flatten=True)
148+
data = json.dumps([
149+
{"id": 1, "name": "Alice", "tags": [{"label": "x"}]},
150+
{"id": 2, "name": "Bob", "tags": [{"label": "y"}]},
151+
])
152+
out = conv.convert(data, table_name="users")
153+
# FK column should be users_id (from parent's id), not users_name
154+
assert '"users_id"' in out or '`users_id`' in out
155+
assert '"users_name"' not in out and '`users_name`' not in out
156+
# Parent table should have id column
157+
assert '"id"' in out or '`id`' in out
158+
159+
@pytest.mark.parametrize("dialect", [Dialect.POSTGRES, Dialect.MYSQL, Dialect.SQLITE])
160+
def test_flatten_prefers_table_specific_id(self, dialect):
161+
"""When parent has '{table}_id' (e.g., users_id), it should be used."""
162+
conv = JSONToSQLConverter(dialect=dialect, flatten=True)
163+
data = json.dumps([
164+
{"users_id": 10, "name": "Alice", "tags": [{"label": "x"}]},
165+
{"users_id": 20, "name": "Bob", "tags": [{"label": "y"}]},
166+
])
167+
out = conv.convert(data, table_name="users")
168+
# FK column should be users_users_id (from parent's users_id)
169+
assert '"users_users_id"' in out or '`users_users_id`' in out
170+
assert '"users_name"' not in out and '`users_name`' not in out
171+
172+
@pytest.mark.parametrize("dialect", [Dialect.POSTGRES, Dialect.MYSQL, Dialect.SQLITE])
173+
def test_flatten_prefers_any_id_suffix(self, dialect):
174+
"""When parent has a singular '*_id' (e.g., user_id), it should be used over 'name'."""
175+
conv = JSONToSQLConverter(dialect=dialect, flatten=True)
176+
data = json.dumps([
177+
{"user_id": 100, "name": "Alice", "tags": [{"label": "x"}]},
178+
{"user_id": 200, "name": "Bob", "tags": [{"label": "y"}]},
179+
])
180+
out = conv.convert(data, table_name="users")
181+
# FK column should be users_user_id (from parent's user_id)
182+
assert '"users_user_id"' in out or '`users_user_id`' in out
183+
assert '"users_name"' not in out and '`users_name`' not in out
184+
# FK type should be numeric (matching parent's user_id type)
185+
if dialect == Dialect.MYSQL:
186+
assert "INT" in out
187+
else:
188+
assert "INTEGER" in out
189+
190+
@pytest.mark.parametrize("dialect", [Dialect.POSTGRES, Dialect.MYSQL, Dialect.SQLITE])
191+
def test_flatten_fallback_to_name_when_no_id(self, dialect):
192+
"""When parent has no ID-like field, 'name' is used as fallback."""
193+
conv = JSONToSQLConverter(dialect=dialect, flatten=True)
194+
data = json.dumps([
195+
{"name": "Alice", "tags": [{"label": "x"}]},
196+
{"name": "Bob", "tags": [{"label": "y"}]},
197+
])
198+
out = conv.convert(data, table_name="users")
199+
# FK column should be users_name (fallback)
200+
assert '"users_name"' in out or '`users_name`' in out
201+
# Parent table should have name column
202+
assert '"name"' in out or '`name`' in out

0 commit comments

Comments
 (0)