Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions .github/workflows/cowork-auto-pr.yml
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,14 @@ jobs:
ensure-pr:
runs-on: ubuntu-latest
steps:
# gh pr create requires a local git checkout to diff head against base;
# without this step every run failed with "not a git repository" and no
# PR was ever opened (fleet-wide defect: 11/11 seeded copies lacked it).
- name: Check out the pushed branch
uses: actions/checkout@v4
with:
ref: ${{ github.ref_name }}
fetch-depth: 0
- name: Open PR for this branch if none exists
env:
GH_TOKEN: ${{ github.token }}
Expand Down
38 changes: 20 additions & 18 deletions tests/test_converters.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,18 @@
supported_formats,
)

try:
import pyarrow.parquet # noqa: F401

_HAS_PYARROW = True
except (ImportError, Exception):
_HAS_PYARROW = False

requires_pyarrow = pytest.mark.skipif(
not _HAS_PYARROW,
reason="Parquet support requires pyarrow: pip install 'datamorph[parquet]'",
)

# ── Fixtures ──────────────────────────────────────────────────────────


Expand Down Expand Up @@ -235,6 +247,7 @@ def test_yaml_to_csv(self, sample_yaml, tmp_path):
# ── Parquet ───────────────────────────────────────────────────────────


@requires_pyarrow
class TestParquetConversion:
def test_csv_to_parquet(self, sample_csv, tmp_path):
output = tmp_path / "output.parquet"
Expand Down Expand Up @@ -315,9 +328,7 @@ def test_avro_write_empty(self, tmp_path):
def test_avro_nullable_first_row(self, tmp_path):
"""Avro should handle nullable fields even when the first row has nulls."""
path = tmp_path / "data.csv"
path.write_text(
"name,age,email\nAlice,,alice@test.com\nBob,30,\nCharlie,25,charlie@test.com\n"
)
path.write_text("name,age,email\nAlice,,alice@test.com\nBob,30,\nCharlie,25,charlie@test.com\n")
avro_file = tmp_path / "out.avro"
result = convert(path, avro_file)
assert not result.errors
Expand Down Expand Up @@ -464,15 +475,14 @@ def test_convert_with_format_override(self, runner, sample_csv, tmp_path):
assert result.exit_code == 0
assert "Converted" in result.output

@requires_pyarrow
def test_convert_to_parquet(self, runner, sample_csv, tmp_path):
output = tmp_path / "out.parquet"
result = runner.invoke(cli, ["convert", str(sample_csv), str(output)])
assert result.exit_code == 0

def test_convert_nonexistent_input(self, runner, tmp_path):
result = runner.invoke(
cli, ["convert", "/nonexistent/file.csv", str(tmp_path / "out.json")]
)
result = runner.invoke(cli, ["convert", "/nonexistent/file.csv", str(tmp_path / "out.json")])
assert result.exit_code != 0

def test_formats_command(self, runner):
Expand Down Expand Up @@ -559,6 +569,7 @@ def test_batch_recursive(self, runner, tmp_path):
# Verify nested directory structure is preserved
assert any("nested" in str(f) for f in out_files)

@requires_pyarrow
def test_batch_to_parquet(self, runner, sample_csv, tmp_path):
"""Batch convert CSV files to Parquet via CLI."""
output_dir = tmp_path / "pq_out"
Expand Down Expand Up @@ -710,12 +721,7 @@ def test_large_json_array(self, tmp_path):
class TestJsonlConversion:
def test_jsonl_to_json(self, tmp_path):
path = tmp_path / "data.jsonl"
path.write_text(
json.dumps({"name": "Alice", "age": 30})
+ "\n"
+ json.dumps({"name": "Bob", "age": 25})
+ "\n"
)
path.write_text(json.dumps({"name": "Alice", "age": 30}) + "\n" + json.dumps({"name": "Bob", "age": 25}) + "\n")
output = tmp_path / "out.json"
result = convert(path, output)
assert not result.errors
Expand All @@ -726,12 +732,7 @@ def test_jsonl_to_json(self, tmp_path):

def test_jsonl_to_csv(self, tmp_path):
path = tmp_path / "data.jsonl"
path.write_text(
json.dumps({"name": "Alice", "age": 30})
+ "\n"
+ json.dumps({"name": "Bob", "age": 25})
+ "\n"
)
path.write_text(json.dumps({"name": "Alice", "age": 30}) + "\n" + json.dumps({"name": "Bob", "age": 25}) + "\n")
output = tmp_path / "out.csv"
result = convert(path, output)
assert not result.errors
Expand Down Expand Up @@ -768,6 +769,7 @@ def test_jsonl_empty(self, tmp_path):
class TestWriterKwargLeak:
"""Regression tests for delimiter kwarg leaking to non-CSV writers."""

@requires_pyarrow
def test_csv_delimiter_to_parquet(self, tmp_path):
"""csv_delimiter should not crash when converting to Parquet."""
csv_path = tmp_path / "data.csv"
Expand Down
Loading