Skip to content
Merged
Show file tree
Hide file tree
Changes from 1 commit
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Prev Previous commit
cowork-bot: batch convert no longer silently drops format-mismatched …
…files - convert_batch returns BatchConversionResult with skipped[] (file + detected format), CLI reports each skip explicitly instead of a green 'complete' while dropping data; +1 regression test (149 pass), ruff clean
  • Loading branch information
cowork-bot committed Aug 26, 2026
commit 9fbb32461ba6f9c22575c319a6196c2feafa37ea
16 changes: 15 additions & 1 deletion src/datamorph/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,7 +118,7 @@ def batch_cmd(
if csv_delimiter != ",":
writer_kwargs["delimiter"] = csv_delimiter

results = convert_batch(
batch = convert_batch(
input_dir,
output_dir,
from_format,
Expand All @@ -127,11 +127,25 @@ def batch_cmd(
recursive=recursive,
**writer_kwargs,
)
results = batch.results

success = [r for r in results if not r.errors]
failed = [r for r in results if r.errors]

console.print("\n[bold]Batch Conversion Complete[/bold]")

# Files that matched the pattern but were skipped due to format mismatch
# are surfaced explicitly instead of vanishing silently.
if batch.skipped:
console.print(
f" Skipped (format mismatch): {len(batch.skipped)}"
" - did not match --from format"
)
for item in batch.skipped:
err_console.print(
f" [yellow]SKIPPED[/yellow] {item['file']} "
f"(detected: {item['detected_format']}, expected: {from_format})"
)
console.print(f" Files: {len(success)} converted, {len(failed)} failed")

if failed:
Expand Down
32 changes: 27 additions & 5 deletions src/datamorph/converters.py
Original file line number Diff line number Diff line change
Expand Up @@ -600,6 +600,21 @@ def _counting(stream: RowStream) -> RowStream:
return result


@dataclass
class BatchConversionResult:
"""Outcome of a batch conversion, including files that were skipped.

``skipped`` records every file that matched ``pattern`` but was NOT
converted because its detected format differed from ``input_format``
(or the format could not be detected at all). Surfacing these prevents
the classic silent failure where a directory conversion quietly drops
mislabeled or foreign-format files while reporting success.
"""

results: list[ConversionResult] = field(default_factory=list)
skipped: list[dict[str, str]] = field(default_factory=list)


def convert_batch(
input_dir: str | Path,
output_dir: str | Path,
Expand All @@ -608,19 +623,26 @@ def convert_batch(
pattern: str = "*",
recursive: bool = False,
**writer_kwargs: Any,
) -> list[ConversionResult]:
) -> BatchConversionResult:

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

P2 Badge Preserve convert_batch's list contract

convert_batch is re-exported as a public API from datamorph.__init__, but this changes its return value from list[ConversionResult] to a non-iterable dataclass. Existing callers that iterate, index, compare to [], or call len(convert_batch(...)) now fail even though the CLI was updated. Keep the result list-compatible (for example by implementing the sequence protocol) or introduce the richer result through a new API.

Useful? React with 👍 / 👎.

"""Convert all matching files in a directory."""
input_dir = Path(input_dir)
output_dir = Path(output_dir)
output_dir.mkdir(parents=True, exist_ok=True)

glob_pattern = f"**/{pattern}" if recursive else pattern
results: list[ConversionResult] = []
batch = BatchConversionResult()

for input_path in sorted(input_dir.glob(glob_pattern)):
if input_path.is_dir():
continue
if detect_format(str(input_path)) != input_format:
detected = detect_format(str(input_path))
if detected != input_format:
batch.skipped.append(
{
"file": str(input_path),
"detected_format": detected or "unknown",
}
)
continue

# Preserve relative path structure
Expand All @@ -637,9 +659,9 @@ def convert_batch(
output_format,
**writer_kwargs,
)
results.append(result)
batch.results.append(result)

return results
return batch


def _format_to_extension(fmt: str) -> str:
Expand Down
31 changes: 27 additions & 4 deletions tests/test_converters.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,6 +3,7 @@
from __future__ import annotations

import json
from pathlib import Path

import pytest
import yaml
Expand Down Expand Up @@ -819,9 +820,10 @@ def test_batch_single_file(self, sample_csv, tmp_path):
"json",
pattern="test.csv",
)
assert len(results) >= 1
assert not results[0].errors
assert results[0].rows_written == 3
assert len(results.results) >= 1
assert results.skipped == []
assert not results.results[0].errors
assert results.results[0].rows_written == 3
assert (output_dir / "test.json").exists()

def test_batch_no_matches(self, tmp_path):
Expand All @@ -835,7 +837,27 @@ def test_batch_no_matches(self, tmp_path):
"json",
pattern="*.csv",
)
assert results == []
assert results.results == []
assert results.skipped == []


class TestBatchSkippedReporting:
def test_format_mismatch_is_reported_not_silent(self, tmp_path):
"""A file matching the pattern but with a different detected format is
recorded in ``skipped`` instead of vanishing without a trace."""
input_dir = tmp_path / "in"
input_dir.mkdir()
(input_dir / "data.csv").write_text("a,b\n1,2\n", encoding="utf-8")
# Mislabeled: .csv extension but JSONL content -> detect_format sees jsonl? No:
# detection is extension-based, so use a foreign extension instead.
(input_dir / "notes.txt").write_text("hello", encoding="utf-8")
output_dir = tmp_path / "out"
results = convert_batch(str(input_dir), str(output_dir), "csv", "json")
names = [Path(item["file"]).name for item in results.skipped]
assert names == ["notes.txt"]
assert results.skipped[0]["detected_format"] in ("txt", "unknown")
# The real csv was still converted.
assert [r.rows_written for r in results.results] == [1]


# ── Type inference ────────────────────────────────────────────────────
Expand Down Expand Up @@ -993,3 +1015,4 @@ def test_scalar_bool_root_round_trip(self, tmp_path):
assert result.rows_read == 1
assert result.rows_written == 1
assert json.loads(out.read_text(encoding="utf-8")) == [{"data": True}]

Loading