Skip to content

Commit 04bd06a

Browse files
committed
test(qa): add comprehensive test suites for DAG, UI, generator, and in-memory execution
1 parent 7d76ad6 commit 04bd06a

5 files changed

Lines changed: 203 additions & 1 deletion

File tree

‎test123.txt‎

Lines changed: 0 additions & 1 deletion
This file was deleted.

‎tests/test_airflow_dag.py‎

Lines changed: 61 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,61 @@
1+
"""
2+
tests/test_airflow_dag.py
3+
--------------------------
4+
Unit and integration tests for Apache Airflow DAG in dags/data_pipeline_dag.py.
5+
"""
6+
7+
import sys
8+
from pathlib import Path
9+
10+
import pytest
11+
12+
# Skip module if apache-airflow is not installed in the current environment
13+
pytest.importorskip("airflow")
14+
15+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
16+
17+
from dags.data_pipeline_dag import _get_safe_run_id, dag, default_args
18+
19+
20+
class TestAirflowDagDefinition:
21+
"""Test suite for Airflow DAG structure and helper functions."""
22+
23+
def test_dag_structure_and_metadata(self):
24+
assert dag is not None
25+
assert dag.dag_id == "dataprep_pipeline"
26+
assert dag.schedule_interval == "@daily"
27+
assert len(dag.tasks) == 5
28+
29+
def test_dag_task_dependencies(self):
30+
task_dict = {t.task_id: t for t in dag.tasks}
31+
assert set(task_dict.keys()) == {
32+
"ingest_data",
33+
"validate_data",
34+
"clean_data",
35+
"transform_data",
36+
"load_data",
37+
}
38+
39+
# Check linear dependency flow: ingest -> validate -> clean -> transform -> load
40+
ingest_task = task_dict["ingest_data"]
41+
validate_task = task_dict["validate_data"]
42+
clean_task = task_dict["clean_data"]
43+
transform_task = task_dict["transform_data"]
44+
load_task = task_dict["load_data"]
45+
46+
assert validate_task in ingest_task.downstream_list
47+
assert clean_task in validate_task.downstream_list
48+
assert transform_task in clean_task.downstream_list
49+
assert load_task in transform_task.downstream_list
50+
51+
def test_get_safe_run_id_sanitizes_illegal_chars(self):
52+
context_with_colon = {"run_id": "scheduled__2026-08-01T12:00:00+00:00"}
53+
safe_id = _get_safe_run_id(context_with_colon)
54+
assert ":" not in safe_id
55+
assert "+" not in safe_id
56+
assert safe_id == "scheduled__2026-08-01T12_00_00_00_00"
57+
58+
def test_get_safe_run_id_default_fallback(self):
59+
context_empty = {}
60+
safe_id = _get_safe_run_id(context_empty)
61+
assert safe_id == "default_run"

‎tests/test_app_ui.py‎

Lines changed: 52 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,52 @@
1+
"""
2+
tests/test_app_ui.py
3+
---------------------
4+
Unit tests for Streamlit App UI helper functions in app.py.
5+
"""
6+
7+
import sys
8+
from pathlib import Path
9+
10+
import pandas as pd
11+
import pytest
12+
13+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
14+
15+
from app import run_pipeline_cached, to_csv_bytes
16+
17+
18+
class TestAppUiHelpers:
19+
"""Test suite for app.py helper functions."""
20+
21+
def test_to_csv_bytes_returns_utf8_bytes(self):
22+
df = pd.DataFrame({"id": [1, 2], "nombre": ["Ana", "Carlos"]})
23+
result_bytes = to_csv_bytes(df)
24+
25+
assert isinstance(result_bytes, bytes)
26+
csv_str = result_bytes.decode("utf-8")
27+
assert "id,nombre" in csv_str
28+
assert "Ana" in csv_str
29+
assert "Carlos" in csv_str
30+
31+
def test_run_pipeline_cached_execution(self):
32+
df_raw = pd.DataFrame(
33+
{
34+
"fecha": ["2024-01-01", "2024-01-02"],
35+
"precio": [100.0, 200.0],
36+
"cantidad": [1, 2],
37+
}
38+
)
39+
40+
before_report, df_clean, df_final, after_report = run_pipeline_cached(
41+
df_raw=df_raw,
42+
normalize=False,
43+
null_thresh=20.0,
44+
dup_thresh=5.0,
45+
)
46+
47+
assert before_report is not None
48+
assert after_report is not None
49+
assert df_clean is not None
50+
assert df_final is not None
51+
assert "total" in df_final.columns
52+
assert len(df_final) == 2

‎tests/test_dataset_generator.py‎

Lines changed: 42 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,42 @@
1+
"""
2+
tests/test_dataset_generator.py
3+
--------------------------------
4+
Unit tests for synthetic dataset generator in generate_dataset.py.
5+
"""
6+
7+
import sys
8+
from pathlib import Path
9+
10+
import pandas as pd
11+
import pytest
12+
13+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
14+
15+
from generate_dataset import generate_dataset, introduce_errors, make_clean_row, random_date
16+
17+
18+
class TestDatasetGenerator:
19+
"""Test suite for synthetic dataset generation utility."""
20+
21+
def test_random_date_returns_valid_iso_string(self):
22+
d_str = random_date()
23+
assert isinstance(d_str, str)
24+
parsed_dt = pd.to_datetime(d_str)
25+
assert parsed_dt.year in [2022, 2023, 2024]
26+
27+
def test_make_clean_row_structure(self):
28+
row = make_clean_row(i=42)
29+
assert isinstance(row, dict)
30+
assert row["id_venta"] == 42
31+
assert "precio" in row
32+
assert "cantidad" in row
33+
assert "fecha" in row
34+
35+
def test_generate_dataset_introduces_expected_dirty_data(self):
36+
df = generate_dataset(n=200)
37+
assert isinstance(df, pd.DataFrame)
38+
assert len(df) > 200 # Should include duplicate rows added by introduce_errors
39+
assert "codigo_venta" in df.columns
40+
41+
# Verify intentional dirty characteristics
42+
assert df.isnull().sum().sum() > 0

‎tests/test_pipeline_inmemory.py‎

Lines changed: 48 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,48 @@
1+
"""
2+
tests/test_pipeline_inmemory.py
3+
--------------------------------
4+
Unit tests for execute_pipeline in-memory df_raw execution path.
5+
"""
6+
7+
import sys
8+
from pathlib import Path
9+
10+
import pandas as pd
11+
import pytest
12+
13+
sys.path.insert(0, str(Path(__file__).resolve().parent.parent))
14+
15+
from src.pipeline import execute_pipeline
16+
17+
18+
class TestPipelineInMemoryExecution:
19+
"""Test suite for execute_pipeline in-memory execution mode."""
20+
21+
def test_execute_pipeline_with_df_raw(self, tmp_path):
22+
out_csv = tmp_path / "inmemory_out.csv"
23+
out_html = tmp_path / "inmemory_report.html"
24+
25+
df_raw = pd.DataFrame(
26+
{
27+
"id_venta": [1, 2, 2],
28+
"fecha": ["2024-01-01", "2024-01-02", "2024-01-02"],
29+
"precio": [10.0, 20.0, 20.0],
30+
"cantidad": [2, 1, 1],
31+
}
32+
)
33+
34+
result = execute_pipeline(
35+
output_path=out_csv,
36+
report_path=out_html,
37+
df_raw=df_raw,
38+
normalize=True,
39+
)
40+
41+
assert result.success is True
42+
assert result.df_raw is not None
43+
assert result.df_final is not None
44+
assert len(result.df_final) == 2 # Duplicate removed
45+
assert out_csv.exists()
46+
assert out_html.exists()
47+
assert "total" in result.df_final.columns
48+
assert "precio_norm" in result.df_final.columns

0 commit comments

Comments
 (0)