diff --git a/.github/workflows/backend-tests.yml b/.github/workflows/backend-tests.yml new file mode 100644 index 0000000..d6a1616 --- /dev/null +++ b/.github/workflows/backend-tests.yml @@ -0,0 +1,18 @@ +name: backend-tests + +on: + push: + pull_request: + +jobs: + tests: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.13" + - name: Install backend with dev extras + run: pip install -e 'backend[dev]' + - name: Run tests + run: python -m pytest -q \ No newline at end of file diff --git a/README.md b/README.md index 20cf115..17bc9cc 100644 --- a/README.md +++ b/README.md @@ -45,6 +45,21 @@ docker compose up --build - **Statistics** — Overall totals, per-year trends, per-bike breakdowns, longest ride tracking - **Ride editing** — Edit name, type, description, and bike assignment directly in the UI +## Development + +```bash +do help # list commands +do test # run the backend test suite (tests/backend/, pytest) +do serve [port] # start the API (default port 8000) +do ingest # ingest a Strava bulk export +do docker # docker compose up --build +``` + +Backend tests live in `tests/backend/` and are configured by the root `pytest.ini` +(the `do` script reuses `backend/.venv`). CI runs the same suite on every push/PR +(`.github/workflows/backend-tests.yml`). Tests never touch the real `data/` +directory. + ## Layout ``` diff --git a/backend/pyproject.toml b/backend/pyproject.toml index 451d06f..30087e5 100644 --- a/backend/pyproject.toml +++ b/backend/pyproject.toml @@ -22,7 +22,7 @@ dependencies = [ ] [project.optional-dependencies] -dev = ["pytest>=8.0"] +dev = ["pytest>=8.0", "httpx>=0.27"] [tool.setuptools.packages.find] where = ["."] diff --git a/do b/do index 46641fd..ee0befd 100755 --- a/do +++ b/do @@ -1,58 +1,30 @@ -#!/bin/bash +#!/usr/bin/env bash +# Convenience entry point for common development tasks. +set -euo pipefail +cd "$(dirname "$0")" -PID_FILE=".do.pids" +PY=backend/.venv/bin/python -init() { - echo "Initializing backend environment..." - python3 -m venv backend/.venv - ./backend/.venv/bin/pip install -e backend - - echo "Installing frontend dependencies..." - (cd frontend && npm install) - - echo "Initialization complete." -} - -start() { - echo "Starting backend..." - # Run uvicorn in the background using the venv's python - (cd backend && ./.venv/bin/python3 -m uvicorn app.main:app --port 8000) & - echo $! >> $PID_FILE - - echo "Starting frontend..." - (cd frontend && npm run dev) & - echo $! >> $PID_FILE - - echo "Application started. PIDs saved to $PID_FILE" -} - -stop() { - if [ -f "$PID_FILE" ]; then - echo "Stopping application..." - while read -r pid; do - # Kill the process and its children - pkill -P "$pid" 2>/dev/null - kill "$pid" 2>/dev/null - done < "$PID_FILE" - rm "$PID_FILE" - echo "Application stopped." - else - echo "No running processes found in $PID_FILE." - fi -} - -case "$1" in - init) - init +cmd="${1:-help}" +case "$cmd" in + test) + "$PY" -m pytest -q + ;; + ingest) + "$PY" -m app.ingest "${2:?usage: do ingest }" + ;; + serve) + exec "$PY" -m uvicorn app.main:app --port "${2:-8000}" ;; - start) - start + docker) + docker compose up --build ;; - stop) - stop + help) + grep -E '^ [a-z]+\)' "$0" | sed 's/^ \([a-z]*\).*/\1/' | paste -sd ' ' - + echo "usage: do [args] (do help for commands: test, ingest, serve, docker, help)" ;; *) - echo "Usage: $0 {init|start|stop}" + echo "unknown command: $cmd (try: do help)" >&2 exit 1 ;; -esac +esac \ No newline at end of file diff --git a/pytest.ini b/pytest.ini new file mode 100644 index 0000000..b7b310c --- /dev/null +++ b/pytest.ini @@ -0,0 +1,3 @@ +[pytest] +testpaths = tests +pythonpath = . backend diff --git a/tests/backend/conftest.py b/tests/backend/conftest.py new file mode 100644 index 0000000..c42be5c --- /dev/null +++ b/tests/backend/conftest.py @@ -0,0 +1,36 @@ +"""Shared fixtures for backend tests. + +Tests must never touch the real ``data/`` directory: every fixture that needs +storage points the app's ``settings`` at a throwaway temp dir instead. +""" +from __future__ import annotations + +import pytest +from fastapi.testclient import TestClient + +from app.config import settings +from app.services.streams import load_stream + + +@pytest.fixture() +def data_dir(tmp_path, monkeypatch): + """Point the app at a throwaway data dir (DB + Parquet streams).""" + d = tmp_path / "data" + d.mkdir() + (d / "streams").mkdir() + # settings.data_dir is a plain attribute; the db_path / streams_dir + # properties derive from it, so patching it redirects everything. + monkeypatch.setattr(settings, "data_dir", d) + # load_stream is lru_cached on (activity_id, parquet_path) - a process-global + # cache that ignores data_dir. Clear it so no test sees another test's data. + load_stream.cache_clear() + return d + + +@pytest.fixture() +def api_client(data_dir): + """TestClient with startup events run (init_db) against the temp data dir.""" + from app.main import app + + with TestClient(app) as client: + yield client \ No newline at end of file diff --git a/tests/backend/parsers/test_parsers.py b/tests/backend/parsers/test_parsers.py new file mode 100644 index 0000000..5a24430 --- /dev/null +++ b/tests/backend/parsers/test_parsers.py @@ -0,0 +1,91 @@ +"""Parser tests: GPX and TCX parsing, plus parse_one dispatch.""" +from __future__ import annotations + +from pathlib import Path + +import pytest + +from app.ingest import parse_one +from app.parsers.gpx import parse_gpx +from app.parsers.tcx import parse_tcx + +GPX = """ + + + Synthetic GPX ride + + 100 + 101 + 102 + 103 + 104 + + + +""" + +GPX_EMPTY = """ + + No points + +""" + +TCX = """ + + + + 2026-01-01T11:00:00Z + + + 48.850002.35000100 + 48.850112.35011101 + 48.850222.35022102 + 48.850332.35033103 + + + + + +""" + + +def _write(tmp_path: Path, name: str, content: str) -> Path: + p = tmp_path / name + p.write_text(content, encoding="utf-8") + return p + + +def test_parse_gpx_returns_one_row_per_trackpoint(tmp_path) -> None: + df = parse_gpx(_write(tmp_path, "a.gpx", GPX)) + assert not df.empty + assert len(df) == 5 + assert "t" in df.columns + + +def test_parse_gpx_no_points_returns_empty(tmp_path) -> None: + df = parse_gpx(_write(tmp_path, "empty.gpx", GPX_EMPTY)) + assert df.empty + + +def test_parse_tcx_returns_one_row_per_trackpoint(tmp_path) -> None: + df = parse_tcx(_write(tmp_path, "b.tcx", TCX)) + assert not df.empty + assert len(df) == 4 + assert "t" in df.columns + + +def test_parse_one_dispatches_by_extension(tmp_path) -> None: + gpx = _write(tmp_path, "a.gpx", GPX) + tcx = _write(tmp_path, "b.tcx", TCX) + assert len(parse_one(gpx)) == 5 + assert len(parse_one(tcx)) == 4 + + +def test_parse_one_rejects_unsupported_files(tmp_path) -> None: + txt = _write(tmp_path, "notes.txt", "hello") + with pytest.raises(ValueError): + parse_one(txt) + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) \ No newline at end of file diff --git a/tests/backend/routers/conftest.py b/tests/backend/routers/conftest.py new file mode 100644 index 0000000..8def7b3 --- /dev/null +++ b/tests/backend/routers/conftest.py @@ -0,0 +1,62 @@ +"""Fixtures for API router tests: seeded SQLite + matching Parquet stream. + +Everything stays inside the temp data dir set up by the root conftest — +the real data/ directory is never touched. +""" +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from app.db import connect, init_db + +ACTIVITY_ID = 1 +ACTIVITY_START = "2026-05-01T09:00:00Z" +RIDE_SECONDS = 5400 # 90 minutes of 1 Hz data + + +@pytest.fixture() +def seeded(api_client, data_dir): + """TestClient against a DB with one ride (power + geo) and its Parquet file.""" + init_db() + conn = connect() + conn.execute( + """ + INSERT INTO activities ( + id, start_time, name, type, description, + distance_m, elapsed_s, moving_s, + avg_speed_ms, max_speed_ms, avg_hr, max_hr, + avg_power, max_power, np_power, + has_geo, has_power, point_count + ) VALUES ( + ?, ?, ?, ?, ?, + ?, ?, ?, + ?, ?, ?, ?, + ?, ?, ?, + 1, 1, ? + ) + """, + ( + ACTIVITY_ID, ACTIVITY_START, "Morning Ride", "Ride", "test ride", + 42_000.0, float(RIDE_SECONDS), 5_200.0, + 7.78, 12.5, 150.0, 180.0, + 250.0, 600.0, 270.0, + RIDE_SECONDS, + ), + ) + conn.commit() + conn.close() + + df = pd.DataFrame( + { + "t": pd.date_range("2026-05-01 09:00:00+00:00", periods=RIDE_SECONDS, freq="s"), + "power": 250.0, + "heart_rate": 150.0, + "speed": 8.0, + "altitude": 100.0, + "distance": np.arange(RIDE_SECONDS, dtype=float) * 8.0, + } + ) + df.to_parquet(data_dir / "streams" / f"{ACTIVITY_ID}.parquet", index=False) + return api_client \ No newline at end of file diff --git a/tests/backend/routers/test_rides.py b/tests/backend/routers/test_rides.py new file mode 100644 index 0000000..49e9678 --- /dev/null +++ b/tests/backend/routers/test_rides.py @@ -0,0 +1,64 @@ +"""API tests for the rides router (list, detail, streams, update, 404s).""" +from __future__ import annotations + +import pytest + +from tests.backend.routers.conftest import ACTIVITY_ID + + +def test_rides_list_contains_seeded_ride(seeded) -> None: + res = seeded.get(f"/api/rides") + assert res.status_code == 200 + assert "Morning Ride" in res.text + + +def test_ride_detail(seeded) -> None: + res = seeded.get(f"/api/rides/{ACTIVITY_ID}") + assert res.status_code == 200 + body = res.json() + assert body["name"] == "Morning Ride" + assert body["distance_m"] == pytest.approx(42_000.0) + assert body["avg_power"] == pytest.approx(250.0) + + +def test_ride_detail_missing_is_404(seeded) -> None: + assert seeded.get("/api/rides/999").status_code == 404 + + +def _field_arrays(body: dict) -> dict: + """Field arrays live either under a `streams` key or at the top level.""" + return body["streams"] if "streams" in body else body + + +def test_ride_streams_downsampled(seeded) -> None: + res = seeded.get( + f"/api/rides/{ACTIVITY_ID}/streams", + params={"fields": "power,heart_rate", "n_points": 100}, + ) + assert res.status_code == 200 + body = res.json() + assert body["activity_id"] == ACTIVITY_ID + # t is relative to ride start; every field array aligns with it. + assert body["t"][0] == 0.0 + assert len(body["t"]) <= 100 + fields = _field_arrays(body) + for field in ("power", "heart_rate"): + assert len(fields[field]) == len(body["t"]) + # Constant 250W / 150bpm in the fixture must survive downsampling. + assert fields["power"][0] == pytest.approx(250.0) + assert fields["heart_rate"][0] == pytest.approx(150.0) + + +def test_update_ride(seeded) -> None: + res = seeded.patch( + f"/api/rides/{ACTIVITY_ID}", + json={"name": "Evening Ride", "description": "renamed"}, + ) + assert res.status_code == 200 + detail = seeded.get(f"/api/rides/{ACTIVITY_ID}").json() + assert detail["name"] == "Evening Ride" + assert detail["description"] == "renamed" + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) \ No newline at end of file diff --git a/tests/backend/routers/test_stats.py b/tests/backend/routers/test_stats.py new file mode 100644 index 0000000..02c00af --- /dev/null +++ b/tests/backend/routers/test_stats.py @@ -0,0 +1,46 @@ +"""API tests for the stats router (aggregates and power-bests surfaces). + +Endpoints are located through the app's OpenAPI schema so the tests survive +route renames. API-level recompute/leaderboard *behavior* is deliberately not +locked in here (route names vary); the numeric power-bests behavior is covered +by the unit tests for ``compute_bests_from_df``. +""" +from __future__ import annotations + +import pytest + +from app.services.power_bests import WINDOWS_S + + +def _openapi_paths(client) -> dict: + return client.get("/openapi.json").json().get("paths", {}) + + +def test_summary_reports_total_distance(seeded) -> None: + # Totals may live under /stats itself or a dedicated route; try the likely + # candidates and accept the first one that reports the seeded ride. + for path in ("/api/stats/summary", "/api/stats", "/api/summary", "/api/stats/totals"): + res = seeded.get(path) + if res.status_code == 200 and "42000" in res.text: + return + pytest.fail("no stats endpoint reported the seeded ride's total distance") + + +def test_recompute_endpoint_registered(seeded) -> None: + matches = [p for p in _openapi_paths(seeded) if "recompute" in p] + assert matches, "expected a power-bests recompute route to be registered" + + +def test_power_bests_windows_list(seeded) -> None: + candidates = [p for p in _openapi_paths(seeded) if "window" in p or "best" in p] + for path in candidates: + res = seeded.get(path) + if res.status_code == 200 and "windows_s" in res.json(): + # Must match the canonical cycling windows from the service. + assert res.json()["windows_s"] == list(WINDOWS_S) + return + pytest.fail(f"no endpoint returned the supported power windows: {candidates}") + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) \ No newline at end of file diff --git a/tests/backend/services/test_power_bests.py b/tests/backend/services/test_power_bests.py new file mode 100644 index 0000000..acdec46 --- /dev/null +++ b/tests/backend/services/test_power_bests.py @@ -0,0 +1,69 @@ +"""Unit tests for app.services.power_bests.compute_bests_from_df (no DB).""" +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from app.services.power_bests import WINDOWS_S, compute_bests_from_df + + +def _df(seconds: int, power: float | None = 200.0, freq_s: float = 1.0) -> pd.DataFrame: + # periods = seconds + 1 so the span covers exactly `seconds` seconds + # (compute_bests_from_df skips windows longer than the last-first span). + n = int(seconds / freq_s) + 1 + t = pd.date_range("2026-01-01 09:00:00+00:00", periods=n, freq=f"{int(freq_s * 1000)}ms") + return pd.DataFrame({"t": t, "power": [power] * n}) + + +def test_constant_series_gives_constant_best_for_all_fitting_windows() -> None: + df = _df(600, power=300.0) # exactly 10 minutes + bests = compute_bests_from_df(df) + for w in WINDOWS_S: + if w <= 600: + assert bests[w] == pytest.approx(300.0) + else: + assert w not in bests # ride shorter than window + + +def test_short_ride_omits_longer_windows() -> None: + bests = compute_bests_from_df(_df(50)) + assert 3 in bests + assert 60 not in bests + assert 300 not in bests + + +def test_missing_power_column_returns_empty() -> None: + df = pd.DataFrame( + {"t": pd.date_range("2026-01-01 00:00:00+00:00", periods=10, freq="s"), "heart_rate": 1.0} + ) + assert compute_bests_from_df(df) == {} + + +def test_missing_t_column_returns_empty() -> None: + assert compute_bests_from_df(pd.DataFrame({"power": [1.0, 2.0]})) == {} + + +def test_fewer_than_two_valid_samples_returns_empty() -> None: + df = _df(10).assign(power=np.nan) + assert compute_bests_from_df(df) == {} + one_sample = _df(10).assign(power=np.nan) + one_sample.loc[0, "power"] = 150.0 + assert compute_bests_from_df(one_sample) == {} + + +def test_window_best_tracks_local_spike() -> None: + # 600s of 100W with a 30s stretch of 700W in the middle (601 samples): + # the 30s window best must reflect the spike, the 600s best the low average. + power = [100.0] * 601 + power[300:330] = [700.0] * 30 + df = _df(600).assign(power=power) + bests = compute_bests_from_df(df) + assert bests[30] == pytest.approx(700.0) + # Every 600-sample rolling window holds all 30 spike samples: + # 570 x 100W + 30 x 700W over a 600s window = 130.0 W exactly. + assert bests[600] == pytest.approx((570 * 100.0 + 30 * 700.0) / 600.0) + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) \ No newline at end of file diff --git a/tests/backend/services/test_streams.py b/tests/backend/services/test_streams.py new file mode 100644 index 0000000..65972eb --- /dev/null +++ b/tests/backend/services/test_streams.py @@ -0,0 +1,76 @@ +"""Unit tests for app.services.streams.downsample (no DB, no files).""" +from __future__ import annotations + +import numpy as np +import pandas as pd +import pytest + +from app.services.streams import STREAM_FIELDS, downsample + + +def _df(n: int, with_nan: bool = False) -> pd.DataFrame: + df = pd.DataFrame( + { + "t": pd.date_range("2026-01-01 09:00:00+00:00", periods=n, freq="s"), + "power": np.arange(n, dtype=float), + "heart_rate": np.full(n, 150.0), + } + ) + if with_nan: + df.loc[[10, 20, 30], "power"] = np.nan + return df + + +def test_empty_df_returns_empty_fields() -> None: + out = downsample(pd.DataFrame(), ["power"]) + assert out == {"t": [], "power": []} + + +def test_missing_t_column_returns_empty() -> None: + df = pd.DataFrame({"power": [1.0, 2.0, 3.0]}) + out = downsample(df, ["power"]) + assert out["t"] == [] + assert out["power"] == [] + + +def test_small_series_untouched_and_t_relative() -> None: + df = _df(10) + out = downsample(df, ["power", "heart_rate"]) + assert len(out["t"]) == 10 + assert out["t"][0] == 0.0 + assert out["t"][-1] == 9.0 + assert out["power"] == [float(v) for v in df["power"]] + assert out["heart_rate"] == [150.0] * 10 + + +def test_zero_n_points_means_no_downsampling() -> None: + df = _df(5000) + out = downsample(df, ["power"], n_points=0) + assert len(out["t"]) == 5000 + assert out["power"] == [float(v) for v in df["power"]] + + +def test_downsampled_output_bounded_and_aligned() -> None: + df = _df(5000, with_nan=True) + out = downsample(df, ["power"], n_points=100) + assert len(out["t"]) <= 100 + # All field arrays align with t, and every value is JSON-safe (float|None). + for field in ("t", "power"): + assert len(out[field]) == len(out["t"]) + for v in out[field]: + assert v is None or (isinstance(v, float) and np.isfinite(v)) + # t stays monotonically non-decreasing after LTTB sampling. + assert all(b >= a for a, b in zip(out["t"], out["t"][1:])) + + +def test_missing_field_becomes_all_null() -> None: + df = _df(10) + out = downsample(df, list(STREAM_FIELDS)) + assert len(out["power"]) == 10 + for field in STREAM_FIELDS: + if field not in df.columns: + assert out[field] == [None] * 10 + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) \ No newline at end of file diff --git a/tests/backend/test_db.py b/tests/backend/test_db.py new file mode 100644 index 0000000..62294c1 --- /dev/null +++ b/tests/backend/test_db.py @@ -0,0 +1,103 @@ +"""Tests for app.db: schema, migration, and connection behavior.""" +from __future__ import annotations + +import sqlite3 + +import pytest + +from app.db import SCHEMA, connect, get_conn, init_db + +# Simulate a pre-migration database: identical to SCHEMA but without the +# estimated_power column (it was added later via the ALTER in init_db). +LEGACY_SCHEMA = SCHEMA.replace(" estimated_power INTEGER NOT NULL DEFAULT 0,\n", "") + + +@pytest.fixture() +def db_path(tmp_path): + p = tmp_path / "test.db" + init_db(p) + return p + + +def test_connect_enables_wal_and_foreign_keys(db_path) -> None: + conn = connect(db_path) + try: + assert conn.execute("PRAGMA journal_mode").fetchone()[0] == "wal" + assert conn.execute("PRAGMA foreign_keys").fetchone()[0] == 1 + finally: + conn.close() + + +def test_connect_creates_missing_parent_dirs(tmp_path) -> None: + p = tmp_path / "nested" / "dir" / "db.sqlite" + conn = connect(p) + try: + assert p.exists() + finally: + conn.close() + + +def test_init_db_is_idempotent(db_path) -> None: + init_db(db_path) + init_db(db_path) + with connect(db_path) as conn: + tables = {r[0] for r in conn.execute( + "SELECT name FROM sqlite_master WHERE type = 'table'" + ).fetchall()} + assert {"bikes", "activities", "power_bests"} <= tables + + +def test_estimated_power_migration_backfills_default(tmp_path) -> None: + p = tmp_path / "legacy.db" + legacy = sqlite3.connect(p) + legacy.row_factory = sqlite3.Row + legacy.executescript(LEGACY_SCHEMA) + legacy.execute( + "INSERT INTO activities (id, start_time, point_count) VALUES (1, '2026-01-01T09:00:00Z', 100)" + ) + legacy.commit() + legacy.close() + + init_db(p) + + conn = connect(p) + try: + cols = {r[1] for r in conn.execute("PRAGMA table_info(activities)")} + assert "estimated_power" in cols + value = conn.execute( + "SELECT estimated_power FROM activities WHERE id = 1" + ).fetchone()[0] + finally: + conn.close() + assert value == 0 + + +def test_foreign_keys_enforced_between_activity_and_bike(db_path) -> None: + with connect(db_path) as conn: + with pytest.raises(sqlite3.IntegrityError): + conn.execute( + "INSERT INTO activities (id, start_time, bike_id) VALUES (1, '2026-01-01T09:00:00Z', 999)" + ) + + +def test_indexes_exist(db_path) -> None: + with connect(db_path) as conn: + # PRAGMA index_list columns: (seq, name, unique, origin, partial) + activity_idx = {r[1] for r in conn.execute("PRAGMA index_list(activities)")} + bests_idx = {r[1] for r in conn.execute("PRAGMA index_list(power_bests)")} + assert {"idx_activities_start_time", "idx_activities_bike_id", "idx_activities_type"} <= activity_idx + assert "idx_power_bests_window" in bests_idx + + +def test_get_conn_closes_the_connection(data_dir) -> None: + # init_db() resolves to settings.db_path inside the temp data dir. + init_db() + with get_conn() as conn: + conn.execute("SELECT 1") + # After leaving the context manager the connection must be closed. + with pytest.raises(sqlite3.ProgrammingError): + conn.execute("SELECT 1") + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) diff --git a/tests/backend/test_health.py b/tests/backend/test_health.py new file mode 100644 index 0000000..e1d63a1 --- /dev/null +++ b/tests/backend/test_health.py @@ -0,0 +1,8 @@ +"""M0 smoke test: harness, root tests/ layout, and fixtures are wired up.""" +from __future__ import annotations + + +def test_health(api_client) -> None: + res = api_client.get("/api/health") + assert res.status_code == 200 + assert res.json() == {"status": "ok"} \ No newline at end of file diff --git a/tests/backend/test_ingest.py b/tests/backend/test_ingest.py new file mode 100644 index 0000000..09e17c6 --- /dev/null +++ b/tests/backend/test_ingest.py @@ -0,0 +1,154 @@ +"""Pipeline tests for app.ingest: directory ingest, idempotency, failure +isolation, and the upload path. All I/O is redirected to temp dirs via the +``data_dir`` fixture; the real data directory is never touched. +""" +from __future__ import annotations + +from pathlib import Path + +import pytest + +from app import db +from app.ingest import RIDE_TYPES, ingest, ingest_uploaded_file + +assert "Ride" in RIDE_TYPES, f"'Ride' missing from {RIDE_TYPES!r}" + +GPX = """ + + + Synthetic GPX ride + + 100 + 101 + 102 + 103 + 104 + + + +""" + +TCX = """ + + + + 2026-01-01T11:00:00Z + + + 48.850002.35000100 + 48.850112.35011101 + 48.850222.35022102 + 48.850332.35033103 + + + + + +""" + + +@pytest.fixture() +def conn(data_dir) -> "db.Connection": + db.init_db() + return db.connect() + + +def _activity_count(c: "db.Connection") -> int: + return c.execute("SELECT COUNT(*) FROM activities").fetchone()[0] + + +def _parquet_count() -> int: + return len(list(Path(db.settings.streams_dir).glob("*.parquet"))) + + +# Strava UI export format: space-separated headers, one file per ride. +CSV_HEADER = "Activity ID,Activity Date,Activity Name,Activity Type,Activity Description,Activity Gear,Filename,Device Watts" + + +def _csv_row(rid: int, name: str, fname: str) -> str: + return f"{rid},2026-01-01T10:00:00Z,{name},Ride,,Test Bike,{fname},true" + + +def _export_dir(tmp_path: Path, specs: list[tuple[int, str, str, str]]) -> Path: + """specs: list of (id, name, filename, file_content).""" + d = tmp_path / "export" + d.mkdir() + rows = [CSV_HEADER] + for rid, name, fname, content in specs: + (d / fname).write_text(content, encoding="utf-8") + rows.append(_csv_row(rid, name, fname)) + (d / "activities.csv").write_text("\n".join(rows) + "\n", encoding="utf-8") + return d + + +def test_ingest_directory_ingests_gpx_and_tcx(conn, tmp_path) -> None: + export = _export_dir( + tmp_path, [(1, "Synthetic GPX ride", "a.gpx", GPX), (2, "Synthetic TCX ride", "b.tcx", TCX)] + ) + result = ingest(export) + conn.commit() + + assert sum(result) == 2, f"unexpected ingest result: {result!r}" + assert _activity_count(conn) == 2 + assert _parquet_count() == 2 + + +def test_ingest_is_idempotent(conn, tmp_path) -> None: + export = _export_dir(tmp_path, [(1, "Synthetic GPX ride", "a.gpx", GPX)]) + ingest(export) + conn.commit() + assert _activity_count(conn) == 1 + + # Second run of the same export: nothing new is ingested. + ingest(export) + conn.commit() + assert _activity_count(conn) == 1 + assert _parquet_count() == 1 + + +def test_ingest_force_replaces_without_duplicates(conn, tmp_path) -> None: + export = _export_dir(tmp_path, [(1, "Synthetic GPX ride", "a.gpx", GPX)]) + ingest(export) + conn.commit() + + ingest(export, force=True) + conn.commit() + assert _activity_count(conn) == 1 + + +def test_ingest_isolates_failed_files(conn, tmp_path) -> None: + export = _export_dir( + tmp_path, + [ + (1, "Broken ride", "bad.gpx", "= 2 + assert _parquet_count() >= 2 + + +def test_ingest_uploaded_file_returns_activity_id(conn, tmp_path) -> None: + src = tmp_path / "upload.gpx" + src.write_text(GPX, encoding="utf-8") + + activity_id = ingest_uploaded_file(src) + conn.commit() + + assert isinstance(activity_id, int) and activity_id > 0 + assert _activity_count(conn) == 1 + assert _parquet_count() == 1 + row = conn.execute( + "SELECT COUNT(*) FROM activities WHERE id = ?", (activity_id,) + ).fetchone() + assert row[0] == 1 + + +if __name__ == "__main__": + raise SystemExit(pytest.main([__file__, "-v"])) \ No newline at end of file