From fcb47107f38b1f7abf7904d52adf137f7b6bc2b5 Mon Sep 17 00:00:00 2001 From: gabbypolito Date: Fri, 25 Sep 2026 14:35:53 -0400 Subject: [PATCH] Remove dead code and fixed README file --- services/data-ngin/README.md | 19 ++-------- services/data-ngin/src/data_ngin/main.py | 45 ------------------------ 2 files changed, 3 insertions(+), 61 deletions(-) delete mode 100644 services/data-ngin/src/data_ngin/main.py diff --git a/services/data-ngin/README.md b/services/data-ngin/README.md index f6790d3e..7327cbb4 100644 --- a/services/data-ngin/README.md +++ b/services/data-ngin/README.md @@ -37,7 +37,6 @@ The **data-ngin** is a modular pipeline designed to fetch, clean, store, and ana - **Modularity:** Designed with interchangeable components for fetchers, loaders, cleaners, and inserters ## Project Structure -- **`src/main.py`**: Entry point for pipeline execution - **Primary Modules**: - **Loader**: Loads metadata and configuration (e.g., `CSVLoader`) - **Fetcher**: Fetches raw data (e.g., `DatabentoFetcher`) @@ -62,7 +61,6 @@ The **data-ngin** is a modular pipeline designed to fetch, clean, store, and ana ├── src │ ├── config │ │ ├── config.yaml -│ ├── main.py │ ├── modules │ │ ├── cleaner │ │ │ ├── cleaner.py @@ -122,7 +120,7 @@ The **data-ngin** is a modular pipeline designed to fetch, clean, store, and ana 1. **Clone the Repository:** ```bash - git clone https://github.com/AlgoGators/data-ngin.git + git clone https://github.com/AlgoGators/algogators.git cd data-ngin ``` @@ -159,11 +157,6 @@ The **data-ngin** is a modular pipeline designed to fetch, clean, store, and ana ## Usage -### Run Pipeline Locally -```bash -poetry run python src/main.py -``` - ### Access Airflow Web Interface 1. Navigate to http://localhost:8080 2. Use default credentials (admin/admin) unless modified @@ -435,19 +428,13 @@ task = PythonOperator(task_id="fetch_data", python_callable=orchestrator.run, da ## Common Development Tasks -1. **Run the pipeline manually** - -```bash -poetry run python src/main.py -``` - -2. **Check database contents** +1. **Check database contents** ```sql SELECT * FROM futures_data.ohlcv_1d LIMIT 10; ``` -3. **Trigger Airflow DAG** +2. **Trigger Airflow DAG** ```bash airflow dags trigger data_pipeline diff --git a/services/data-ngin/src/data_ngin/main.py b/services/data-ngin/src/data_ngin/main.py deleted file mode 100644 index 22dd0ee0..00000000 --- a/services/data-ngin/src/data_ngin/main.py +++ /dev/null @@ -1,45 +0,0 @@ -import asyncio -import logging -import os - -from data_ngin.application.orchestrator import Orchestrator -from data_ngin.utils.dynamic_loader import DEFAULT_CONFIG_PATH, load_config - - -def main() -> None: - """ - Main entry point for the data pipeline. - - TO-DO: - - Create file for interacting with database and pulling data (look into ORMs) - - Change OHLCV class to to handle more than just _1d - - Dockerize and implement Airflow - """ - # Configure logging - logging.basicConfig(level=logging.INFO, format="%(asctime)s - %(levelname)s - %(message)s") - - try: - # Load configuration - file_path = os.path.dirname(DEFAULT_CONFIG_PATH) - for filename in os.listdir(file_path): - file_path = os.path.join(file_path, filename) - config_path = DEFAULT_CONFIG_PATH - logging.info(f"Loading configuration from {config_path}") - config = load_config(config_path) - - # Initialize the Orchestrator - logging.info("Initializing orchestrator...") - orchestrator = Orchestrator(config=config) - - # Run the pipeline - logging.info("Starting the data pipeline...") - asyncio.run(orchestrator.run()) - - logging.info("Pipeline execution completed successfully.") - - except Exception as e: - logging.error(f"Pipeline execution failed: {e}") - - -if __name__ == "__main__": - main()