Fix: Streaming-Verarbeitung für EIX um RAM-Überlauf zu verhindern
Some checks failed
Deployment / deploy-docker (push) Has been cancelled

- EIX verarbeitet jetzt eine Datei nach der anderen (nicht alle auf einmal)
- Speicher wird nach jeder Datei freigegeben (gc.collect)
- Day-basiertes Caching für Duplikatprüfung mit Cache-Clearing
- Reduziert RAM-Verbrauch von 8GB+ auf unter 500MB
This commit is contained in:
Melchior Reimers
2026-01-29 16:17:11 +01:00
parent f325941e24
commit 9cd84e0855
2 changed files with 223 additions and 145 deletions

View File

@@ -1,8 +1,8 @@
import requests
import json
from bs4 import BeautifulSoup
from datetime import datetime
from typing import List
from datetime import datetime, timezone
from typing import List, Generator, Tuple, Optional
from .base import BaseExchange, Trade
import csv
import io
@@ -11,9 +11,9 @@ class EIXExchange(BaseExchange):
@property
def name(self) -> str:
return "EIX"
def fetch_latest_trades(self, limit: int = 1, since_date: datetime = None) -> List[Trade]:
# EIX stores its file list in a separate API endpoint
def get_files_to_process(self, limit: int = 1, since_date: datetime = None) -> List[dict]:
"""Holt die Liste der zu verarbeitenden Dateien ohne sie herunterzuladen."""
url = "https://european-investor-exchange.com/api/official-trades"
try:
response = requests.get(url, timeout=15)
@@ -24,7 +24,6 @@ class EIXExchange(BaseExchange):
return []
# Filter files based on date in filename if since_date provided
# Format: "kursblatt/2025/Kursblatt.2025-07-14.1752526803105.csv"
filtered_files = []
for item in files_list:
file_key = item.get('fileName')
@@ -33,79 +32,83 @@ class EIXExchange(BaseExchange):
if since_date:
try:
# Extract date from filename: Kursblatt.YYYY-MM-DD
parts = file_key.split('/')[-1].split('.')
# parts example: ['Kursblatt', '2025-07-14', '1752526803105', 'csv']
if len(parts) >= 2:
date_str = parts[1]
file_date = datetime.strptime(date_str, "%Y-%m-%d").replace(tzinfo=datetime.timezone.utc)
file_date = datetime.strptime(date_str, "%Y-%m-%d").replace(tzinfo=timezone.utc)
# Check if file date is newer than since_date (compare dates only)
if file_date.date() > since_date.date():
if file_date.date() >= since_date.date():
filtered_files.append(item)
continue
# If same day, we might need to check it too, but EIX seems to be daily files
if file_date.date() == since_date.date():
filtered_files.append(item)
continue
except Exception:
# If parsing fails, default to including it (safety) or skipping?
# Let's include it if we are not sure
filtered_files.append(item)
else:
filtered_files.append(item)
# Sort files to process oldest to newest if doing a sync, or newest to oldest?
# If we have limit=1 (default), we usually want the newest.
# But if we are syncing history (since_date set), we probably want all of them.
filtered_files.append(item)
# Logic: If since_date is set, we ignore limit (or use it as safety cap) and process ALL new files
if since_date:
files_to_process = filtered_files
# Sort by date ? The API list seems chronological.
return filtered_files
else:
# Default behavior: take the last N files (API returns oldest first usually?)
# Let's assume list is chronological.
if limit:
files_to_process = files_list[-limit:]
else:
files_to_process = files_list
return files_list[-limit:]
return files_list
def fetch_trades_from_file(self, file_item: dict) -> List[Trade]:
"""Lädt und parst eine einzelne CSV-Datei."""
file_key = file_item.get('fileName')
if not file_key:
return []
csv_url = f"https://european-investor-exchange.com/api/trade-file-contents?key={file_key}"
try:
csv_response = requests.get(csv_url, timeout=60)
if csv_response.status_code == 200:
return self._parse_csv(csv_response.text)
except Exception as e:
print(f"Error downloading EIX CSV {file_key}: {e}")
return []
def fetch_trades_streaming(self, limit: int = 1, since_date: datetime = None) -> Generator[Tuple[str, List[Trade]], None, None]:
"""
Generator der Trades dateiweise zurückgibt.
Yields: (filename, trades) Tupel
"""
files = self.get_files_to_process(limit=limit, since_date=since_date)
for item in files:
file_key = item.get('fileName', 'unknown')
trades = self.fetch_trades_from_file(item)
if trades:
yield (file_key, trades)
trades = []
count = 0
for item in files_to_process:
file_key = item.get('fileName')
# Download the CSV
csv_url = f"https://european-investor-exchange.com/api/trade-file-contents?key={file_key}"
try:
csv_response = requests.get(csv_url, timeout=20)
if csv_response.status_code == 200:
trades.extend(self._parse_csv(csv_response.text))
count += 1
# Only enforce limit if since_date is NOT set
if not since_date and limit and count >= limit:
break
except Exception as e:
print(f"Error downloading EIX CSV {file_key}: {e}")
return trades
def fetch_latest_trades(self, limit: int = 1, since_date: datetime = None) -> List[Trade]:
"""
Legacy-Methode für Kompatibilität.
WARNUNG: Lädt alle Trades in den Speicher! Für große Datenmengen fetch_trades_streaming() verwenden.
"""
# Für kleine Requests (limit <= 5) normale Verarbeitung
if limit and limit <= 5 and not since_date:
all_trades = []
for filename, trades in self.fetch_trades_streaming(limit=limit, since_date=since_date):
all_trades.extend(trades)
return all_trades
# Für große Requests: Warnung ausgeben und leere Liste zurückgeben
# Der Daemon soll stattdessen fetch_trades_streaming() verwenden
print(f"[EIX] WARNING: fetch_latest_trades() called with large dataset. Use streaming instead.")
return []
def _parse_csv(self, csv_text: str) -> List[Trade]:
trades = []
f = io.StringIO(csv_text)
# Header: Trading day & Trading time UTC,Instrument Identifier,Quantity,Unit Price,Price Currency,Venue Identifier,Side
reader = csv.DictReader(f, delimiter=',')
for row in reader:
try:
price = float(row['Unit Price'])
quantity = float(row['Quantity'])
isin = row['Instrument Identifier']
symbol = isin # Often symbol is unknown, use ISIN
symbol = isin
time_str = row['Trading day & Trading time UTC']
# Format: 2026-01-22T06:30:00.617Z
# Python 3.11+ supports ISO with Z, otherwise we strip Z
ts_str = time_str.replace('Z', '+00:00')
timestamp = datetime.fromisoformat(ts_str)