mirror of
https://git.ianrenton.com/ian/spothole.git
synced 2026-09-20 14:27:42 +00:00
85 lines
3.9 KiB
Python
85 lines
3.9 KiB
Python
import logging
|
|
from datetime import datetime
|
|
from threading import Thread
|
|
|
|
import pytz
|
|
from requests.exceptions import ConnectionError, ConnectTimeout, ReadTimeout
|
|
|
|
from core.constants import HTTP_HEADERS
|
|
from core.url_data_cache import URLDataCache
|
|
from providers.activityrefdata.activity_ref_data_provider import ActivityRefDataProvider
|
|
|
|
logger = logging.getLogger(__name__)
|
|
|
|
|
|
class FileDownloadActivityRefDataProvider(ActivityRefDataProvider):
|
|
"""Generic activity ref data provider class for providers that fetch their data from the web by downloading a
|
|
file."""
|
|
|
|
def __init__(self, sig_name, provider_config, url, poll_interval):
|
|
"""Set up the provider, note poll_interval is in *days*."""
|
|
super().__init__(sig_name, provider_config)
|
|
self._url = url
|
|
self._poll_interval = poll_interval
|
|
self._thread = None
|
|
self._url_data_cache = URLDataCache(f"activity_ref_data_{sig_name}")
|
|
|
|
def start(self):
|
|
# Fire off the polling thread. It will poll immediately on startup, then sleep for poll_interval between
|
|
# subsequent polls, so start() returns immediately and the application can continue starting.
|
|
logger.info(f"Set up query of {self.sig_name} activity ref data every {self._poll_interval!s} days.")
|
|
self._thread = Thread(target=self._run, name=f"FileDownloadActivityRefDataProvider-{self.sig_name}", daemon=True)
|
|
self._thread.start()
|
|
|
|
def stop(self):
|
|
super().stop()
|
|
if self._thread:
|
|
self._thread.join(timeout=12)
|
|
if self._thread.is_alive():
|
|
logger.warning(f"{self.sig_name} activity ref data worker thread did not exit on time and will be killed.")
|
|
|
|
def _run(self):
|
|
while True:
|
|
self._poll()
|
|
if self._stop_event.wait(timeout=self._poll_interval * 60 * 60 * 24):
|
|
break
|
|
|
|
def _poll(self):
|
|
try:
|
|
# Request data from API. Use the data cache (with a TTL of 1 day) here, not as the main mechanism for
|
|
# caching, but just so continual restarts of the software during testing don't hammer the servers.
|
|
logger.debug(f"Downloading {self.sig_name} activity ref data...")
|
|
http_response = self._url_data_cache.get(self._url, headers=HTTP_HEADERS)
|
|
# Check response code was good
|
|
if http_response.ok:
|
|
# Pass off to the subclass for processing
|
|
new_data = self._http_response_to_data(http_response)
|
|
# Add the new data to the activity ref data store
|
|
if new_data:
|
|
self._add_data(new_data)
|
|
|
|
self.status = "OK"
|
|
self.last_update_time = datetime.now(pytz.UTC)
|
|
logger.debug(f"Received activity ref data for {self.sig_name}")
|
|
else:
|
|
self.status = "Error"
|
|
logger.warning(f"HTTP {http_response.status_code} when downloading activity ref data for {self.sig_name}.")
|
|
|
|
except ConnectionError:
|
|
self.status = "Error"
|
|
logger.warning(f"Connection error when downloading activity ref data for {self.sig_name}.")
|
|
except (ConnectTimeout, ReadTimeout):
|
|
self.status = "Error"
|
|
logger.warning(f"Timeout when downloading activity ref data for {self.sig_name}.")
|
|
except Exception:
|
|
self.status = "Error"
|
|
logger.exception(f"Exception in HTTP Activity Ref Data Provider ({self.sig_name})")
|
|
self._stop_event.wait(timeout=1)
|
|
|
|
def _http_response_to_data(self, http_response):
|
|
"""Convert an HTTP response returned by the server into activity ref data. The whole response is provided here
|
|
so the subclass implementations can check for HTTP status codes if necessary, and handle the response as
|
|
JSON, CSV, whatever the remote file actually is."""
|
|
|
|
raise NotImplementedError("Subclasses must implement this method")
|