import io from time import sleep import pandas as pd from core.enums import ActivityRefType from data.activity_ref import ActivityRef from providers.activityrefdata.file_download_activity_ref_data_provider import FileDownloadActivityRefDataProvider class DMVE(FileDownloadActivityRefDataProvider): """Activity ref data provider for Diploma Monumentos y Vestigios de EspaƱa""" POLL_INTERVAL_DAYS = 365 ACTIVITY = "DMVE" DATA_URL = "https://www.acracb.org/dmve/descargas/General/directorio_referencias_dmve.xls" def __init__(self, provider_config): super().__init__(self.ACTIVITY, provider_config, self.DATA_URL, self.POLL_INTERVAL_DAYS) def _http_response_to_data(self, http_response): new_data = [] file_stream = io.BytesIO(http_response.content) # Despide the .xls extension this is actually an xlsx file, so we need openpyxl not xlrd df = pd.read_excel(file_stream, engine="openpyxl", header=None) for index, row in df.iterrows(): ref = row.iloc[0] name = row.iloc[1] # Skip the header row and blank rows if str(ref) == "REF.": continue if pd.isna(ref) or pd.isna(name): continue if ref and name: new_data.append(ActivityRef(sig=self.ACTIVITY, id=ref.strip(), name=name.strip(), ref_type=ActivityRefType.BUILDING)) # Bail out if a stop has been requested, i.e. the program is shutting down - no need to parse the rest of # the data in this case if self._stop_event.is_set(): break # Very short pause. This will extend the time to handle activity refs by a few seconds but will ensure some time # is available for other threads e.g. the web server. sleep(0.001) return new_data