mirror of
https://git.ianrenton.com/ian/spothole.git
synced 2026-09-20 06:17:41 +00:00
51 lines
1.9 KiB
Python
51 lines
1.9 KiB
Python
import io
|
|
from time import sleep
|
|
|
|
import pandas as pd
|
|
|
|
from core.enums import ActivityName, ActivityRefType
|
|
from data.activity_ref import ActivityRef
|
|
from providers.activityrefdata.file_download_activity_ref_data_provider import FileDownloadActivityRefDataProvider
|
|
|
|
|
|
class DMVE(FileDownloadActivityRefDataProvider):
|
|
"""Activity ref data provider for Diploma Monumentos y Vestigios de España"""
|
|
|
|
POLL_INTERVAL_DAYS = 365
|
|
ACTIVITY = ActivityName.DMVE
|
|
DATA_URL = "https://www.acracb.org/dmve/descargas/General/directorio_referencias_dmve.xls"
|
|
|
|
def __init__(self, provider_config):
|
|
super().__init__(self.ACTIVITY, provider_config, self.DATA_URL, self.POLL_INTERVAL_DAYS)
|
|
|
|
def _http_response_to_data(self, http_response):
|
|
new_data = []
|
|
|
|
file_stream = io.BytesIO(http_response.content)
|
|
# Despide the .xls extension this is actually an xlsx file, so we need openpyxl not xlrd
|
|
df = pd.read_excel(file_stream, engine="openpyxl", header=None)
|
|
|
|
for index, row in df.iterrows():
|
|
ref = row.iloc[0]
|
|
name = row.iloc[1]
|
|
|
|
# Skip the header row and blank rows
|
|
if str(ref) == "REF.":
|
|
continue
|
|
if pd.isna(ref) or pd.isna(name):
|
|
continue
|
|
|
|
if ref and name:
|
|
new_data.append(ActivityRef(sig=self.ACTIVITY, id=ref.strip(), name=name.strip(), ref_type=ActivityRefType.BUILDING))
|
|
|
|
# Bail out if a stop has been requested, i.e. the program is shutting down - no need to parse the rest of
|
|
# the data in this case
|
|
if self._stop_event.is_set():
|
|
break
|
|
|
|
# Very short pause. This will extend the time to handle activity refs by a few seconds but will ensure some time
|
|
# is available for other threads e.g. the web server.
|
|
sleep(0.001)
|
|
|
|
return new_data
|