from time import sleep from bs4 import BeautifulSoup from core.enums import ActivityName, ActivityRefType from data.activity_ref import ActivityRef from providers.activityrefdata.file_download_activity_ref_data_provider import FileDownloadActivityRefDataProvider class PGA(FileDownloadActivityRefDataProvider): """Activity ref data provider for Polish Gmina Award""" POLL_INTERVAL_DAYS = 30 ACTIVITY = ActivityName.PGA DATA_URL = "http://www.spga.pl/lista_pga2.php" def __init__(self, provider_config): super().__init__(self.ACTIVITY, provider_config, self.DATA_URL, self.POLL_INTERVAL_DAYS) def _http_response_to_data(self, http_response): new_data = [] soup = BeautifulSoup(http_response.text, "html.parser") # Iterate through tables in the page for table in soup.find_all("table"): header_cells = table.find_all(["th", "td"], limit=10) header_texts = [c.get_text(strip=True) for c in header_cells] # If it has "PGA" and "Nazwa" in the header, it's the main data table if any("PGA" in t for t in header_texts) and any("Nazwa" in t for t in header_texts): # Iterate through all rows except the first rows = table.find_all("tr") for row in rows[1:]: cells = row.find_all("td") ref_id = cells[0].get_text(strip=True) name = cells[1].get_text(strip=True) if not ref_id: continue new_data.append( ActivityRef( sig=self.ACTIVITY, id=ref_id, name=name, ref_type=ActivityRefType.REGION, ) ) # Bail out if a stop has been requested, i.e. the program is shutting down - no need to parse the rest # of the data in this case if self._stop_event.is_set(): break # Very short pause. This will extend the time to handle activity refs by a few seconds but will ensure some time # is available for other threads e.g. the web server. sleep(0.001) return new_data