217 lines
7.2 KiB
Python
217 lines
7.2 KiB
Python
import datetime
|
|
import re
|
|
|
|
from bs4 import BeautifulSoup, Tag
|
|
|
|
from gallery.sketch.parse.core import Parser
|
|
from gallery.sketch.parse.date import parse_date
|
|
from gallery.sketch.parse.html import (
|
|
TableParser,
|
|
TableValueParser,
|
|
TagIterParser,
|
|
TagParser,
|
|
)
|
|
from gallery.sketch.weather.model import (
|
|
Cloudness,
|
|
Location,
|
|
Precipitation,
|
|
WeatherResponse,
|
|
WeatherValue,
|
|
WindDirection,
|
|
)
|
|
from gallery.sketch.weather.parser import fix_dates_sequence, parse_wind_direction
|
|
|
|
|
|
class LocationNameParser(TagParser[str]):
|
|
selector = ".intoLeftNavi > .verticalBottom > a"
|
|
|
|
def extract(self, data: Tag) -> str:
|
|
return data.text.strip()
|
|
|
|
|
|
class DatesParser(TagIterParser[tuple[datetime.datetime, int]]):
|
|
selector = "#forecastTable > tr.forecastDate > td"
|
|
|
|
def extract(self, data: Tag) -> tuple[datetime.datetime, int]:
|
|
colspan = int(data.attrs["colspan"])
|
|
index = data.parent.index(data)
|
|
if index == 1:
|
|
colspan -= 1
|
|
date_str = ",".join(data.text.strip().split(",")[-2:])
|
|
date = parse_date(date_str)
|
|
return (date, colspan)
|
|
|
|
|
|
class TimeParser(TableValueParser[datetime.datetime]):
|
|
selector = "tr.forecastTime > td:not(:first-of-type)"
|
|
model_key = "date"
|
|
|
|
def extract(self, data: Tag) -> datetime.datetime:
|
|
hour = int(data.text.strip())
|
|
return datetime.datetime.min.replace(hour=hour)
|
|
|
|
|
|
class CloudnessParser(TableValueParser[Cloudness]):
|
|
selector = "tr:nth-of-type(3) > td:not(:first-of-type)"
|
|
model_key = "sky.cloudness"
|
|
|
|
CLOUDNESS_MAP = {
|
|
"Ясно": Cloudness.CLEAR,
|
|
"Малооблачно": Cloudness.PARTLY_CLOUDY,
|
|
"Облачно с прояснениями": Cloudness.PARTLY_CLOUDY,
|
|
"Переменная облачность": Cloudness.PARTLY_CLOUDY,
|
|
"Небольшая облачность": Cloudness.CLOUDY,
|
|
"Облачно": Cloudness.CLOUDY,
|
|
"Значительная облачность": Cloudness.MAINLY_CLOUDY,
|
|
"Пасмурная погода": Cloudness.MAINLY_CLOUDY,
|
|
}
|
|
|
|
def extract(self, data: Tag) -> Cloudness:
|
|
onmouseover = data.select_one(".cc_0").next_element.attrs["onmouseover"]
|
|
cloudness_str = re.compile("<b>(.*?)</b>").search(onmouseover).group(1).strip()
|
|
return self.CLOUDNESS_MAP[cloudness_str]
|
|
|
|
|
|
class PrecipitationParser(TableValueParser[tuple[Precipitation, float]]):
|
|
selector = "tr:nth-of-type(4) > td:not(:first-of-type)"
|
|
model_key = ["sky.precipitation", "precipitation"]
|
|
|
|
PRECIPITATION_MAP = {
|
|
"Явления погоды отсутствуют": Precipitation.NO,
|
|
"Cлабый дождь": Precipitation.SMALL_RAIN,
|
|
}
|
|
|
|
def extract(self, data: Tag) -> tuple[Precipitation, float]:
|
|
onmouseover = data.select_one(".pr_0").attrs["onmouseover"]
|
|
precipitation_str = onmouseover.split(",")[1].strip(" '")
|
|
precipitation_value_matched = re.compile("(\\d+\\.\\d+) мм").search(precipitation_str)
|
|
precipitation_value = 0
|
|
if precipitation_value_matched:
|
|
# one hour value
|
|
precipitation_value = float(precipitation_value_matched.group(1)) * 6
|
|
if "(" in precipitation_str:
|
|
precipitation_str = precipitation_str.split(" (")[0]
|
|
return self.PRECIPITATION_MAP[precipitation_str], precipitation_value
|
|
|
|
|
|
class TemperatureParser(TableValueParser[list[int]]):
|
|
selector = "tr:nth-of-type(5) > td:not(:first-of-type)"
|
|
model_key = "temperature"
|
|
|
|
def extract(self, data: Tag) -> list[int]:
|
|
return [int(data.select_one(".t_0").text.lstrip("+").strip())]
|
|
|
|
|
|
class PressureParser(TableValueParser[list[int]]):
|
|
selector = "tr:nth-last-of-type(8) > td:not(:first-of-type)"
|
|
model_key = "pressure"
|
|
|
|
def extract(self, data: Tag) -> list[int]:
|
|
return [int(data.select_one(".p_0").text.strip())]
|
|
|
|
|
|
class WindSpeedParser(TableValueParser[int]):
|
|
selector = "tr:nth-last-of-type(7) > td:not(:first-of-type)"
|
|
model_key = "wind.speed"
|
|
|
|
def extract(self, data: Tag) -> int:
|
|
element = data.select_one(".wv_0")
|
|
if element:
|
|
return int(element.text.strip() or "0")
|
|
return 0
|
|
|
|
|
|
class WindGustParser(WindSpeedParser):
|
|
selector = "tr:nth-last-of-type(6) > td:not(:first-of-type)"
|
|
model_key = "wind.gust"
|
|
|
|
|
|
class WindDirectionParser(TableValueParser[WindDirection | None]):
|
|
selector = "tr:nth-last-of-type(5) > td:not(:first-of-type)"
|
|
model_key = "wind.direction"
|
|
|
|
def extract(self, data: Tag) -> WindDirection | None:
|
|
return parse_wind_direction(data.text.strip())
|
|
|
|
|
|
class HumidityParser(TableValueParser[int]):
|
|
selector = "tr:nth-last-of-type(4) > td:not(:first-of-type)"
|
|
model_key = "humidity"
|
|
|
|
def extract(self, data: Tag) -> int:
|
|
return int(data.text.strip())
|
|
|
|
|
|
class RP5TableParser(TableParser[WeatherValue]):
|
|
selector = "#forecastTable"
|
|
value_parsers = [
|
|
TimeParser(),
|
|
CloudnessParser(),
|
|
PrecipitationParser(),
|
|
TemperatureParser(),
|
|
PressureParser(),
|
|
WindSpeedParser(),
|
|
WindGustParser(),
|
|
WindDirectionParser(),
|
|
HumidityParser(),
|
|
]
|
|
|
|
def create_model(self, data: dict) -> WeatherValue:
|
|
return WeatherValue.model_validate(data)
|
|
|
|
|
|
class RP5ForecastParser(Parser[str, list[WeatherResponse]]):
|
|
location_parser = LocationNameParser()
|
|
dates_parser = DatesParser()
|
|
table_parser = RP5TableParser()
|
|
|
|
def parse(self, data: str) -> list[WeatherResponse]:
|
|
soup = BeautifulSoup(data, features="html.parser")
|
|
location = self.location_parser.parse(soup)
|
|
dates = self.dates_parser.parse(soup)
|
|
values = self.table_parser.parse(soup)
|
|
responses = []
|
|
for date, count in dates:
|
|
item_values, values = values[:count], values[count:]
|
|
for value in item_values:
|
|
value.date = date.replace(hour=value.date.hour)
|
|
responses.append(
|
|
WeatherResponse(
|
|
location=location,
|
|
date=date,
|
|
period="day",
|
|
values=item_values,
|
|
)
|
|
)
|
|
fix_dates_sequence(responses)
|
|
return responses
|
|
|
|
|
|
class RP5LocationParser(Parser[str, list[Location]]):
|
|
|
|
def parse(self, data: str) -> list[Location]:
|
|
soup = BeautifulSoup(data, features="html.parser")
|
|
table = soup.select_one(".searchResults")
|
|
result = []
|
|
for row in table.select("tr")[1:]:
|
|
cells = row.select("td")
|
|
name = cells[0].text.strip()
|
|
location_id = cells[0].select_one("a").attrs["href"].lstrip("/")
|
|
country = cells[1].text.strip()
|
|
district = cells[2].text.strip()
|
|
subdistrict = cells[3].text.strip()
|
|
result.append(
|
|
Location(
|
|
id=location_id,
|
|
name=name,
|
|
provider="rp5", # TODO: provider module const
|
|
lat=0,
|
|
lon=0,
|
|
country=country,
|
|
country_code="ru", # TODO: contry_code from country
|
|
district=district,
|
|
subdistrict=subdistrict,
|
|
)
|
|
)
|
|
return result
|