Skip to content
Open

Dev #16

Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions .gitignore
Original file line number Diff line number Diff line change
Expand Up @@ -129,3 +129,4 @@ dmypy.json
# Pyre type checker
.pyre/
/poetry.lock
cache/
16 changes: 16 additions & 0 deletions .pre-commit-config.yaml
Original file line number Diff line number Diff line change
@@ -0,0 +1,16 @@
# See https://pre-commit.com for more information
# See https://pre-commit.com/hooks.html for more hooks
repos:
- repo: https://github.com/pre-commit/pre-commit-hooks
rev: v3.2.0
hooks:
- id: trailing-whitespace
- id: end-of-file-fixer
- id: check-yaml
- id: check-added-large-files
- id: check-case-conflict
- id: requirements-txt-fixer
- repo: https://github.com/psf/black
rev: 22.10.0
hooks:
- id: black
18 changes: 0 additions & 18 deletions .readthedocs.yml
Original file line number Diff line number Diff line change
@@ -1,6 +1,3 @@
# Read the Docs configuration file for Sphinx projects
# See https://docs.readthedocs.io/en/stable/config-file/v2.html for details

# Required
version: 2

Expand All @@ -9,27 +6,12 @@ build:
os: ubuntu-22.04
tools:
python: "3.10"
# You can also specify other tool versions:
# nodejs: "20"
# rust: "1.70"
# golang: "1.20"

# Build documentation in the "docs/" directory with Sphinx
sphinx:
configuration: docs/source/conf.py
# You can configure Sphinx to use a different builder, for instance use the dirhtml builder for simpler URLs
# builder: "dirhtml"
# Fail on all warnings to avoid broken references
# fail_on_warning: true

# Optionally build your docs in additional formats such as PDF and ePub
# formats:
# - pdf
# - epub

# Optional but recommended, declare the Python requirements required
# to build your documentation
# See https://docs.readthedocs.io/en/stable/guides/reproducible-builds.html
python:
install:
- requirements: requirements.txt
19 changes: 17 additions & 2 deletions Makefile
Original file line number Diff line number Diff line change
@@ -1,10 +1,25 @@
FILES = SOIKA/*
FILES = ./factfinder/src/*

lint:
@echo "Running pylint check..."
python -m pylint ${FILES}

format:
@echo "Starting black formatting..."
python -m black ${FILES}

test:
python -m pytest
@echo "Starting tests..."
python -m pytest

activate:
@echo "Activating virtual environment..."
poetry shell

install:
@echo "Installing poetry dependencies..."
poetry install
poetry run pre-commit install

setup: install activate
finish: lint test
1 change: 1 addition & 0 deletions factfinder/src/event_detection.py
Original file line number Diff line number Diff line change
Expand Up @@ -18,6 +18,7 @@ class EventDetection:
It is based on the application of semantic clustering method (BERTopic)
on the texts in the context of urban spatial model
"""

def __init__(self):
np.random.seed(42)
self.population_filepath = None
Expand Down
512 changes: 256 additions & 256 deletions factfinder/src/exceptions_countries.csv

Large diffs are not rendered by default.

51 changes: 34 additions & 17 deletions factfinder/src/geocoder.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@
In this scenario texts are comments in social networks (e.g. Vkontakte).
Thus the model was trained on the corpus of comments on Russian language.
"""
import numpy as np
import numpy as np
import re
import warnings
from typing import List, Optional
Expand All @@ -28,19 +28,16 @@
from natasha import (
Segmenter,
MorphVocab,

NewsEmbedding,
NewsMorphTagger,
NewsSyntaxParser,
NewsNERTagger,

PER,
NamesExtractor,
DatesExtractor,
MoneyExtractor,
AddrExtractor,

Doc
Doc,
)

segmenter = Segmenter()
Expand Down Expand Up @@ -239,9 +236,20 @@ class Geocoder:
dir_path = os.path.dirname(os.path.realpath(__file__))

global_crs: int = 4326
exceptions = pd.merge(pd.read_csv(os.path.join(dir_path, "exceptions_countries.csv"), encoding="utf-8", sep=",")
,pd.read_csv(os.path.join(dir_path, "exсeptions_city.csv"), encoding="utf-8", sep=","),
on='Сокращенное наименование', how='outer')
exceptions = pd.merge(
pd.read_csv(
os.path.join(dir_path, "exceptions_countries.csv"),
encoding="utf-8",
sep=",",
),
pd.read_csv(
os.path.join(dir_path, "exсeptions_city.csv"),
encoding="utf-8",
sep=",",
),
on="Сокращенное наименование",
how="outer",
)

def __init__(
self,
Expand Down Expand Up @@ -283,35 +291,40 @@ def extract_ner_street(self, text: str) -> pd.Series:
return pd.Series([res, score])
else:
return pd.Series([None, None])


except IndexError:
return pd.Series([None, None])

# Блок с Наташей
def get_ner_address_natasha(row, exceptions, text_col): #input: string, list, series... output: string
def get_ner_address_natasha(
row, exceptions, text_col
): # input: string, list, series... output: string
if row["Street"] == None or row["Street"] == np.nan:
i = row[text_col]
location_final = []
i = re.sub(r'\[.*?\]', '', i)
i = re.sub(r"\[.*?\]", "", i)
doc = Doc(i)
doc.segment(segmenter)
doc.tag_morph(morph_tagger)
doc.parse_syntax(syntax_parser)
doc.tag_ner(ner_tagger)
for span in doc.spans:
span.normalize(morph_vocab)
location = list(filter(lambda x: x.type == 'LOC', doc.spans))
location = list(filter(lambda x: x.type == "LOC", doc.spans))
for span in location:
if span.normal.lower() not in exceptions['Сокращенное наименование'].str.lower().values:
if (
span.normal.lower()
not in exceptions["Сокращенное наименование"]
.str.lower()
.values
):
location_final.append(span)
location_final = [(span.text) for span in location_final]
if not location_final:
return None
return location_final[0]
else:
return row["Street"]


@staticmethod
def get_stem(street_names_df: pd.DataFrame) -> pd.DataFrame:
Expand Down Expand Up @@ -415,8 +428,12 @@ def get_street(
df[["Street", "Score"]] = df[text_column].progress_apply(
lambda t: self.extract_ner_street(t)
)
df["Street"] = df[[text_column,"Street"]].progress_apply(
lambda row: Geocoder.get_ner_address_natasha(row, self.exceptions, text_column), axis=1)
df["Street"] = df[[text_column, "Street"]].progress_apply(
lambda row: Geocoder.get_ner_address_natasha(
row, self.exceptions, text_column
),
axis=1,
)
df = df[df.Street.notna()]
df = df[df["Street"].str.contains("[а-яА-Я]")]

Expand Down
30 changes: 30 additions & 0 deletions factfinder/src/test.ipynb
Original file line number Diff line number Diff line change
@@ -0,0 +1,30 @@
{
"cells": [
{
"cell_type": "code",
"execution_count": null,
"metadata": {},
"outputs": [],
"source": [
"def get_city_data(city_name):\n",
" api = overpy.Overpass()\n",
" query = f\"\"\"\n",
" area[name=\"{city_name}\"]->.searchArea;\n",
" (node(area.searchArea)[\"place\"=\"city\"];\n",
" way(area.searchArea)[\"place\"=\"city\"];\n",
" relation(area.searchArea)[\"place\"=\"city\"];);\n",
" out center;\n",
" \"\"\"\n",
" result = api.query(query)\n",
" return result"
]
}
],
"metadata": {
"language_info": {
"name": "python"
}
},
"nbformat": 4,
"nbformat_minor": 2
}
1 change: 1 addition & 0 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -36,6 +36,7 @@ autodocsumm = "^0.2.11"
flake8 = "^6.0.0"
isort = "^5.12.0"
black = "^23.1.0"
pylint = "^3.0.2"

[tool.poetry.group.test.dependencies]
pytest = "^7.4.3"
Expand Down
16 changes: 10 additions & 6 deletions tests/test_events_modelling.py
Original file line number Diff line number Diff line change
Expand Up @@ -8,23 +8,27 @@
path_to_population = "data/raw/population.geojson"
path_to_data = "data/processed/messages.geojson"


@pytest.fixture
def gdf():
gdf = gpd.read_file(path_to_data)
gdf = gdf.head(6)
return gdf


def test_event_detection(gdf):
expected_name = "0_фурштатская_штукатурного слоя_слоя_отслоение"
expected_risk = 0.405
expected_messages = [4, 5, 3, 2]
event_model = EventDetection()
messages, events, connections = event_model.run(
gdf, path_to_population, 'Санкт-Петербург', 32636, min_event_size=3
)
event_name = events.iloc[0]['name']
event_risk = events.iloc[0]['risk'].round(3)
event_messages = [int(mid) for mid in events.iloc[0]['message_ids'].split(', ')]
gdf, path_to_population, "Санкт-Петербург", 32636, min_event_size=3
)
event_name = events.iloc[0]["name"]
event_risk = events.iloc[0]["risk"].round(3)
event_messages = [
int(mid) for mid in events.iloc[0]["message_ids"].split(", ")
]
assert event_name == expected_name
assert event_risk == expected_risk
assert all(mid in event_messages for mid in expected_messages)
assert all(mid in event_messages for mid in expected_messages)
9 changes: 3 additions & 6 deletions tests/test_location.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,16 +16,13 @@
# result = Location().query(input_address)
# assert result.address == geocode_result


@pytest.mark.parametrize(
"input_address,geocode_result",
[
(
'возле дома на Итальянской 17 постоянно мусорят!!!',
'Итальянской 17'
),
("возле дома на Итальянской 17 постоянно мусорят!!!", "Итальянской 17"),
],
)

def test_geolocator(input_address, geocode_result):
result = Geocoder().extract_ner_street(input_address)
assert result.loc[0] == geocode_result
assert result.loc[0] == geocode_result