Compare commits
12
Commits
76d890980c
..
main
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
196972b90b | ||
|
|
d205625ef4 | ||
|
|
6060930208 | ||
|
|
0cd5377442 | ||
|
|
29459d5386 | ||
|
|
4f1835c8f8 | ||
|
|
32bd83f054 | ||
|
|
2cbf2af0de | ||
|
|
13d47be9c1 | ||
|
|
14e314822d | ||
|
|
3953d6d2ff | ||
|
|
a6325468ca |
@@ -13,7 +13,9 @@ def main(folder: str = "plots"):
|
||||
timestamp = timestamp.replace(":", "-")
|
||||
plot_df = create_plot_df(datetime.datetime.now(), _df_state)
|
||||
print(plot_df.sum(1))
|
||||
fig.savefig(Path(folder) / f"digital_plot_{timestamp}.png", dpi=300)
|
||||
fig.savefig(
|
||||
Path(folder) / f"digital_plot_{timestamp}.png", dpi=300, bbox_inches="tight"
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
+17
-4
@@ -10,10 +10,23 @@ import numpy as np
|
||||
import pandas as pd
|
||||
import scipy
|
||||
|
||||
from wsgi import create_fig, create_plot_df, plot
|
||||
from wsgi import create_fig, create_plot_df, get_tables, plot
|
||||
|
||||
|
||||
def create_dfs():
|
||||
def create_dfs(url: str = "https://beschaeftigtenbefragung.verdi.de/"):
|
||||
try:
|
||||
df, df_state, curr_datetime = get_tables(url)
|
||||
|
||||
df = df.sort_values(
|
||||
["Digitale Befragung", "Bundesland", "Bezirk"],
|
||||
ascending=[False, True, True],
|
||||
)
|
||||
|
||||
df_state = df_state.sort_values("Landesbezirk")
|
||||
plot_df = create_plot_df(curr_datetime, df_state)
|
||||
|
||||
except Exception as e:
|
||||
print(e)
|
||||
last_file = sorted(Path("data").iterdir())[-1]
|
||||
key = last_file.name[:10]
|
||||
|
||||
@@ -86,7 +99,7 @@ def main():
|
||||
date_range,
|
||||
vals,
|
||||
label=f"Lineare Regression ($R^2={reg.rvalue**2:.3f}$)",
|
||||
color="tab:blue",
|
||||
color="tab:green",
|
||||
zorder=1,
|
||||
)
|
||||
plt.plot(
|
||||
@@ -108,7 +121,7 @@ def main():
|
||||
|
||||
plt.gca().set_xticks([target_time])
|
||||
plt.title("Projektion Teilnahme an Digitaler Beschäftigtenbefragung")
|
||||
plt.savefig("plots/regression.png")
|
||||
plt.savefig("plots/regression.png", bbox_inches="tight", dpi=300)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
@@ -46,13 +46,13 @@ app.config.from_mapping(config)
|
||||
cache = Cache(app)
|
||||
|
||||
|
||||
def get_tables(url: str) -> tuple[pd.DataFrame, pd.DataFrame]:
|
||||
def get_tables(url: str) -> tuple[pd.DataFrame, pd.DataFrame, datetime.datetime]:
|
||||
bez_data = get_bez_data(["bez_data_0", "bez_data_2"], url)
|
||||
|
||||
df = construct_dataframe(bez_data=bez_data[0], special_tag="stud")
|
||||
df_state = construct_dataframe(bez_data=bez_data[1])
|
||||
|
||||
return df, df_state
|
||||
return df, df_state, datetime.datetime.now()
|
||||
|
||||
|
||||
def create_plot_df(
|
||||
@@ -113,7 +113,10 @@ def plot(
|
||||
fix_lims: bool = True,
|
||||
max_shading_date=None,
|
||||
) -> str:
|
||||
fig = plt.figure(dpi=300)
|
||||
fig = plt.figure(dpi=300, figsize=(8.5, 5))
|
||||
|
||||
target_time = pd.Timestamp("2023-10-01")
|
||||
plt.axvline(x=target_time, color="tab:green", linestyle=":")
|
||||
|
||||
if fix_lims:
|
||||
for total_target in total_targets:
|
||||
@@ -173,7 +176,7 @@ def plot(
|
||||
idx = np.argmin(nearest_target)
|
||||
|
||||
ceil_val = max(max_val, total_targets[idx])
|
||||
plt.ylim(0, ceil_val * 1.025)
|
||||
plt.ylim(0, ceil_val * 1.04)
|
||||
plt.legend()
|
||||
|
||||
# use timezone offset to center tick labels
|
||||
@@ -202,14 +205,14 @@ def plot(
|
||||
|
||||
# fill weekends
|
||||
if max_shading_date is None:
|
||||
max_shading_date = df.index.max() + datetime.timedelta(days=3)
|
||||
max_shading_date = df.index.max() + datetime.timedelta(days=4)
|
||||
days = pd.date_range(start="2023-08-14", end=max_shading_date)
|
||||
for idx, day in enumerate(days[:-1]):
|
||||
if day.weekday() >= 5:
|
||||
plt.gca().axvspan(days[idx], days[idx + 1], alpha=0.2, color="gray")
|
||||
|
||||
# reset xlim
|
||||
plt.xlim(xlim)
|
||||
plt.xlim((xlim[0], pd.Timestamp("2023-10-02")))
|
||||
|
||||
plt.tight_layout()
|
||||
|
||||
@@ -224,7 +227,7 @@ def create_fig(
|
||||
):
|
||||
curr_datetime = datetime.datetime.now()
|
||||
try:
|
||||
df, df_state = get_tables(url)
|
||||
df, df_state, curr_datetime = get_tables(url)
|
||||
|
||||
df = df.sort_values(
|
||||
["Digitale Befragung", "Bundesland", "Bezirk"],
|
||||
@@ -251,7 +254,7 @@ def create_fig(
|
||||
{"Digitale Befragung": "Int32"}
|
||||
)
|
||||
|
||||
plot_df = create_plot_df(curr_datetime)
|
||||
plot_df = create_plot_df(curr_datetime, df_state)
|
||||
annotate_current = False
|
||||
timestamp = Markup(f'<font color="red">{key} 10:00:00</font>')
|
||||
|
||||
@@ -287,12 +290,38 @@ def convert_fig_to_svg(fig: plt.Figure) -> str:
|
||||
def _print_as_html(
|
||||
df: pd.DataFrame,
|
||||
output_str: list[str],
|
||||
total: int | None = None,
|
||||
df_state: pd.DataFrame | None = None,
|
||||
dropna: bool = True,
|
||||
) -> list[str]:
|
||||
df = df.astype({"Digitale Befragung": "Int32"})
|
||||
missing_df = (
|
||||
df[["Digitale Befragung"]]
|
||||
.isna()
|
||||
.join(df[["Landesbezirk"]])
|
||||
.groupby("Landesbezirk")
|
||||
.sum()
|
||||
)
|
||||
total = df_state["Digitale Befragung"].sum() if df_state is not None else None
|
||||
|
||||
if df_state is not None:
|
||||
for idx, row in missing_df.loc[
|
||||
missing_df["Digitale Befragung"] == 1
|
||||
].iterrows():
|
||||
df_tmp = df.loc[df["Landesbezirk"] == idx]
|
||||
df_state_tmp = df_state.loc[df_state["Landesbezirk"] == idx]
|
||||
missing_idx = df_tmp.loc[df_tmp.isna().any(axis=1)].iloc[0].name
|
||||
df["Digitale Befragung"].loc[missing_idx] = (
|
||||
df_state_tmp["Digitale Befragung"].sum()
|
||||
- df_tmp["Digitale Befragung"].sum()
|
||||
)
|
||||
|
||||
df = df.sort_values(
|
||||
["Digitale Befragung", "Landesbezirk", "Bezirk"],
|
||||
ascending=[False, True, True],
|
||||
)
|
||||
if dropna:
|
||||
df = df.dropna()
|
||||
|
||||
with pd.option_context("display.max_rows", None):
|
||||
table = df.to_html(
|
||||
index_names=False,
|
||||
@@ -315,11 +344,12 @@ def _print_as_html(
|
||||
]
|
||||
)
|
||||
if total and (diff := total - df["Digitale Befragung"].sum()):
|
||||
tfoot.extend(
|
||||
[
|
||||
" <tr>",
|
||||
" <td>Weitere Bezirke</td>",
|
||||
]
|
||||
tfoot.append(" <tr>")
|
||||
num_missing = missing_df["Digitale Befragung"].sum()
|
||||
tfoot.append(
|
||||
f" <td>Weitere Bezirke ({num_missing})</td>"
|
||||
if num_missing
|
||||
else f" <td>Weitere Bezirke</td>"
|
||||
)
|
||||
for i in range(len(df.columns) - 2):
|
||||
tfoot.append(" <td></td>")
|
||||
@@ -366,9 +396,7 @@ def state_dashboard(state: str):
|
||||
|
||||
output_str = []
|
||||
output_str = _print_as_html(df_state, output_str, dropna=False)
|
||||
output_str = _print_as_html(
|
||||
df, output_str, total=df_state["Digitale Befragung"].sum(), dropna=False
|
||||
)
|
||||
output_str = _print_as_html(df, output_str, df_state=df_state, dropna=False)
|
||||
|
||||
return render_template(
|
||||
"base.html",
|
||||
@@ -396,9 +424,7 @@ def dashboard():
|
||||
|
||||
output_str = []
|
||||
output_str = _print_as_html(df_state, output_str, dropna=False)
|
||||
output_str = _print_as_html(
|
||||
df, output_str, total=df_state["Digitale Befragung"].sum()
|
||||
)
|
||||
output_str = _print_as_html(df, output_str, df_state)
|
||||
|
||||
return render_template(
|
||||
"base.html",
|
||||
@@ -408,5 +434,13 @@ def dashboard():
|
||||
)
|
||||
|
||||
|
||||
@app.route("/total")
|
||||
@cache.cached(timeout=60)
|
||||
def total_result(url: str = "https://beschaeftigtenbefragung.verdi.de/"):
|
||||
df, df_state, curr_datetime = get_tables(url)
|
||||
total = df_state["Digitale Befragung"].sum().item()
|
||||
return f"{total}"
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
app.run()
|
||||
|
||||
Reference in New Issue
Block a user