From 69031cd4db8d05413db0e1c53fc00a184cb3f0f7 Mon Sep 17 00:00:00 2001 From: Ryuki Kayano <20t1479y@student.gs.chiba-u.jp> Date: Thu, 9 Jul 2026 12:49:09 +0900 Subject: [PATCH] =?UTF-8?q?*=20major=20change=20plot=5Fdata=E9=96=A2?= =?UTF-8?q?=E6=95=B0=E3=82=92=E8=BF=BD=E5=8A=A0=E3=81=97=E3=81=9F=E3=80=82?= =?UTF-8?q?=20-=20axs[0]=EF=BC=9Abox=20plot=E3=81=A7=E4=BB=96=E7=A0=94?= =?UTF-8?q?=E7=A9=B6=E5=AE=A4=E3=81=A8=E3=81=AE=E6=AF=94=E8=BC=83=E3=80=82?= =?UTF-8?q?=20-=20axs[1]=EF=BC=9AGP=E3=81=AE=E5=88=86=E5=B8=83=E3=81=A8?= =?UTF-8?q?=E5=AF=BE=E8=B1=A1=E7=A0=94=E7=A9=B6=E5=AE=A4=E3=81=AE=E5=BF=97?= =?UTF-8?q?=E6=9C=9B=E8=80=85=E3=81=AE=E4=BD=8D=E7=BD=AE=E3=82=92=E5=8F=AF?= =?UTF-8?q?=E8=A6=96=E5=8C=96=E3=80=82=20-=20base.txt=E3=82=92=E8=AA=AD?= =?UTF-8?q?=E3=82=80=E3=81=A8=E3=80=81base.png=E3=81=A7=E4=BF=9D=E5=AD=98?= =?UTF-8?q?=E3=80=82=20*=20miner=20change=20print(df=5Fr)=E9=83=A8?= =?UTF-8?q?=E5=88=86=E3=82=92=E3=80=81GP=20mean=E3=81=A7=E3=82=BD=E3=83=BC?= =?UTF-8?q?=E3=83=88=E3=81=97=E3=81=A6=E3=80=81=E9=99=8D=E9=A0=86=E3=81=A7?= =?UTF-8?q?=E8=A1=A8=E7=A4=BA=E3=81=99=E3=82=8B=E3=82=88=E3=81=86=E3=81=AB?= =?UTF-8?q?=E5=A4=89=E6=9B=B4=E3=81=97=E3=81=9F=E3=80=82?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- analyze_lab_placement_survey.py | 376 +++++++++++++++++++++++++++----- 1 file changed, 327 insertions(+), 49 deletions(-) diff --git a/analyze_lab_placement_survey.py b/analyze_lab_placement_survey.py index 24a7fbb..1d9a8e8 100755 --- a/analyze_lab_placement_survey.py +++ b/analyze_lab_placement_survey.py @@ -1,7 +1,6 @@ #!/usr/bin/env python import argparse import re - import numpy as np import pandas as pd @@ -29,15 +28,13 @@ def extract_values_from_body(body): # -------------------------------------------------- # Case 1: 「値 カウント」表がある場合 # -------------------------------------------------- - m = re.search(r'^値\s+カウント\s*$', body, flags=re.MULTILINE) + m = re.search(r"^値\s+カウント\s*$", body, flags=re.MULTILINE) if m: - table_part = body[m.end():] + table_part = body[m.end() :] count_rows = re.findall( - r'^\s*(-?\d+(?:\.\d+)?)\s+(\d+)\s*$', - table_part, - flags=re.MULTILINE + r"^\s*(-?\d+(?:\.\d+)?)\s+(\d+)\s*$", table_part, flags=re.MULTILINE ) values = [] @@ -67,7 +64,7 @@ def extract_values_from_body(body): continue # 例: 365(修正済) - m = re.match(r'^(-?\d+(?:\.\d+)?)', line) + m = re.match(r"^(-?\d+(?:\.\d+)?)", line) if m: values.append(float(m.group(1))) @@ -96,31 +93,31 @@ def parse_google_form_results(text): # 研究室1 # 7 件の回答 m = re.search( - r'^研究室\s*[0-9]+[^\n]*\n' - r'(?:[^\d\n]*?)' - r'[0-9]+\s*件の回答', + r"^研究室\s*[0-9]+[^\n]*\n" + r"(?:[^\d\n]*?)" + r"[0-9]+\s*件の回答", text, - flags=re.MULTILINE + flags=re.MULTILINE, ) if not m: raise ValueError("個別研究室ブロックが見つかりません") - text = text[m.start():] + text = text[m.start() :] # 各研究室ブロックを取得 pattern = re.compile( - r'^研究室\s*([0-9]+)[^\n]*\n' - r'(?:[^\d\n]*?)' - r'([0-9]+)\s*件の回答\s*\n' - r'(.*?)' - r'(?=' - r'^研究室\s*[0-9]+[^\n]*\n' - r'(?:[^\d\n]*?)' - r'[0-9]+\s*件の回答' - r'|\Z' - r')', - flags=re.MULTILINE | re.DOTALL + r"^研究室\s*([0-9]+)[^\n]*\n" + r"(?:[^\d\n]*?)" + r"([0-9]+)\s*件の回答\s*\n" + r"(.*?)" + r"(?=" + r"^研究室\s*[0-9]+[^\n]*\n" + r"(?:[^\d\n]*?)" + r"[0-9]+\s*件の回答" + r"|\Z" + r")", + flags=re.MULTILINE | re.DOTALL, ) rows = [] @@ -136,16 +133,10 @@ def parse_google_form_results(text): values = extract_values_from_body(body) if len(values) != n_answers: - print( - f"警告: 研究室{lab_no}: " - f"回答数={n_answers}, 抽出数={len(values)}" - ) + print(f"警告: 研究室{lab_no}: 回答数={n_answers}, 抽出数={len(values)}") for value in values: - rows.append({ - "研究室番号": lab_no, - "値": value - }) + rows.append({"研究室番号": lab_no, "値": value}) df = pd.DataFrame(rows) @@ -166,13 +157,15 @@ def summarize_by_lab(df): # 編入生は 0 として入れてある。 df_regular = df_lab.loc[df_lab["値"] > 10] - rows.append({ - "研究室": lab_no, - "在校生": df_regular.shape[0], - "編入生": df_lab.shape[0] - df_regular.shape[0], - "GP mean": df_regular["値"].mean(), - "GP std": df_regular["値"].std(), - }) + rows.append( + { + "研究室": lab_no, + "在校生": df_regular.shape[0], + "編入生": df_lab.shape[0] - df_regular.shape[0], + "GP mean": df_regular["値"].mean(), + "GP std": df_regular["値"].std(), + } + ) return pd.DataFrame(rows).set_index("研究室") @@ -180,13 +173,7 @@ def summarize_by_lab(df): def print_summary(df, df_r): print("-" * 52) print("回答者数:", df.shape[0]) - print( - f"{'研究室':>6} " - f"{'在校生':>6} " - f"{'編入生':>6} " - f"{'GP mean':>10} " - f"{'GP std':>10}" - ) + print(f"{'研究室':>6} {'在校生':>6} {'編入生':>6} {'GP mean':>10} {'GP std':>10}") print("-" * 52) for n, row in df_r.iterrows(): @@ -202,14 +189,300 @@ def print_summary(df, df_r): ) +def apply_plot_style(): + """ + 論文図として使いやすい matplotlib の描画設定を適用する。 + + 仕様: + - seaborn や scienceplots には依存しない。 + - タイトルは付けず、軸ラベルと目盛だけで情報を伝える。 + - フォントサイズ、線幅、余白、保存時 DPI を論文図向けに調整する。 + - カラーマップは既定で tab10 を使う。 + """ + + import matplotlib.pyplot as plt + + plt.rcParams.update( + { + "figure.dpi": 120, + "savefig.dpi": 300, + "savefig.bbox": "tight", + "font.family": "sans-serif", + "font.sans-serif": ["Arial", "Helvetica", "DejaVu Sans"], + "font.size": 11, + "axes.labelsize": 12, + "axes.linewidth": 1.0, + "axes.spines.top": False, + "axes.spines.right": False, + "xtick.labelsize": 10, + "ytick.labelsize": 10, + "xtick.direction": "out", + "ytick.direction": "out", + "xtick.major.size": 4, + "ytick.major.size": 4, + "xtick.major.width": 1.0, + "ytick.major.width": 1.0, + "legend.frameon": False, + "pdf.fonttype": 42, + "ps.fonttype": 42, + } + ) + + +def filter_plot_data(data, lower=10, upper=None): + """ + GP の描画対象データを抽出する。 + + Parameters + ---------- + data : pandas.DataFrame + 「研究室番号」と「値」列を持つデータフレーム。 + lower : float, default 10 + この値より大きい回答を描画対象にする。既定値では編入生を除く。 + upper : float or None, default None + 指定した場合、この値未満の回答だけを描画対象にする。 + + Returns + ------- + pandas.DataFrame + フィルタ後のデータ。元データは変更しない。 + + Raises + ------ + KeyError + 必須列が存在しない場合。 + ValueError + フィルタ後に描画対象データが残らない場合。 + """ + + required_columns = {"研究室番号", "値"} + missing_columns = required_columns - set(data.columns) + if missing_columns: + raise KeyError(f"必須列がありません: {', '.join(sorted(missing_columns))}") + + df = data.loc[data["値"] > lower, ["研究室番号", "値"]].copy() + if upper is not None: + df = df.loc[df["値"] < upper].copy() + + if df.empty: + raise ValueError("描画対象のデータがありません。lower/upper を確認してください。") + + return df + + +def make_histogram_bins(values, bin_width): + """ + GP ヒストグラム用のビン境界を作成する。 + + 仕様: + - 最小値側は bin_width の倍数に切り下げる。 + - 最大値側は bin_width の倍数に切り上げ、最大値が最後のビンに入るようにする。 + - bin_width は正の数でなければならない。 + """ + + if bin_width <= 0: + raise ValueError("bin_width は正の数を指定してください。") + + min_edge = np.floor(values.min() / bin_width) * bin_width + max_edge = np.ceil(values.max() / bin_width) * bin_width + bin_width + return np.arange(min_edge, max_edge, bin_width) + + +def style_axis(ax, grid_axis="y"): + """ + 軸まわりの体裁を統一する。 + + 仕様: + - 上枠と右枠を消す。 + - 目盛を外向きにする。 + - 指定軸方向に薄いグリッドを入れ、文字やデータ点と競合しない濃度にする。 + """ + + ax.spines["top"].set_visible(False) + ax.spines["right"].set_visible(False) + ax.tick_params(direction="out", length=4, width=1) + ax.grid(axis=grid_axis, color="0.88", linewidth=0.8) + ax.set_axisbelow(True) + + +def draw_lab_boxplot(ax, df, labs, color, rng): + """ + 研究室別 GP 分布を箱ひげ図と実測点で描画する。 + + 仕様: + - 箱ひげ図は外れ値を非表示にし、実測点を重ねて全回答の分布を見せる。 + - 実測点には固定乱数の jitter を与え、同じ値の点の重なりを避ける。 + - x 軸のカテゴリ順は labs の順序に従う。 + """ + + positions = np.arange(1, len(labs) + 1) + grouped_values = [ + df.loc[df["研究室番号"] == lab, "値"].to_numpy(dtype=float) for lab in labs + ] + + box = ax.boxplot( + grouped_values, + positions=positions, + widths=0.55, + patch_artist=True, + showfliers=False, + medianprops={"color": "black", "linewidth": 1.2}, + boxprops={"facecolor": color, "edgecolor": "black", "linewidth": 1.0}, + whiskerprops={"color": "black", "linewidth": 1.0}, + capprops={"color": "black", "linewidth": 1.0}, + ) + + for patch in box["boxes"]: + patch.set_alpha(0.55) + + for x_position, values in zip(positions, grouped_values): + jitter = rng.uniform(-0.16, 0.16, size=len(values)) + ax.scatter( + np.full(len(values), x_position) + jitter, + values, + s=18, + color="black", + alpha=0.45, + linewidths=0, + zorder=3, + ) + + ax.set_xlim(0.4, len(labs) + 0.6) + ax.set_xticks(positions) + ax.set_xticklabels([str(lab) for lab in labs]) + ax.set_xlabel("Laboratory") + ax.set_ylabel("GP") + + +def annotate_target_values(ax, target_values, color): + """ + ヒストグラム上に対象研究室の GP を縦線と数値で示す。 + + 仕様: + - 同じ GP 値は 1 本の線にまとめる。 + - ラベルは軸上端の内側に置き、棒や軸ラベルとの重なりを避ける。 + - 近い値が複数ある場合はラベル高さを段階的にずらす。 + """ + + unique_values = pd.Series(target_values).value_counts().sort_index() + for i, (value, count) in enumerate(unique_values.items()): + ax.axvline(value, color=color, lw=1.3, alpha=0.85, linestyle="--") + label = f"{value:g}" if count == 1 else f"{value:g} x{count}" + ax.text( + value, + 0.96 - 0.11 * (i % 3), + label, + transform=ax.get_xaxis_transform(), + ha="center", + va="top", + rotation=90, + color=color, + fontsize=9, + bbox={"facecolor": "white", "edgecolor": "none", "alpha": 0.75, "pad": 1.2}, + ) + + +def draw_gp_histogram(ax, df, target_values, bin_width, color, target_color): + """ + 全研究室の GP ヒストグラムを描画する。 + + 仕様: + - bin_width ごとの頻度を表示する。 + - 対象研究室の GP は点ではなく縦線で示し、全体分布内の位置を読みやすくする。 + - 研究室別箱ひげ図と同じ主色を使い、対象研究室は tab10 の別色で強調する。 + """ + + bins = make_histogram_bins(df["値"], bin_width) + ax.hist( + df["値"], + bins=bins, + color=color, + alpha=0.62, + edgecolor="black", + linewidth=0.8, + ) + + if len(target_values) > 0: + annotate_target_values(ax, target_values, target_color) + + ax.set_xlabel("GP") + ax.set_ylabel("Frequency") + + +def plot_data(data, lower=10, upper=None, target_lab=10, bin_width=10): + """ + 研究室配属希望調査の GP 分布を描画する。 + + Parameters + ---------- + data : pandas.DataFrame + 「研究室番号」と「値」列を持つデータフレーム。 + lower : float, default 10 + 描画対象に含める GP の下限。値が lower より大きい回答だけを描く。 + upper : float or None, default None + 描画対象に含める GP の上限。None の場合は上限を設けない。 + target_lab : int, default 10 + ヒストグラム上で GP 値を縦線表示する研究室番号。 + bin_width : float, default 10 + ヒストグラムのビン幅。 + Returns + ------- + tuple[matplotlib.figure.Figure, numpy.ndarray] + 生成した Figure と 2 個の Axes。 + + 仕様: + - matplotlib と pandas/numpy のみで描画する。 + - 図タイトルは付けない。 + - 上段に研究室別の箱ひげ図と個別点、下段に全体 GP ヒストグラムを描く。 + - 編入生は既定で除外するため、内部値 0 など lower 以下の値は描画しない。 + - 色は基本的に tab10 を使用する。 + """ + + import matplotlib.pyplot as plt + + df = data.copy() + + df = filter_plot_data(df, lower=lower, upper=upper) + labs = sorted(df["研究室番号"].unique()) + target_values = df.loc[df["研究室番号"] == target_lab, "値"].to_numpy(dtype=float) + + apply_plot_style() + colors = plt.get_cmap("tab10").colors + rng = np.random.default_rng(20260709) + + fig, axs = plt.subplots( + 2, + 1, + figsize=(8.0, 6.2), + gridspec_kw={"height_ratios": [1.1, 1]}, + constrained_layout=True, + ) + + draw_lab_boxplot(axs[0], df, labs, colors[0], rng) + draw_gp_histogram(axs[1], df, target_values, bin_width, colors[0], colors[3]) + + for ax in axs: + style_axis(ax) + + plt.show() + return fig, axs + def main(): parser = argparse.ArgumentParser( description="Google Forms のコピペ結果から研究室別GPを集計する" ) - parser.add_argument("googleform_str", help="Google Forms からコピーしたテキストファイル") parser.add_argument( - "-o", "--output", - help="Excel 出力ファイル名。指定しない場合は出力しない" + "googleform_str", help="Google Forms からコピーしたテキストファイル" + ) + parser.add_argument( + "-o", "--output", help="Excel 出力ファイル名。指定しない場合は出力しない" + ) + + parser.add_argument( + "-p", + "--plot", + action="store_true", + help="各種plotを作成するかどうか。指定しない場合は作成しない。", ) args = parser.parse_args() @@ -218,10 +491,15 @@ def main(): text = f.read() df = parse_google_form_results(text) - df_r = summarize_by_lab(df) + df_r = summarize_by_lab(df).sort_values(by="GP mean", ascending=False) print_summary(df, df_r) + if args.plot: + fig, axs = plot_data(df) + fig.savefig(args.googleform_str.replace(".txt", ".png"), dpi=300) + print(args.googleform_str.replace(".txt", ".png"), "was created.") + if args.output: with pd.ExcelWriter(args.output) as writer: df.to_excel(writer, sheet_name="raw")