""" Replacement Figure 2 for https://psychsafety.com/childhood-ses-workplace-risks/ Childhood SES vs workplace safety ratings, respondents under 50 vs 50 and over. Data: the public CSV linked from the article. Colours match Figures 1 and 3. """ import numpy as np import pandas as pd import matplotlib.pyplot as plt from scipy import stats df = pd.read_csv("Public-Copy-of-socioeconomic-background-and-interpersonal-risks.csv") df.columns = ["SES", "PS", "OPP", "Age", "Submitted", "Token"] # one row per respondent (the export includes partial saves), as in the published script a = (df.groupby("Token") .agg(SES=("SES", "mean"), PS=("PS", "mean"), OPP=("OPP", "mean"), Age=("Age", "last")) .reset_index()) a["Group"] = a["Age"].map({"20-30": "Under 50", "30-40": "Under 50", "40-50": "Under 50", "50-60": "50 and over", "60-70": "50 and over", "70 and above": "50 and over"}) ORANGE, BLUE = "#e69f00", "#56b4e9" INK, INK2, GRID = "#1a1a1a", "#555555", "#d0d0d0" styles = {"Under 50": dict(color=ORANGE, ls="-"), "50 and over": dict(color=BLUE, ls="--")} panels = [("PS", "Interpersonal psychological safety"), ("OPP", "Opportunity-taking")] plt.rcParams.update({"font.family": "DejaVu Sans", "font.size": 12, "axes.edgecolor": INK, "axes.labelcolor": INK, "xtick.color": INK, "ytick.color": INK}) fig, axes = plt.subplots(1, 2, figsize=(13, 6.2), sharey=True) xs = np.linspace(0, 6, 200) for ax, (col, title) in zip(axes, panels): for grp in ["Under 50", "50 and over"]: d = a[a["Group"] == grp][["SES", col]].dropna() x, y = d["SES"].values, d[col].values n = len(d) fit = stats.linregress(x, y) r = fit.rvalue yhat = fit.intercept + fit.slope * xs # 95% confidence band for the fitted mean resid = y - (fit.intercept + fit.slope * x) s = np.sqrt(np.sum(resid ** 2) / (n - 2)) se = s * np.sqrt(1 / n + (xs - x.mean()) ** 2 / np.sum((x - x.mean()) ** 2)) t = stats.t.ppf(0.975, n - 2) st = styles[grp] ax.fill_between(xs, yhat - t * se, yhat + t * se, color=st["color"], alpha=0.16, lw=0) ax.plot(xs, yhat, color=st["color"], ls=st["ls"], lw=2.6, label=f"{grp} (n = {n}, r = {r:.2f})") # group means at each SES level with at least 5 respondents g = d.assign(S=d["SES"].round()).groupby("S")[col].agg(["mean", "count"]) g = g[g["count"] >= 5] ax.scatter(g.index, g["mean"], s=46, marker="o", color=st["color"], edgecolor="white", linewidth=1.2, zorder=3) # direct label at the right-hand end of each line ax.annotate(grp, xy=(6, yhat[-1]), xytext=(8, 0), textcoords="offset points", va="center", ha="left", fontsize=11, color=INK) ax.set_title(title, fontsize=13, color=INK, loc="left", pad=10) ax.set_xlim(0, 6) ax.set_xticks(range(7)) ax.set_xlabel("Childhood SES (0–6)") ax.grid(True, ls="--", lw=0.8, color=GRID) ax.set_axisbelow(True) for side in ["top", "right"]: ax.spines[side].set_visible(False) ax.legend(loc="upper left", frameon=True, framealpha=0.95, edgecolor=GRID, fontsize=10.5) axes[0].set_ylabel("Safety rating (0–6)") axes[0].set_ylim(2.0, 5.6) fig.text(0.985, 0.02, "psychsafety.com", ha="right", va="bottom", fontsize=11, color=INK) fig.text(0.015, 0.02, "Lines: linear fit with 95% confidence band. Dots: average rating at each SES level (5+ respondents).", ha="left", va="bottom", fontsize=9.5, color=INK2) fig.tight_layout(rect=(0, 0.05, 0.97, 1)) fig.subplots_adjust(wspace=0.34, right=0.9) fig.savefig("Figure-2-SES-and-safety-by-age-group.png", dpi=170, facecolor="white") print("saved")