.
import matplotlib.pyplot as plt
plt.plot(df["X"], df["Y"], marker='o', color='blue')
plt.title("Line Plot")
plt.xlabel("X")
plt.ylabel("Y")
plt.show()
plt.scatter(df["X"], df["Y"], color='red')
plt.title("Scatter Plot")
plt.xlabel("X")
plt.ylabel("Y")
plt.show()
plt.bar(df["Category"], df["Value"], color='green')
plt.title("Bar Plot")
plt.xlabel("Category")
plt.ylabel("Value")
plt.show()
plt.hist(df["Value"], bins=10, color='purple', edgecolor='black')
plt.title("Histogram")
plt.xlabel("Value")
plt.ylabel("Frequency")
plt.show()
.
3. ترسیم نمودار
ترسیم نمودار خطی
import matplotlib.pyplot as plt
plt.plot(df["X"], df["Y"], marker='o', color='blue')
plt.title("Line Plot")
plt.xlabel("X")
plt.ylabel("Y")
plt.show()
ترسیم نمودار پراکندگی
plt.scatter(df["X"], df["Y"], color='red')
plt.title("Scatter Plot")
plt.xlabel("X")
plt.ylabel("Y")
plt.show()
ترسیم نمودار میلهای
plt.bar(df["Category"], df["Value"], color='green')
plt.title("Bar Plot")
plt.xlabel("Category")
plt.ylabel("Value")
plt.show()
ترسیم نمودار هیستوگرام
plt.hist(df["Value"], bins=10, color='purple', edgecolor='black')
plt.title("Histogram")
plt.xlabel("Value")
plt.ylabel("Frequency")
plt.show()
.
.
داشتن کد هر آزمون به شکل آماده و به روز کردن آن طی زمان و نیاز در تحلیل آماری با پایتون بسیار کمک کننده است.
from scipy import stats
t_stat, p = stats.ttest_rel(df["Before"], df["After"])
print("t =", t_stat, "p =", p)
from scipy import stats
t_stat, p = stats.ttest_ind(df["Group1"], df["Group2"])
print("t =", t_stat, "p =", p)
from scipy import stats
u_stat, p = stats.mannwhitneyu(df["Group1"], df["Group2"], alternative="two-sided")
print("U-statistic:", u_stat, " P-value:", p)
from scipy.stats import chi2_contingency
contingency = pd.crosstab(df["Gender"], df["Disease"])
chi2, p, dof, exp = chi2_contingency(contingency)
print("Chi2:", chi2, "P-value:", p, "Degrees of freedom:", dof)
from scipy.stats import fisher_exact
table = [[10, 5], [3, 12]]
odds, p = fisher_exact(table)
print("Odds Ratio:", odds, "P-value:", p)
from scipy import stats
f_stat, p = stats.f_oneway(df["Group1"], df["Group2"], df["Group3"])
print("F-statistic:", f_stat, "P-value:", p)
from scipy.stats import kruskal
h_stat, p = kruskal(df["Group1"], df["Group2"], df["Group3"])
print(f"H-statistic = {h_stat}, p-value = {p}")
import pingouin as pg
pg.rm_anova(dv='Score', within='Time', subject='ID', data=df, detailed=True)
from scipy.stats import pearsonr
r, p = pearsonr(df["X"], df["Y"])
print(r, p)
from scipy.stats import spearmanr
r, p = spearmanr(df["X"], df["Y"])
print(r, p)
import statsmodels.api as sm
X = sm.add_constant(df["X"])
model = sm.OLS(df["Y"], X).fit()
print(model.summary())
X = sm.add_constant(df[["X1", "X2"]])
model = sm.OLS(df["Y"], X).fit()
print(model.summary())
from sklearn.decomposition import PCA
from sklearn.preprocessing import StandardScaler
X = StandardScaler().fit_transform(df)
pca = PCA(n_components=2)
components = pca.fit_transform(X)
.
2. اجرای آزمون
داشتن کد هر آزمون به شکل آماده و به روز کردن آن طی زمان و نیاز در تحلیل آماری با پایتون بسیار کمک کننده است.
# Paired t-test
from scipy import stats
t_stat, p = stats.ttest_rel(df["Before"], df["After"])
print("t =", t_stat, "p =", p)
# Unpaired t-Test
from scipy import stats
t_stat, p = stats.ttest_ind(df["Group1"], df["Group2"])
print("t =", t_stat, "p =", p)
# Mann-Whitney U Test
from scipy import stats
u_stat, p = stats.mannwhitneyu(df["Group1"], df["Group2"], alternative="two-sided")
print("U-statistic:", u_stat, " P-value:", p)
# Chi-Square Test
from scipy.stats import chi2_contingency
contingency = pd.crosstab(df["Gender"], df["Disease"])
chi2, p, dof, exp = chi2_contingency(contingency)
print("Chi2:", chi2, "P-value:", p, "Degrees of freedom:", dof)
# Fisher’s Exact Test
from scipy.stats import fisher_exact
table = [[10, 5], [3, 12]]
odds, p = fisher_exact(table)
print("Odds Ratio:", odds, "P-value:", p)
# One-way ANOVA
from scipy import stats
f_stat, p = stats.f_oneway(df["Group1"], df["Group2"], df["Group3"])
print("F-statistic:", f_stat, "P-value:", p)
# Kruskal-Wallis Test
from scipy.stats import kruskal
h_stat, p = kruskal(df["Group1"], df["Group2"], df["Group3"])
print(f"H-statistic = {h_stat}, p-value = {p}")
# Repeated Measures ANOVA
import pingouin as pg
pg.rm_anova(dv='Score', within='Time', subject='ID', data=df, detailed=True)
# Pearson Correlation
from scipy.stats import pearsonr
r, p = pearsonr(df["X"], df["Y"])
print(r, p)
# Spearman Correlation
from scipy.stats import spearmanr
r, p = spearmanr(df["X"], df["Y"])
print(r, p)
# Simple Linear Regression
import statsmodels.api as sm
X = sm.add_constant(df["X"])
model = sm.OLS(df["Y"], X).fit()
print(model.summary())
# Multiple Linear Regression
X = sm.add_constant(df[["X1", "X2"]])
model = sm.OLS(df["Y"], X).fit()
print(model.summary())
# Principal Component Analysis (PCA)
from sklearn.decomposition import PCA
from sklearn.preprocessing import StandardScaler
X = StandardScaler().fit_transform(df)
pca = PCA(n_components=2)
components = pca.fit_transform(X)
.
PCA.xlsx
10.9 KB
Principal component analysis (PCA)
آنالیز مؤلفههای اصلی (PCA) روشی برای خلاصهکردن دادههای زیاد به چند عامل اصلی است که بیشترین تفاوت یا تغییر را در دادهها نشان میدهند.
با کد زیر و نمونه داده در اکسل پیوست میتوانید این آنالیز را انجام دهید.
دقت کنید که با توجه به فایل، خط کد زیر را در کد اصلاح نمایید:
excel_path = r"C:\Users\Office\Desktop\PCA.xlsx"
نتیجه دقیقاً در پوشهای ذخیره میشوند که در ابتدای کد تعریف کردهاید.
همان مسیری که اسکریپت پایتون شما اجرا شده است + پوشهٔ pca_outputs
توضیح کامل این آنالیز در کانال آنالیزهای آماری قرار دارد.
@PALDataanalysis
.
import os
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
from matplotlib.patches import Ellipse
from sklearn.preprocessing import StandardScaler
from sklearn.decomposition import PCA
from scipy.spatial import ConvexHull
# 1) File paths and column definitions
excel_path = r"C:\Users\Office\Desktop\PCA.xlsx"
sheet_name = 0
id_col = "Subject ID"
group_col = "Group"
feature_cols = ["Variable 1", "Variable 2", "Variable 3", "Variable 4",
"Variable 5", "Variable 6", "Variable 7", "Variable 8"]
group_colors = {"T1": "blue", "T2": "red"}
out_dir = "pca_outputs"
os.makedirs(out_dir, exist_ok=True)
# 2) Read data
df = pd.read_excel(excel_path, sheet_name=sheet_name)
if id_col not in df.columns:
raise ValueError(f"ID column '{id_col}' not found.")
if group_col not in df.columns:
raise ValueError(f"Group column '{group_col}' not found.")
mask = df[feature_cols].notna().all(axis=1) & df[group_col].notna()
df_clean = df.loc[mask].copy()
# 3) Standardize data
X_raw = df_clean[feature_cols].to_numpy(float)
scaler = StandardScaler()
X_z = scaler.fit_transform(X_raw)
# 4) PCA computation
pca = PCA(n_components=2)
scores = pca.fit_transform(X_z)
loadings = pca.components_.T
explained = pca.explained_variance_ratio_
df_clean["PC1"] = scores[:, 0]
df_clean["PC2"] = scores[:, 1]
# 5) Draw convex hull polygons
def draw_convex_hull(ax, points, color, alpha=0.15):
if len(points) < 3:
return
hull = ConvexHull(points)
hull_points = points[hull.vertices]
ax.fill(hull_points[:, 0], hull_points[:, 1],
color=color, alpha=alpha)
ax.plot(hull_points[:, 0], hull_points[:, 1],
color=color, linewidth=2)
# 6) PCA score plot with convex hulls
plt.figure(figsize=(7, 6))
ax = plt.gca()
ax.axhline(0, linestyle="--", color="lightgray")
ax.axvline(0, linestyle="--", color="lightgray")
for grp in df_clean[group_col].unique():
subset = df_clean[df_clean[group_col] == grp]
color = group_colors.get(grp, "black")
ax.scatter(subset["PC1"], subset["PC2"],
color=color, label=grp, alpha=0.9)
pts = subset[["PC1", "PC2"]].to_numpy()
draw_convex_hull(ax, pts, color)
# Add subject labels
for _, row in df_clean.iterrows():
ax.text(row["PC1"], row["PC2"] + 0.04,
str(row[id_col]), fontsize=8)
ax.set_xlabel(f"PC1 ({explained[0]*100:.1f}%)")
ax.set_ylabel(f"PC2 ({explained[1]*100:.1f}%)")
ax.set_title("PCA Score Plot (Convex Hull Groups)")
ax.legend()
plt.tight_layout()
plt.savefig(os.path.join(out_dir, "PCA_score_hull.jpg"), dpi=300)
plt.savefig(os.path.join(out_dir, "PCA_score_hull.svg"))
plt.show()
# 7) PCA Loading plot (PC1 vs PC2)
plt.figure(figsize=(7, 6))
pc1_load = loadings[:, 0]
pc2_load = loadings[:, 1]
plt.axhline(0, linestyle="--", color="lightgray")
plt.axvline(0, linestyle="--", color="lightgray")
plt.scatter(pc1_load, pc2_load, s=80)
for i, feat in enumerate(feature_cols):
plt.text(pc1_load[i], pc2_load[i] + 0.02, feat,
fontsize=9, ha="center")
plt.xlabel("PC1 Loading")
plt.ylabel("PC2 Loading")
plt.title("PCA Loading Plot (PC1 vs PC2)")
plt.tight_layout()
plt.savefig(os.path.join(out_dir, "PCA_loadings.jpg"), dpi=300)
plt.savefig(os.path.join(out_dir, "PCA_loadings.svg"))
plt.show()
# 8) Explained variance plot
plt.figure(figsize=(7, 4))
x = np.arange(1, 3)
plt.bar(x, explained * 100)
for i, v in enumerate(explained * 100):
plt.text(i + 1, v + 0.5, f"{v:.1f}%",
ha="center", fontsize=10)
plt.ylim(0, (explained * 100).max() + 10)
plt.xticks([1, 2], ["PC1", "PC2"])
plt.xlabel("Principal Component")
plt.ylabel("Explained Variance (%)")
plt.title("Explained Variance of PCA Components")
plt.tight_layout()
plt.savefig(os.path.join(out_dir, "PCA_variance.jpg"), dpi=300)
plt.savefig(os.path.join(out_dir, "PCA_variance.jpg"))
plt.show()
print("\n✔️ All PCA plots generated and saved in:", os.path.abspath(out_dir))
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
from matplotlib.patches import Ellipse
from sklearn.preprocessing import StandardScaler
from sklearn.decomposition import PCA
from scipy.spatial import ConvexHull
# 1) File paths and column definitions
excel_path = r"C:\Users\Office\Desktop\PCA.xlsx"
sheet_name = 0
id_col = "Subject ID"
group_col = "Group"
feature_cols = ["Variable 1", "Variable 2", "Variable 3", "Variable 4",
"Variable 5", "Variable 6", "Variable 7", "Variable 8"]
group_colors = {"T1": "blue", "T2": "red"}
out_dir = "pca_outputs"
os.makedirs(out_dir, exist_ok=True)
# 2) Read data
df = pd.read_excel(excel_path, sheet_name=sheet_name)
if id_col not in df.columns:
raise ValueError(f"ID column '{id_col}' not found.")
if group_col not in df.columns:
raise ValueError(f"Group column '{group_col}' not found.")
mask = df[feature_cols].notna().all(axis=1) & df[group_col].notna()
df_clean = df.loc[mask].copy()
# 3) Standardize data
X_raw = df_clean[feature_cols].to_numpy(float)
scaler = StandardScaler()
X_z = scaler.fit_transform(X_raw)
# 4) PCA computation
pca = PCA(n_components=2)
scores = pca.fit_transform(X_z)
loadings = pca.components_.T
explained = pca.explained_variance_ratio_
df_clean["PC1"] = scores[:, 0]
df_clean["PC2"] = scores[:, 1]
# 5) Draw convex hull polygons
def draw_convex_hull(ax, points, color, alpha=0.15):
if len(points) < 3:
return
hull = ConvexHull(points)
hull_points = points[hull.vertices]
ax.fill(hull_points[:, 0], hull_points[:, 1],
color=color, alpha=alpha)
ax.plot(hull_points[:, 0], hull_points[:, 1],
color=color, linewidth=2)
# 6) PCA score plot with convex hulls
plt.figure(figsize=(7, 6))
ax = plt.gca()
ax.axhline(0, linestyle="--", color="lightgray")
ax.axvline(0, linestyle="--", color="lightgray")
for grp in df_clean[group_col].unique():
subset = df_clean[df_clean[group_col] == grp]
color = group_colors.get(grp, "black")
ax.scatter(subset["PC1"], subset["PC2"],
color=color, label=grp, alpha=0.9)
pts = subset[["PC1", "PC2"]].to_numpy()
draw_convex_hull(ax, pts, color)
# Add subject labels
for _, row in df_clean.iterrows():
ax.text(row["PC1"], row["PC2"] + 0.04,
str(row[id_col]), fontsize=8)
ax.set_xlabel(f"PC1 ({explained[0]*100:.1f}%)")
ax.set_ylabel(f"PC2 ({explained[1]*100:.1f}%)")
ax.set_title("PCA Score Plot (Convex Hull Groups)")
ax.legend()
plt.tight_layout()
plt.savefig(os.path.join(out_dir, "PCA_score_hull.jpg"), dpi=300)
plt.savefig(os.path.join(out_dir, "PCA_score_hull.svg"))
plt.show()
# 7) PCA Loading plot (PC1 vs PC2)
plt.figure(figsize=(7, 6))
pc1_load = loadings[:, 0]
pc2_load = loadings[:, 1]
plt.axhline(0, linestyle="--", color="lightgray")
plt.axvline(0, linestyle="--", color="lightgray")
plt.scatter(pc1_load, pc2_load, s=80)
for i, feat in enumerate(feature_cols):
plt.text(pc1_load[i], pc2_load[i] + 0.02, feat,
fontsize=9, ha="center")
plt.xlabel("PC1 Loading")
plt.ylabel("PC2 Loading")
plt.title("PCA Loading Plot (PC1 vs PC2)")
plt.tight_layout()
plt.savefig(os.path.join(out_dir, "PCA_loadings.jpg"), dpi=300)
plt.savefig(os.path.join(out_dir, "PCA_loadings.svg"))
plt.show()
# 8) Explained variance plot
plt.figure(figsize=(7, 4))
x = np.arange(1, 3)
plt.bar(x, explained * 100)
for i, v in enumerate(explained * 100):
plt.text(i + 1, v + 0.5, f"{v:.1f}%",
ha="center", fontsize=10)
plt.ylim(0, (explained * 100).max() + 10)
plt.xticks([1, 2], ["PC1", "PC2"])
plt.xlabel("Principal Component")
plt.ylabel("Explained Variance (%)")
plt.title("Explained Variance of PCA Components")
plt.tight_layout()
plt.savefig(os.path.join(out_dir, "PCA_variance.jpg"), dpi=300)
plt.savefig(os.path.join(out_dir, "PCA_variance.jpg"))
plt.show()
print("\n✔️ All PCA plots generated and saved in:", os.path.abspath(out_dir))
.
در کنار کانال آنالیز آماری (زیرمجموعه همین تیم) مواردی که تا اینجا مطرح شد ابزار جدیدی برای تحلیلهای آماری در اختیارمان قرار داد.
در تحلیل آماری با پایتون، چون که یک بخش آن ترسیم گراف است، نیاز است به موضوع مدیریت رنگ در پایتون بپردازیم.
.
در کنار کانال آنالیز آماری (زیرمجموعه همین تیم) مواردی که تا اینجا مطرح شد ابزار جدیدی برای تحلیلهای آماری در اختیارمان قرار داد.
مطمئنا اگر بدانیم قرار است چه نوع تحلیل و مسیر آماری را برای دادهخای خود پیاده کنیم، به راحتی امکان استفاده از پایتون را در این مسیر داریم.
در تحلیل آماری با پایتون، چون که یک بخش آن ترسیم گراف است، نیاز است به موضوع مدیریت رنگ در پایتون بپردازیم.
.
.
در Matplotlib، برای پارامتر color میتوان از نامهای استاندارد رنگ استفاده کرد.
این رنگها از نامهای CSS گرفته شدهاند.
.
روش تعریف رنگ در پایتون (Matplotlib)
در Matplotlib، برای پارامتر color میتوان از نامهای استاندارد رنگ استفاده کرد.
این رنگها از نامهای CSS گرفته شدهاند.
.
.
1. نام ساده رنگها (CSS Names)
2. کد HEX یا هگزادسیمال برای تعریف دقیق رنگ
3. کد RGB یا RGBA با مقدار بین صفر و یک
4. حروف تکحرفی (سریع ولی محدود)
5. استفاده ازColormapها (پالت رنگ آماده) برای دادههای پیوسته
@Python_Education_Azizi
.
روشهای استفاده از رنگ در پایتون (Matplotlib)
1. نام ساده رنگها (CSS Names)
2. کد HEX یا هگزادسیمال برای تعریف دقیق رنگ
3. کد RGB یا RGBA با مقدار بین صفر و یک
4. حروف تکحرفی (سریع ولی محدود)
5. استفاده ازColormapها (پالت رنگ آماده) برای دادههای پیوسته
@Python_Education_Azizi
.
.
مانند
"red", "blue", "green", "yellow", "purple", "orange", "black", "white", "pink", "brown", "cyan", "magenta", "gray", "gold", "silver" and …
.
1. نام ساده رنگها (CSS Names)
مانند
"red", "blue", "green", "yellow", "purple", "orange", "black", "white", "pink", "brown", "cyan", "magenta", "gray", "gold", "silver" and …
.
.
مانند
color=(0.2, 0.4, 0.6) # RGB
color=(0.2, 0.4, 0.6, 0.8) # RGBA (با شفافیت)
.
3. کد RGB یا RGBA با مقدار بین صفر و یک
مانند
color=(0.2, 0.4, 0.6) # RGB
color=(0.2, 0.4, 0.6, 0.8) # RGBA (با شفافیت)
.
.
معروفترینها:
"viridis", "plasma", "inferno", "magma", "cividis", "coolwarm", "rainbow", "jet", "spring", "summer", "autumn", "winter"
.
5. استفاده ازColormapها (پالت رنگ آماده) برای دادههای پیوسته
معروفترینها:
"viridis", "plasma", "inferno", "magma", "cividis", "coolwarm", "rainbow", "jet", "spring", "summer", "autumn", "winter"
.
.
import matplotlib.pyplot as plt
import numpy as np
cmap = plt.cm.viridis # 'plasma', 'coolwarm', 'tab10', ...
x = np.linspace(0, 10, 100)
y = np.sin(x)
colors = cmap(np.linspace(0, 1, len(x)))
plt.figure(figsize=(8, 4))
plt.scatter(x, y, c=colors, s=50)
plt.title("Matplotlib Colormap - Viridis")
plt.show()
.
تمرین استفاده ازColormapها
import matplotlib.pyplot as plt
import numpy as np
cmap = plt.cm.viridis # 'plasma', 'coolwarm', 'tab10', ...
x = np.linspace(0, 10, 100)
y = np.sin(x)
colors = cmap(np.linspace(0, 1, len(x)))
plt.figure(figsize=(8, 4))
plt.scatter(x, y, c=colors, s=50)
plt.title("Matplotlib Colormap - Viridis")
plt.show()
.
.
import seaborn as sns
import matplotlib.pyplot as plt
palettes = ["deep", "muted", "pastel", "bright", "dark", "colorblind",
"Set1", "Set2", "Set3", "Paired", "Accent", "Pastel1", "Pastel2", "Dark2",
"coolwarm", "RdBu_r", "Spectral"]
plt.figure(figsize=(10, 8))
for i, palette in enumerate(palettes):
colors = sns.color_palette(palette, 8)
plt.subplot(len(palettes)//3+1, 3, i+1)
sns.palplot(colors)
plt.title(palette, fontsize=8)
plt.tight_layout()
plt.show()
.
نمایش همه این پالتها با کد
لطفا اجرا کنید.
import seaborn as sns
import matplotlib.pyplot as plt
palettes = ["deep", "muted", "pastel", "bright", "dark", "colorblind",
"Set1", "Set2", "Set3", "Paired", "Accent", "Pastel1", "Pastel2", "Dark2",
"coolwarm", "RdBu_r", "Spectral"]
plt.figure(figsize=(10, 8))
for i, palette in enumerate(palettes):
colors = sns.color_palette(palette, 8)
plt.subplot(len(palettes)//3+1, 3, i+1)
sns.palplot(colors)
plt.title(palette, fontsize=8)
plt.tight_layout()
plt.show()
.
.
import matplotlib.pyplot as plt
import numpy as np
cmaps = plt.colormaps()
gradient = np.linspace(0, 1, 256).reshape(1, -1)
fig, axs = plt.subplots(len(cmaps)//5, 5, figsize=(15, 40))
for ax, cmap_name in zip(axs.flat, cmaps):
ax.imshow(gradient, aspect='auto', cmap=plt.get_cmap(cmap_name))
ax.set_title(cmap_name, fontsize=8)
ax.axis('off')
plt.tight_layout()
plt.savefig("colormaps.jpeg", format="jpeg", dpi=300)
plt.show()
.
نمایش همه این پالتها با کد
لطفا اجرا کنید.
import matplotlib.pyplot as plt
import numpy as np
cmaps = plt.colormaps()
gradient = np.linspace(0, 1, 256).reshape(1, -1)
fig, axs = plt.subplots(len(cmaps)//5, 5, figsize=(15, 40))
for ax, cmap_name in zip(axs.flat, cmaps):
ax.imshow(gradient, aspect='auto', cmap=plt.get_cmap(cmap_name))
ax.set_title(cmap_name, fontsize=8)
ax.axis('off')
plt.tight_layout()
plt.savefig("colormaps.jpeg", format="jpeg", dpi=300)
plt.show()
.