Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 5d71524f81 | |||
| d32476ef94 |
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,8 @@
|
|||||||
|
,Exercises,Culture,Beauty,Education,Society,Spirit
|
||||||
|
Park,0.6367377065045589,0.5386259837748125,0.609527381412311,0.6666205318079247,0.5894275000421331,0.5971978092460599
|
||||||
|
Residential,0.6898067312223374,0.5322888366243355,0.7043563888714474,0.7003848393051737,0.6356124898043054,0.671536167178851
|
||||||
|
Garden,0.569207243129314,0.42043962025354226,0.520982025379601,0.5745723285479705,0.47015163689224226,0.4720301687679101
|
||||||
|
Rooftop,0.5220196490736146,0.39671027279755816,0.559846265287038,0.5289815361048281,0.46041084713416613,0.5313786451855366
|
||||||
|
Recreation,0.4416516078536275,0.4118096932349253,0.414973028889102,0.48397115121605433,0.347143218999983,0.3437593248955575
|
||||||
|
Agriculture,-0.30525120464643385,-0.09456459047576876,-0.41726188321313135,-0.3345389043778018,-0.3341241701147672,-0.37637058454953964
|
||||||
|
Nature,-0.44980612831551553,-0.27299890014702094,-0.49976092665649713,-0.4610579352258267,-0.4605091031334855,-0.5381766537399497
|
||||||
|
@@ -0,0 +1,7 @@
|
|||||||
|
,0,1
|
||||||
|
Exercises,0.007027806093845654,0.004957784418992443
|
||||||
|
Culture,-0.12773486154946914,-0.001881737518742339
|
||||||
|
Beauty,0.05188265296211081,0.0051213397312073914
|
||||||
|
Education,0.020143274994230118,0.013800122537297493
|
||||||
|
Society,0.004096923788506599,-0.004799612673251565
|
||||||
|
Spirit,0.04544596636820759,-0.018212152335741184
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
,0,1
|
||||||
|
Park,0.01929915023578449,-0.0026557268484349067
|
||||||
|
Residential,0.03611771387245799,-0.006098756785044107
|
||||||
|
Garden,0.024297413495923986,0.008449977834710572
|
||||||
|
Rooftop,0.03634674358580918,-0.009949590505925508
|
||||||
|
Recreation,-0.0006198278964895602,0.015176373315937821
|
||||||
|
Agriculture,-0.14550165539847498,-0.015634536842211624
|
||||||
|
Nature,-0.14586553315107212,0.01084942463742428
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
,Dirty,Unsafe,Danger
|
||||||
|
Park,-0.40659517420422864,-0.3652691650630517,-0.47632864603181757
|
||||||
|
Residential,-0.42361014190625074,-0.37318965467898796,-0.4711066759095078
|
||||||
|
Garden,-0.3109599181970276,-0.27653098840256857,-0.34126139088328367
|
||||||
|
Rooftop,-0.313359215935556,-0.3361077310042239,-0.35612431235937536
|
||||||
|
Recreation,-0.09348749429342133,-0.14803279299655103,-0.27167796733643157
|
||||||
|
Agriculture,0.48572656432897177,0.46262486117052976,0.5273673898837783
|
||||||
|
Nature,0.5129174934902375,0.5064560749634449,0.5363089258301386
|
||||||
|
@@ -0,0 +1,4 @@
|
|||||||
|
,0,1
|
||||||
|
Dirty,-0.021750742704197926,-0.023688662634614326
|
||||||
|
Unsafe,-0.031685779157853394,0.02094725474813505
|
||||||
|
Danger,0.056173988030702275,0.002810907103117069
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
,0,1
|
||||||
|
Park,-0.05277737050209137,0.021966851260274492
|
||||||
|
Residential,-0.039828498949265316,0.030086993744197595
|
||||||
|
Garden,-0.011237261666548438,0.01787364745984657
|
||||||
|
Rooftop,0.0025503277042415763,-0.014900920261128709
|
||||||
|
Recreation,-0.05985983449204827,-0.034402423139858064
|
||||||
|
Agriculture,0.04093472746557441,-0.0031882467528099086
|
||||||
|
Nature,0.03188949642613842,0.0004223966469934847
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
,MEAN RES,MEAN CES,MEAN DES
|
||||||
|
Park,0.6110543201654319,0.6768852442644577,-0.47060044983099286
|
||||||
|
Residential,0.6407085222739133,0.7291962678143591,-0.4708562921098795
|
||||||
|
Garden,0.44165260890843555,0.5480223668648628,-0.34198141262933823
|
||||||
|
Rooftop,0.49600027293784166,0.5419994714753112,-0.3688582414111523
|
||||||
|
Recreation,0.3745893900657618,0.40850035774422844,-0.17674812102484044
|
||||||
|
Agriculture,-0.3502787611798973,-0.39573598649446984,0.5625037794065111
|
||||||
|
Nature,-0.4734036284834367,-0.5391708531031724,0.582247211793259
|
||||||
|
@@ -0,0 +1,4 @@
|
|||||||
|
,0,1
|
||||||
|
MEAN RES,-0.2186071848607324,-0.007839144868552231
|
||||||
|
MEAN CES,-0.2574749215265179,0.007317945054910379
|
||||||
|
MEAN DES,0.6682022605408963,0.00045551629187779
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
,0,1
|
||||||
|
Park,-0.28217527664101055,-0.004698104759826525
|
||||||
|
Residential,-0.2891839961693309,0.00131726592526606
|
||||||
|
Garden,-0.18810995860092022,0.013756464641007708
|
||||||
|
Rooftop,-0.20631526503069358,-0.0065124662879235936
|
||||||
|
Recreation,-0.07849003909114283,-0.0039717946663235335
|
||||||
|
Agriculture,0.6635981475555077,0.002205946282469859
|
||||||
|
Nature,0.8024031622133635,-0.002091864403477851
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
,Temperature,Noise,Stormwind,Respiratory
|
||||||
|
Park,0.5943691697194959,0.4789900955834734,0.5400353044339286,0.5612018360594374
|
||||||
|
Residential,0.6098741786551782,0.4993073835325437,0.5888125616106002,0.6297059731338867
|
||||||
|
Garden,0.45042886274863986,0.3963838785463788,0.42396144532666635,0.41516036007093543
|
||||||
|
Rooftop,0.49442453924504476,0.4358175328312086,0.4842898044085772,0.4021375963717873
|
||||||
|
Recreation,0.40664357632988,0.3565204984864607,0.39747369992230713,0.3250145065589509
|
||||||
|
Agriculture,-0.3272930077585318,-0.23403946925952315,-0.3063551523911346,-0.320369510797842
|
||||||
|
Nature,-0.44797645118533835,-0.30105163246668354,-0.4709764053605818,-0.45237173056171975
|
||||||
|
@@ -0,0 +1,5 @@
|
|||||||
|
,0,1
|
||||||
|
Temperature,-0.023338638107689796,-0.006213751498479767
|
||||||
|
Noise,0.06553553877191058,-0.0005751574284129316
|
||||||
|
Stormwind,-0.021996594532673455,-0.013022115419183451
|
||||||
|
Respiratory,-0.019898887972482068,0.02012270293988676
|
||||||
|
@@ -0,0 +1,8 @@
|
|||||||
|
,0,1
|
||||||
|
Park,-0.02240002337805916,0.007917221215969647
|
||||||
|
Residential,-0.02799189713117847,0.015859200211763763
|
||||||
|
Garden,-0.008475103510200898,0.0016829493791823964
|
||||||
|
Rooftop,-0.0060566811778039985,-0.017070154943948847
|
||||||
|
Recreation,-0.004992858046068623,-0.015594566177349648
|
||||||
|
Agriculture,0.05380057559170161,0.0006539273597548665
|
||||||
|
Nature,0.11812861315235715,0.010393415861558994
|
||||||
|
Binary file not shown.
|
After Width: | Height: | Size: 194 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 172 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 189 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 185 KiB |
@@ -0,0 +1,90 @@
|
|||||||
|
import pandas as pd
|
||||||
|
import numpy as np
|
||||||
|
import matplotlib.pyplot as plt
|
||||||
|
import prince
|
||||||
|
import os
|
||||||
|
|
||||||
|
# Đường dẫn file dữ liệu
|
||||||
|
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Data_VN_filter_v7.xlsx'
|
||||||
|
output_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\2_CA_and_Spearman'
|
||||||
|
|
||||||
|
# Load data
|
||||||
|
df = pd.read_excel(file_path)
|
||||||
|
|
||||||
|
# Extract variables
|
||||||
|
ugs_cols = ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
|
||||||
|
mean_cols = ['MEAN RES', 'MEAN CES', 'MEAN DES']
|
||||||
|
colors = ['green', 'purple', 'red']
|
||||||
|
|
||||||
|
print("Running CA for MEAN RES, MEAN CES, MEAN DES...")
|
||||||
|
|
||||||
|
# Lọc các biến có trong data
|
||||||
|
available_means = [c for c in mean_cols if c in df.columns]
|
||||||
|
if not available_means:
|
||||||
|
print("Không tìm thấy các cột MEAN trong dữ liệu.")
|
||||||
|
exit()
|
||||||
|
|
||||||
|
data = df[ugs_cols + available_means].dropna()
|
||||||
|
|
||||||
|
# Tạo ma trận tương quan Spearman
|
||||||
|
corr = data.corr(method='spearman').loc[ugs_cols, available_means]
|
||||||
|
|
||||||
|
# Dịch chuyển ma trận để tất cả các giá trị đều dương (CA yêu cầu bảng tần số hoặc giá trị dương)
|
||||||
|
corr_shifted = corr + 1
|
||||||
|
|
||||||
|
# Initialize CA
|
||||||
|
ca = prince.CA(n_components=2, n_iter=3, copy=True, check_input=True, engine='scipy', random_state=42)
|
||||||
|
ca = ca.fit(corr_shifted)
|
||||||
|
|
||||||
|
# Extract column and row coordinates
|
||||||
|
row_coords = ca.row_coordinates(corr_shifted) # UGS
|
||||||
|
col_coords = ca.column_coordinates(corr_shifted) # Mean Services
|
||||||
|
|
||||||
|
# Save to CSV
|
||||||
|
group_name = "MEAN_ESS_DES"
|
||||||
|
corr.to_csv(os.path.join(output_dir, f'CA_{group_name}_Correlation_Matrix.csv'))
|
||||||
|
row_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_UGS_Coords.csv'))
|
||||||
|
col_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_Services_Coords.csv'))
|
||||||
|
|
||||||
|
# Plot Biplot
|
||||||
|
fig, ax = plt.subplots(figsize=(10, 8))
|
||||||
|
|
||||||
|
# Plot UGS points
|
||||||
|
ax.scatter(row_coords[0], row_coords[1], c='blue', label='UGS Types', s=80, marker='o', edgecolors='black')
|
||||||
|
|
||||||
|
# Plot Mean Services points with corresponding colors
|
||||||
|
for i, col in enumerate(available_means):
|
||||||
|
c_idx = mean_cols.index(col)
|
||||||
|
color = colors[c_idx]
|
||||||
|
ax.scatter(col_coords.loc[col, 0], col_coords.loc[col, 1], c=color, marker='s', s=100, edgecolors='black')
|
||||||
|
|
||||||
|
# Labels
|
||||||
|
for i, txt in enumerate(ugs_cols):
|
||||||
|
ax.annotate(txt, (row_coords.iloc[i, 0], row_coords.iloc[i, 1]),
|
||||||
|
xytext=(-10, 10), textcoords='offset points',
|
||||||
|
color='blue', fontweight='bold', fontsize=12, ha='right', va='bottom')
|
||||||
|
|
||||||
|
for i, txt in enumerate(available_means):
|
||||||
|
c_idx = mean_cols.index(txt)
|
||||||
|
color = colors[c_idx]
|
||||||
|
ax.annotate(txt, (col_coords.loc[txt, 0], col_coords.loc[txt, 1]),
|
||||||
|
xytext=(10, -10), textcoords='offset points',
|
||||||
|
color=color, fontsize=12, fontweight='bold', ha='left', va='top')
|
||||||
|
|
||||||
|
# Add dummy plots for legend
|
||||||
|
ax.scatter([], [], c='green', marker='s', label='MEAN RES (Điều hòa)')
|
||||||
|
ax.scatter([], [], c='purple', marker='s', label='MEAN CES (Văn hóa)')
|
||||||
|
ax.scatter([], [], c='red', marker='s', label='MEAN DES (Bất lợi)')
|
||||||
|
|
||||||
|
ax.axhline(0, color='grey', linestyle='--', linewidth=1)
|
||||||
|
ax.axvline(0, color='grey', linestyle='--', linewidth=1)
|
||||||
|
ax.set_title('CA Biplot: UGS vs MEAN RES, MEAN CES, MEAN DES', fontsize=16)
|
||||||
|
ax.set_xlabel('Component 0', fontsize=12)
|
||||||
|
ax.set_ylabel('Component 1', fontsize=12)
|
||||||
|
ax.legend(loc='best', fontsize=12)
|
||||||
|
plt.grid(True, linestyle=':', alpha=0.6)
|
||||||
|
|
||||||
|
plt.savefig(os.path.join(output_dir, f'CA_plot_{group_name}.png'), dpi=300, bbox_inches='tight')
|
||||||
|
plt.close()
|
||||||
|
|
||||||
|
print("Finished MEAN ESS/DES.")
|
||||||
@@ -0,0 +1,87 @@
|
|||||||
|
import pandas as pd
|
||||||
|
import numpy as np
|
||||||
|
import matplotlib.pyplot as plt
|
||||||
|
import prince
|
||||||
|
import os
|
||||||
|
|
||||||
|
# Đường dẫn file dữ liệu
|
||||||
|
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Data_VN_filter_v7.xlsx'
|
||||||
|
output_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\2_CA_and_Spearman'
|
||||||
|
|
||||||
|
# Load data
|
||||||
|
df = pd.read_excel(file_path)
|
||||||
|
|
||||||
|
# Extract variables
|
||||||
|
ugs_cols = ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
|
||||||
|
|
||||||
|
groups = {
|
||||||
|
'RES': ['Temperature', 'Noise', 'Stormwind', 'Respiratory'],
|
||||||
|
'CES': ['Exercises', 'Culture', 'Beauty', 'Education', 'Society', 'Spirit'],
|
||||||
|
'DES': ['Dirty', 'Unsafe', 'Danger']
|
||||||
|
}
|
||||||
|
|
||||||
|
colors = {
|
||||||
|
'RES': 'green',
|
||||||
|
'CES': 'purple',
|
||||||
|
'DES': 'red'
|
||||||
|
}
|
||||||
|
|
||||||
|
for group_name, cols in groups.items():
|
||||||
|
available_cols = [c for c in cols if c in df.columns]
|
||||||
|
if not available_cols:
|
||||||
|
print(f"No columns found for {group_name}")
|
||||||
|
continue
|
||||||
|
|
||||||
|
print(f"Running CA for {group_name}...")
|
||||||
|
|
||||||
|
data = df[ugs_cols + available_cols].dropna()
|
||||||
|
corr = data.corr(method='spearman').loc[ugs_cols, available_cols]
|
||||||
|
|
||||||
|
corr_shifted = corr + 1
|
||||||
|
|
||||||
|
# Initialize CA
|
||||||
|
ca = prince.CA(n_components=2, n_iter=3, copy=True, check_input=True, engine='scipy', random_state=42)
|
||||||
|
ca = ca.fit(corr_shifted)
|
||||||
|
|
||||||
|
# Extract column and row coordinates
|
||||||
|
row_coords = ca.row_coordinates(corr_shifted) # UGS
|
||||||
|
col_coords = ca.column_coordinates(corr_shifted) # Services
|
||||||
|
|
||||||
|
# Save to CSV
|
||||||
|
corr.to_csv(os.path.join(output_dir, f'CA_{group_name}_Correlation_Matrix.csv'))
|
||||||
|
row_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_UGS_Coords.csv'))
|
||||||
|
col_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_Services_Coords.csv'))
|
||||||
|
|
||||||
|
# Plot Biplot
|
||||||
|
fig, ax = plt.subplots(figsize=(10, 8))
|
||||||
|
|
||||||
|
# Plot UGS points
|
||||||
|
ax.scatter(row_coords[0], row_coords[1], c='blue', label='UGS Types', s=80, marker='o', edgecolors='black')
|
||||||
|
|
||||||
|
# Plot Services points
|
||||||
|
color = colors[group_name]
|
||||||
|
ax.scatter(col_coords[0], col_coords[1], c=color, marker='s', s=80, edgecolors='black', label=group_name)
|
||||||
|
|
||||||
|
# Labels
|
||||||
|
for i, txt in enumerate(ugs_cols):
|
||||||
|
ax.annotate(txt, (row_coords.iloc[i, 0], row_coords.iloc[i, 1]),
|
||||||
|
xytext=(-10, 10), textcoords='offset points',
|
||||||
|
color='blue', fontweight='bold', fontsize=12, ha='right', va='bottom')
|
||||||
|
|
||||||
|
for i, txt in enumerate(available_cols):
|
||||||
|
ax.annotate(txt, (col_coords.iloc[i, 0], col_coords.iloc[i, 1]),
|
||||||
|
xytext=(10, -10), textcoords='offset points',
|
||||||
|
color=color, fontsize=12, fontweight='bold', ha='left', va='top')
|
||||||
|
|
||||||
|
ax.axhline(0, color='grey', linestyle='--', linewidth=1)
|
||||||
|
ax.axvline(0, color='grey', linestyle='--', linewidth=1)
|
||||||
|
ax.set_title(f'CA Biplot: UGS vs {group_name}', fontsize=16)
|
||||||
|
ax.set_xlabel('Component 0', fontsize=12)
|
||||||
|
ax.set_ylabel('Component 1', fontsize=12)
|
||||||
|
ax.legend(loc='best', fontsize=12)
|
||||||
|
plt.grid(True, linestyle=':', alpha=0.6)
|
||||||
|
|
||||||
|
plt.savefig(os.path.join(output_dir, f'CA_plot_{group_name}.png'), dpi=300, bbox_inches='tight')
|
||||||
|
plt.close()
|
||||||
|
|
||||||
|
print(f"Finished {group_name}.")
|
||||||
Binary file not shown.
+25
-25
@@ -1,25 +1,25 @@
|
|||||||
,Beta (B),P-value,Odds Ratio EXP(B),Significance
|
,Beta (B),S.E.,P-value,Odds Ratio EXP(B),Significance
|
||||||
const,-13.3839,0.0078,0.0,**
|
const,-13.3839,5.0267,0.0078,0.0,**
|
||||||
Frequency,0.7183,0.0098,2.051,**
|
Frequency,0.7183,0.278,0.0098,2.051,**
|
||||||
Career_6,4.1539,0.0105,63.6799,*
|
Career_6,4.1539,1.6229,0.0105,63.6799,*
|
||||||
Rooftop,1.7017,0.0123,5.4833,*
|
Rooftop,1.7017,0.6798,0.0123,5.4833,*
|
||||||
Income,0.0045,0.0229,1.0045,*
|
Income,0.0045,0.002,0.0229,1.0045,*
|
||||||
Transportation_4,4.6473,0.0284,104.3069,*
|
Transportation_4,4.6473,2.1208,0.0284,104.3069,*
|
||||||
MEAN CES,1.8488,0.0288,6.3523,*
|
MEAN CES,1.8488,0.8458,0.0288,6.3523,*
|
||||||
Career_4,2.8862,0.0334,17.9242,*
|
Career_4,2.8862,1.3569,0.0334,17.9242,*
|
||||||
Career_3,3.4267,0.0368,30.7762,*
|
Career_3,3.4267,1.6413,0.0368,30.7762,*
|
||||||
Garden,-1.4026,0.0446,0.246,*
|
Garden,-1.4026,0.6984,0.0446,0.246,*
|
||||||
MEAN DES,-0.4982,0.0586,0.6076,
|
MEAN DES,-0.4982,0.2635,0.0586,0.6076,
|
||||||
Transportation_3,2.1343,0.0739,8.4512,
|
Transportation_3,2.1343,1.1942,0.0739,8.4512,
|
||||||
Literacy,0.5519,0.0906,1.7365,
|
Literacy,0.5519,0.3261,0.0906,1.7365,
|
||||||
Gender_2,1.1137,0.1406,3.0455,
|
Gender_2,1.1137,0.7557,0.1406,3.0455,
|
||||||
Nature,-0.6135,0.2466,0.5415,
|
Nature,-0.6135,0.5294,0.2466,0.5415,
|
||||||
MEAN RES,-0.6547,0.2723,0.5196,
|
MEAN RES,-0.6547,0.5964,0.2723,0.5196,
|
||||||
Residential,0.6836,0.391,1.981,
|
Residential,0.6836,0.7969,0.391,1.981,
|
||||||
Time,-0.1722,0.5721,0.8418,
|
Time,-0.1722,0.3048,0.5721,0.8418,
|
||||||
Distance,-0.161,0.6202,0.8513,
|
Distance,-0.161,0.3249,0.6202,0.8513,
|
||||||
Transportation_2,0.6629,0.6341,1.9404,
|
Transportation_2,0.6629,1.3928,0.6341,1.9404,
|
||||||
Recreation,-0.1187,0.7927,0.8881,
|
Recreation,-0.1187,0.4515,0.7927,0.8881,
|
||||||
Career_2,0.347,0.8054,1.4148,
|
Career_2,0.347,1.4082,0.8054,1.4148,
|
||||||
Park,-0.1024,0.8656,0.9027,
|
Park,-0.1024,0.6051,0.8656,0.9027,
|
||||||
Agriculture,-0.0665,0.8894,0.9356,
|
Agriculture,-0.0665,0.4784,0.8894,0.9356,
|
||||||
|
|||||||
|
@@ -0,0 +1,75 @@
|
|||||||
|
import sys
|
||||||
|
sys.stdout.reconfigure(encoding='utf-8')
|
||||||
|
import pandas as pd
|
||||||
|
import numpy as np
|
||||||
|
import os
|
||||||
|
|
||||||
|
base_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\4_Logistic_Regression'
|
||||||
|
|
||||||
|
file_donation = os.path.join(base_dir, 'Logistic_Results_Donation_Firth.csv')
|
||||||
|
file_decision = os.path.join(base_dir, 'Logistic_Results_Decision_Final.csv')
|
||||||
|
|
||||||
|
def process_results(filepath, model_name="M₁"):
|
||||||
|
df = pd.read_csv(filepath, index_col=0)
|
||||||
|
|
||||||
|
# Check if we have S.E. or std err
|
||||||
|
se_col = 'S.E.' if 'S.E.' in df.columns else ('std err' if 'std err' in df.columns else None)
|
||||||
|
if not se_col:
|
||||||
|
# For some files it might be 'SE'
|
||||||
|
if 'SE' in df.columns: se_col = 'SE'
|
||||||
|
|
||||||
|
# Recreate necessary metrics for JASP format
|
||||||
|
estimate = df['Beta (B)']
|
||||||
|
se = df[se_col] if se_col else 0
|
||||||
|
odds_ratio = df['Odds Ratio EXP(B)']
|
||||||
|
p_val = df['P-value']
|
||||||
|
|
||||||
|
# Calculate Z and Wald
|
||||||
|
z_stat = estimate / se
|
||||||
|
wald = z_stat ** 2
|
||||||
|
lower = np.exp(estimate - 1.96 * se)
|
||||||
|
upper = np.exp(estimate + 1.96 * se)
|
||||||
|
|
||||||
|
# Rename index to match JASP
|
||||||
|
# JASP formats: e.g. "Gender_2" -> "Gender (2)", "const" -> "(Intercept)"
|
||||||
|
def map_name(name):
|
||||||
|
if name == 'const': return '(Intercept)'
|
||||||
|
if '_' in name:
|
||||||
|
parts = name.split('_')
|
||||||
|
return f"{parts[0]} ({parts[1]})"
|
||||||
|
return name
|
||||||
|
|
||||||
|
new_index = [map_name(str(i)) for i in df.index]
|
||||||
|
|
||||||
|
# Build the target dataframe
|
||||||
|
out_df = pd.DataFrame({
|
||||||
|
'Model': [model_name] + [np.nan] * (len(df) - 1),
|
||||||
|
'': new_index,
|
||||||
|
'Estimate': estimate.values,
|
||||||
|
'Standard Error': se.values if se_col else [np.nan]*len(df),
|
||||||
|
'Odds Ratio': odds_ratio.values,
|
||||||
|
'z': z_stat.values,
|
||||||
|
'Wald Statistic': wald.values,
|
||||||
|
'df': [1] * len(df),
|
||||||
|
'p': p_val.values,
|
||||||
|
'Lower bound': lower.values,
|
||||||
|
'Upper bound': upper.values
|
||||||
|
})
|
||||||
|
|
||||||
|
# Format p-values < .00001
|
||||||
|
out_df['p'] = out_df['p'].apply(lambda x: '< .00001' if pd.notnull(x) and x < 0.00001 else (round(x, 5) if pd.notnull(x) else x))
|
||||||
|
out_df = out_df.round(5)
|
||||||
|
return out_df
|
||||||
|
|
||||||
|
out_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\4_Logistic_Regression\KẾT_QUẢ_FINAL_JASP_LAYOUT.xlsx'
|
||||||
|
|
||||||
|
with pd.ExcelWriter(out_path, engine='openpyxl') as writer:
|
||||||
|
if os.path.exists(file_donation):
|
||||||
|
df_don = process_results(file_donation, "M₁")
|
||||||
|
df_don.to_excel(writer, sheet_name='Donation_New', index=False)
|
||||||
|
|
||||||
|
if os.path.exists(file_decision):
|
||||||
|
df_dec = process_results(file_decision, "M₁")
|
||||||
|
df_dec.to_excel(writer, sheet_name='Decision_New', index=False)
|
||||||
|
|
||||||
|
print(f"Exported successfully to {out_path}")
|
||||||
@@ -69,6 +69,7 @@ model_dec = sm.Logit(y_dec, X)
|
|||||||
res_dec = model_dec.fit(method='newton', maxiter=1000, disp=False)
|
res_dec = model_dec.fit(method='newton', maxiter=1000, disp=False)
|
||||||
sum_dec = pd.DataFrame({
|
sum_dec = pd.DataFrame({
|
||||||
'Beta (B)': res_dec.params,
|
'Beta (B)': res_dec.params,
|
||||||
|
'S.E.': res_dec.bse,
|
||||||
'P-value': res_dec.pvalues,
|
'P-value': res_dec.pvalues,
|
||||||
'Odds Ratio EXP(B)': np.exp(res_dec.params)
|
'Odds Ratio EXP(B)': np.exp(res_dec.params)
|
||||||
}).round(4)
|
}).round(4)
|
||||||
|
|||||||
Binary file not shown.
@@ -0,0 +1,102 @@
|
|||||||
|
import sys
|
||||||
|
sys.stdout.reconfigure(encoding='utf-8')
|
||||||
|
import pandas as pd
|
||||||
|
import numpy as np
|
||||||
|
import openpyxl
|
||||||
|
from openpyxl.utils.dataframe import dataframe_to_rows
|
||||||
|
from openpyxl.styles import Font, Alignment
|
||||||
|
import os
|
||||||
|
import shutil
|
||||||
|
|
||||||
|
base_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8'
|
||||||
|
orig_file = os.path.join(base_dir, 'KẾT QUẢ PHÂN TÍCH.xlsx')
|
||||||
|
final_file = os.path.join(base_dir, 'Project_Code_and_Results', 'KẾT_QUẢ_PHÂN_TÍCH_MASTER.xlsx')
|
||||||
|
|
||||||
|
# Copy original to new file
|
||||||
|
shutil.copy(orig_file, final_file)
|
||||||
|
|
||||||
|
# Load workbook
|
||||||
|
wb = openpyxl.load_workbook(final_file)
|
||||||
|
|
||||||
|
# Helper to process CSV into JASP layout dataframe
|
||||||
|
def get_jasp_df(filepath, model_name="M₁"):
|
||||||
|
df = pd.read_csv(filepath, index_col=0)
|
||||||
|
se_col = 'S.E.' if 'S.E.' in df.columns else ('std err' if 'std err' in df.columns else 'SE')
|
||||||
|
|
||||||
|
estimate = df['Beta (B)']
|
||||||
|
se = df[se_col]
|
||||||
|
odds_ratio = df['Odds Ratio EXP(B)']
|
||||||
|
p_val = df['P-value']
|
||||||
|
|
||||||
|
z_stat = estimate / se
|
||||||
|
wald = z_stat ** 2
|
||||||
|
lower = np.exp(estimate - 1.96 * se)
|
||||||
|
upper = np.exp(estimate + 1.96 * se)
|
||||||
|
|
||||||
|
def map_name(name):
|
||||||
|
if name == 'const': return '(Intercept)'
|
||||||
|
if '_' in name:
|
||||||
|
parts = name.split('_')
|
||||||
|
return f"{parts[0]} ({parts[1]})"
|
||||||
|
return name
|
||||||
|
|
||||||
|
new_index = [map_name(str(i)) for i in df.index]
|
||||||
|
|
||||||
|
out_df = pd.DataFrame({
|
||||||
|
'Model': [model_name] + [np.nan] * (len(df) - 1),
|
||||||
|
'': new_index,
|
||||||
|
'Estimate': estimate.values,
|
||||||
|
'Standard Error': se.values,
|
||||||
|
'Odds Ratio': odds_ratio.values,
|
||||||
|
'z': z_stat.values,
|
||||||
|
'Wald Statistic': wald.values,
|
||||||
|
'df': [1] * len(df),
|
||||||
|
'p': p_val.values,
|
||||||
|
'Lower bound': lower.values,
|
||||||
|
'Upper bound': upper.values
|
||||||
|
})
|
||||||
|
|
||||||
|
# Format
|
||||||
|
out_df['p'] = out_df['p'].apply(lambda x: '< .00001' if pd.notnull(x) and x < 0.00001 else (round(x, 5) if pd.notnull(x) else x))
|
||||||
|
out_df = out_df.round(5)
|
||||||
|
return out_df
|
||||||
|
|
||||||
|
# Replace a sheet's content
|
||||||
|
def replace_sheet(sheet_name, csv_path):
|
||||||
|
idx = wb.sheetnames.index(sheet_name)
|
||||||
|
del wb[sheet_name]
|
||||||
|
ws = wb.create_sheet(sheet_name, idx)
|
||||||
|
|
||||||
|
# Write Title
|
||||||
|
ws.cell(row=1, column=1, value="Logistic Regression (Updated with Firth/Optimized)").font = Font(bold=True)
|
||||||
|
ws.cell(row=3, column=1, value="Coefficients").font = Font(bold=True)
|
||||||
|
|
||||||
|
# Headers
|
||||||
|
df_jasp = get_jasp_df(csv_path)
|
||||||
|
headers = list(df_jasp.columns)
|
||||||
|
|
||||||
|
for c_idx, col_name in enumerate(headers, 1):
|
||||||
|
cell = ws.cell(row=4, column=c_idx, value=col_name)
|
||||||
|
cell.font = Font(bold=True)
|
||||||
|
cell.alignment = Alignment(horizontal='center')
|
||||||
|
|
||||||
|
ws.cell(row=3, column=10, value="95% Confidence interval").font = Font(bold=True)
|
||||||
|
ws.cell(row=3, column=10).alignment = Alignment(horizontal='center')
|
||||||
|
|
||||||
|
# Write Data
|
||||||
|
for r_idx, row in enumerate(dataframe_to_rows(df_jasp, index=False, header=False), 5):
|
||||||
|
for c_idx, value in enumerate(row, 1):
|
||||||
|
# Check for NaN float
|
||||||
|
if isinstance(value, float) and np.isnan(value):
|
||||||
|
ws.cell(row=r_idx, column=c_idx, value="")
|
||||||
|
else:
|
||||||
|
ws.cell(row=r_idx, column=c_idx, value=value)
|
||||||
|
|
||||||
|
path_don = os.path.join(base_dir, 'Project_Code_and_Results', '4_Logistic_Regression', 'Logistic_Results_Donation_Firth.csv')
|
||||||
|
path_dec = os.path.join(base_dir, 'Project_Code_and_Results', '4_Logistic_Regression', 'Logistic_Results_Decision_Final.csv')
|
||||||
|
|
||||||
|
replace_sheet('Donation', path_don)
|
||||||
|
replace_sheet('Decision', path_dec)
|
||||||
|
|
||||||
|
wb.save(final_file)
|
||||||
|
print("Tạo thành công MASTER file!")
|
||||||
@@ -0,0 +1,22 @@
|
|||||||
|
import sys
|
||||||
|
sys.stdout.reconfigure(encoding='utf-8')
|
||||||
|
import pandas as pd
|
||||||
|
import os
|
||||||
|
|
||||||
|
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\KẾT QUẢ PHÂN TÍCH.xlsx'
|
||||||
|
|
||||||
|
if not os.path.exists(file_path):
|
||||||
|
print("File not found")
|
||||||
|
sys.exit()
|
||||||
|
|
||||||
|
print(f"Reading file: {file_path}")
|
||||||
|
xl = pd.ExcelFile(file_path)
|
||||||
|
print(f"Sheet names: {xl.sheet_names}\n")
|
||||||
|
|
||||||
|
for sheet in xl.sheet_names:
|
||||||
|
print(f"--- Sheet: {sheet} ---")
|
||||||
|
df = xl.parse(sheet)
|
||||||
|
print(f"Shape: {df.shape}")
|
||||||
|
print(f"Columns: {list(df.columns)}")
|
||||||
|
print(f"First few rows:\n{df.head(3)}\n")
|
||||||
|
|
||||||
@@ -0,0 +1,11 @@
|
|||||||
|
import sys
|
||||||
|
sys.stdout.reconfigure(encoding='utf-8')
|
||||||
|
import pandas as pd
|
||||||
|
|
||||||
|
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\KẾT QUẢ PHÂN TÍCH.xlsx'
|
||||||
|
df_donation = pd.read_excel(file_path, sheet_name='Donation')
|
||||||
|
df_decision = pd.read_excel(file_path, sheet_name='Decision')
|
||||||
|
|
||||||
|
print("--- Donation Sheet Layout ---")
|
||||||
|
print(df_donation.head(20).to_string())
|
||||||
|
|
||||||
Reference in New Issue
Block a user