2 Commits

28 changed files with 497 additions and 25 deletions
Binary file not shown.
Binary file not shown.
@@ -0,0 +1,8 @@
,Exercises,Culture,Beauty,Education,Society,Spirit
Park,0.6367377065045589,0.5386259837748125,0.609527381412311,0.6666205318079247,0.5894275000421331,0.5971978092460599
Residential,0.6898067312223374,0.5322888366243355,0.7043563888714474,0.7003848393051737,0.6356124898043054,0.671536167178851
Garden,0.569207243129314,0.42043962025354226,0.520982025379601,0.5745723285479705,0.47015163689224226,0.4720301687679101
Rooftop,0.5220196490736146,0.39671027279755816,0.559846265287038,0.5289815361048281,0.46041084713416613,0.5313786451855366
Recreation,0.4416516078536275,0.4118096932349253,0.414973028889102,0.48397115121605433,0.347143218999983,0.3437593248955575
Agriculture,-0.30525120464643385,-0.09456459047576876,-0.41726188321313135,-0.3345389043778018,-0.3341241701147672,-0.37637058454953964
Nature,-0.44980612831551553,-0.27299890014702094,-0.49976092665649713,-0.4610579352258267,-0.4605091031334855,-0.5381766537399497
1 Exercises Culture Beauty Education Society Spirit
2 Park 0.6367377065045589 0.5386259837748125 0.609527381412311 0.6666205318079247 0.5894275000421331 0.5971978092460599
3 Residential 0.6898067312223374 0.5322888366243355 0.7043563888714474 0.7003848393051737 0.6356124898043054 0.671536167178851
4 Garden 0.569207243129314 0.42043962025354226 0.520982025379601 0.5745723285479705 0.47015163689224226 0.4720301687679101
5 Rooftop 0.5220196490736146 0.39671027279755816 0.559846265287038 0.5289815361048281 0.46041084713416613 0.5313786451855366
6 Recreation 0.4416516078536275 0.4118096932349253 0.414973028889102 0.48397115121605433 0.347143218999983 0.3437593248955575
7 Agriculture -0.30525120464643385 -0.09456459047576876 -0.41726188321313135 -0.3345389043778018 -0.3341241701147672 -0.37637058454953964
8 Nature -0.44980612831551553 -0.27299890014702094 -0.49976092665649713 -0.4610579352258267 -0.4605091031334855 -0.5381766537399497
@@ -0,0 +1,7 @@
,0,1
Exercises,0.007027806093845654,0.004957784418992443
Culture,-0.12773486154946914,-0.001881737518742339
Beauty,0.05188265296211081,0.0051213397312073914
Education,0.020143274994230118,0.013800122537297493
Society,0.004096923788506599,-0.004799612673251565
Spirit,0.04544596636820759,-0.018212152335741184
1 0 1
2 Exercises 0.007027806093845654 0.004957784418992443
3 Culture -0.12773486154946914 -0.001881737518742339
4 Beauty 0.05188265296211081 0.0051213397312073914
5 Education 0.020143274994230118 0.013800122537297493
6 Society 0.004096923788506599 -0.004799612673251565
7 Spirit 0.04544596636820759 -0.018212152335741184
@@ -0,0 +1,8 @@
,0,1
Park,0.01929915023578449,-0.0026557268484349067
Residential,0.03611771387245799,-0.006098756785044107
Garden,0.024297413495923986,0.008449977834710572
Rooftop,0.03634674358580918,-0.009949590505925508
Recreation,-0.0006198278964895602,0.015176373315937821
Agriculture,-0.14550165539847498,-0.015634536842211624
Nature,-0.14586553315107212,0.01084942463742428
1 0 1
2 Park 0.01929915023578449 -0.0026557268484349067
3 Residential 0.03611771387245799 -0.006098756785044107
4 Garden 0.024297413495923986 0.008449977834710572
5 Rooftop 0.03634674358580918 -0.009949590505925508
6 Recreation -0.0006198278964895602 0.015176373315937821
7 Agriculture -0.14550165539847498 -0.015634536842211624
8 Nature -0.14586553315107212 0.01084942463742428
@@ -0,0 +1,8 @@
,Dirty,Unsafe,Danger
Park,-0.40659517420422864,-0.3652691650630517,-0.47632864603181757
Residential,-0.42361014190625074,-0.37318965467898796,-0.4711066759095078
Garden,-0.3109599181970276,-0.27653098840256857,-0.34126139088328367
Rooftop,-0.313359215935556,-0.3361077310042239,-0.35612431235937536
Recreation,-0.09348749429342133,-0.14803279299655103,-0.27167796733643157
Agriculture,0.48572656432897177,0.46262486117052976,0.5273673898837783
Nature,0.5129174934902375,0.5064560749634449,0.5363089258301386
1 Dirty Unsafe Danger
2 Park -0.40659517420422864 -0.3652691650630517 -0.47632864603181757
3 Residential -0.42361014190625074 -0.37318965467898796 -0.4711066759095078
4 Garden -0.3109599181970276 -0.27653098840256857 -0.34126139088328367
5 Rooftop -0.313359215935556 -0.3361077310042239 -0.35612431235937536
6 Recreation -0.09348749429342133 -0.14803279299655103 -0.27167796733643157
7 Agriculture 0.48572656432897177 0.46262486117052976 0.5273673898837783
8 Nature 0.5129174934902375 0.5064560749634449 0.5363089258301386
@@ -0,0 +1,4 @@
,0,1
Dirty,-0.021750742704197926,-0.023688662634614326
Unsafe,-0.031685779157853394,0.02094725474813505
Danger,0.056173988030702275,0.002810907103117069
1 0 1
2 Dirty -0.021750742704197926 -0.023688662634614326
3 Unsafe -0.031685779157853394 0.02094725474813505
4 Danger 0.056173988030702275 0.002810907103117069
@@ -0,0 +1,8 @@
,0,1
Park,-0.05277737050209137,0.021966851260274492
Residential,-0.039828498949265316,0.030086993744197595
Garden,-0.011237261666548438,0.01787364745984657
Rooftop,0.0025503277042415763,-0.014900920261128709
Recreation,-0.05985983449204827,-0.034402423139858064
Agriculture,0.04093472746557441,-0.0031882467528099086
Nature,0.03188949642613842,0.0004223966469934847
1 0 1
2 Park -0.05277737050209137 0.021966851260274492
3 Residential -0.039828498949265316 0.030086993744197595
4 Garden -0.011237261666548438 0.01787364745984657
5 Rooftop 0.0025503277042415763 -0.014900920261128709
6 Recreation -0.05985983449204827 -0.034402423139858064
7 Agriculture 0.04093472746557441 -0.0031882467528099086
8 Nature 0.03188949642613842 0.0004223966469934847
@@ -0,0 +1,8 @@
,MEAN RES,MEAN CES,MEAN DES
Park,0.6110543201654319,0.6768852442644577,-0.47060044983099286
Residential,0.6407085222739133,0.7291962678143591,-0.4708562921098795
Garden,0.44165260890843555,0.5480223668648628,-0.34198141262933823
Rooftop,0.49600027293784166,0.5419994714753112,-0.3688582414111523
Recreation,0.3745893900657618,0.40850035774422844,-0.17674812102484044
Agriculture,-0.3502787611798973,-0.39573598649446984,0.5625037794065111
Nature,-0.4734036284834367,-0.5391708531031724,0.582247211793259
1 MEAN RES MEAN CES MEAN DES
2 Park 0.6110543201654319 0.6768852442644577 -0.47060044983099286
3 Residential 0.6407085222739133 0.7291962678143591 -0.4708562921098795
4 Garden 0.44165260890843555 0.5480223668648628 -0.34198141262933823
5 Rooftop 0.49600027293784166 0.5419994714753112 -0.3688582414111523
6 Recreation 0.3745893900657618 0.40850035774422844 -0.17674812102484044
7 Agriculture -0.3502787611798973 -0.39573598649446984 0.5625037794065111
8 Nature -0.4734036284834367 -0.5391708531031724 0.582247211793259
@@ -0,0 +1,4 @@
,0,1
MEAN RES,-0.2186071848607324,-0.007839144868552231
MEAN CES,-0.2574749215265179,0.007317945054910379
MEAN DES,0.6682022605408963,0.00045551629187779
1 0 1
2 MEAN RES -0.2186071848607324 -0.007839144868552231
3 MEAN CES -0.2574749215265179 0.007317945054910379
4 MEAN DES 0.6682022605408963 0.00045551629187779
@@ -0,0 +1,8 @@
,0,1
Park,-0.28217527664101055,-0.004698104759826525
Residential,-0.2891839961693309,0.00131726592526606
Garden,-0.18810995860092022,0.013756464641007708
Rooftop,-0.20631526503069358,-0.0065124662879235936
Recreation,-0.07849003909114283,-0.0039717946663235335
Agriculture,0.6635981475555077,0.002205946282469859
Nature,0.8024031622133635,-0.002091864403477851
1 0 1
2 Park -0.28217527664101055 -0.004698104759826525
3 Residential -0.2891839961693309 0.00131726592526606
4 Garden -0.18810995860092022 0.013756464641007708
5 Rooftop -0.20631526503069358 -0.0065124662879235936
6 Recreation -0.07849003909114283 -0.0039717946663235335
7 Agriculture 0.6635981475555077 0.002205946282469859
8 Nature 0.8024031622133635 -0.002091864403477851
@@ -0,0 +1,8 @@
,Temperature,Noise,Stormwind,Respiratory
Park,0.5943691697194959,0.4789900955834734,0.5400353044339286,0.5612018360594374
Residential,0.6098741786551782,0.4993073835325437,0.5888125616106002,0.6297059731338867
Garden,0.45042886274863986,0.3963838785463788,0.42396144532666635,0.41516036007093543
Rooftop,0.49442453924504476,0.4358175328312086,0.4842898044085772,0.4021375963717873
Recreation,0.40664357632988,0.3565204984864607,0.39747369992230713,0.3250145065589509
Agriculture,-0.3272930077585318,-0.23403946925952315,-0.3063551523911346,-0.320369510797842
Nature,-0.44797645118533835,-0.30105163246668354,-0.4709764053605818,-0.45237173056171975
1 Temperature Noise Stormwind Respiratory
2 Park 0.5943691697194959 0.4789900955834734 0.5400353044339286 0.5612018360594374
3 Residential 0.6098741786551782 0.4993073835325437 0.5888125616106002 0.6297059731338867
4 Garden 0.45042886274863986 0.3963838785463788 0.42396144532666635 0.41516036007093543
5 Rooftop 0.49442453924504476 0.4358175328312086 0.4842898044085772 0.4021375963717873
6 Recreation 0.40664357632988 0.3565204984864607 0.39747369992230713 0.3250145065589509
7 Agriculture -0.3272930077585318 -0.23403946925952315 -0.3063551523911346 -0.320369510797842
8 Nature -0.44797645118533835 -0.30105163246668354 -0.4709764053605818 -0.45237173056171975
@@ -0,0 +1,5 @@
,0,1
Temperature,-0.023338638107689796,-0.006213751498479767
Noise,0.06553553877191058,-0.0005751574284129316
Stormwind,-0.021996594532673455,-0.013022115419183451
Respiratory,-0.019898887972482068,0.02012270293988676
1 0 1
2 Temperature -0.023338638107689796 -0.006213751498479767
3 Noise 0.06553553877191058 -0.0005751574284129316
4 Stormwind -0.021996594532673455 -0.013022115419183451
5 Respiratory -0.019898887972482068 0.02012270293988676
@@ -0,0 +1,8 @@
,0,1
Park,-0.02240002337805916,0.007917221215969647
Residential,-0.02799189713117847,0.015859200211763763
Garden,-0.008475103510200898,0.0016829493791823964
Rooftop,-0.0060566811778039985,-0.017070154943948847
Recreation,-0.004992858046068623,-0.015594566177349648
Agriculture,0.05380057559170161,0.0006539273597548665
Nature,0.11812861315235715,0.010393415861558994
1 0 1
2 Park -0.02240002337805916 0.007917221215969647
3 Residential -0.02799189713117847 0.015859200211763763
4 Garden -0.008475103510200898 0.0016829493791823964
5 Rooftop -0.0060566811778039985 -0.017070154943948847
6 Recreation -0.004992858046068623 -0.015594566177349648
7 Agriculture 0.05380057559170161 0.0006539273597548665
8 Nature 0.11812861315235715 0.010393415861558994
Binary file not shown.

After

Width:  |  Height:  |  Size: 194 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 172 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 189 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 185 KiB

@@ -0,0 +1,90 @@
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import prince
import os
# Đường dẫn file dữ liệu
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Data_VN_filter_v7.xlsx'
output_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\2_CA_and_Spearman'
# Load data
df = pd.read_excel(file_path)
# Extract variables
ugs_cols = ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
mean_cols = ['MEAN RES', 'MEAN CES', 'MEAN DES']
colors = ['green', 'purple', 'red']
print("Running CA for MEAN RES, MEAN CES, MEAN DES...")
# Lọc các biến có trong data
available_means = [c for c in mean_cols if c in df.columns]
if not available_means:
print("Không tìm thấy các cột MEAN trong dữ liệu.")
exit()
data = df[ugs_cols + available_means].dropna()
# Tạo ma trận tương quan Spearman
corr = data.corr(method='spearman').loc[ugs_cols, available_means]
# Dịch chuyển ma trận để tất cả các giá trị đều dương (CA yêu cầu bảng tần số hoặc giá trị dương)
corr_shifted = corr + 1
# Initialize CA
ca = prince.CA(n_components=2, n_iter=3, copy=True, check_input=True, engine='scipy', random_state=42)
ca = ca.fit(corr_shifted)
# Extract column and row coordinates
row_coords = ca.row_coordinates(corr_shifted) # UGS
col_coords = ca.column_coordinates(corr_shifted) # Mean Services
# Save to CSV
group_name = "MEAN_ESS_DES"
corr.to_csv(os.path.join(output_dir, f'CA_{group_name}_Correlation_Matrix.csv'))
row_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_UGS_Coords.csv'))
col_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_Services_Coords.csv'))
# Plot Biplot
fig, ax = plt.subplots(figsize=(10, 8))
# Plot UGS points
ax.scatter(row_coords[0], row_coords[1], c='blue', label='UGS Types', s=80, marker='o', edgecolors='black')
# Plot Mean Services points with corresponding colors
for i, col in enumerate(available_means):
c_idx = mean_cols.index(col)
color = colors[c_idx]
ax.scatter(col_coords.loc[col, 0], col_coords.loc[col, 1], c=color, marker='s', s=100, edgecolors='black')
# Labels
for i, txt in enumerate(ugs_cols):
ax.annotate(txt, (row_coords.iloc[i, 0], row_coords.iloc[i, 1]),
xytext=(-10, 10), textcoords='offset points',
color='blue', fontweight='bold', fontsize=12, ha='right', va='bottom')
for i, txt in enumerate(available_means):
c_idx = mean_cols.index(txt)
color = colors[c_idx]
ax.annotate(txt, (col_coords.loc[txt, 0], col_coords.loc[txt, 1]),
xytext=(10, -10), textcoords='offset points',
color=color, fontsize=12, fontweight='bold', ha='left', va='top')
# Add dummy plots for legend
ax.scatter([], [], c='green', marker='s', label='MEAN RES (Điều hòa)')
ax.scatter([], [], c='purple', marker='s', label='MEAN CES (Văn hóa)')
ax.scatter([], [], c='red', marker='s', label='MEAN DES (Bất lợi)')
ax.axhline(0, color='grey', linestyle='--', linewidth=1)
ax.axvline(0, color='grey', linestyle='--', linewidth=1)
ax.set_title('CA Biplot: UGS vs MEAN RES, MEAN CES, MEAN DES', fontsize=16)
ax.set_xlabel('Component 0', fontsize=12)
ax.set_ylabel('Component 1', fontsize=12)
ax.legend(loc='best', fontsize=12)
plt.grid(True, linestyle=':', alpha=0.6)
plt.savefig(os.path.join(output_dir, f'CA_plot_{group_name}.png'), dpi=300, bbox_inches='tight')
plt.close()
print("Finished MEAN ESS/DES.")
@@ -0,0 +1,87 @@
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import prince
import os
# Đường dẫn file dữ liệu
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Data_VN_filter_v7.xlsx'
output_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\2_CA_and_Spearman'
# Load data
df = pd.read_excel(file_path)
# Extract variables
ugs_cols = ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
groups = {
'RES': ['Temperature', 'Noise', 'Stormwind', 'Respiratory'],
'CES': ['Exercises', 'Culture', 'Beauty', 'Education', 'Society', 'Spirit'],
'DES': ['Dirty', 'Unsafe', 'Danger']
}
colors = {
'RES': 'green',
'CES': 'purple',
'DES': 'red'
}
for group_name, cols in groups.items():
available_cols = [c for c in cols if c in df.columns]
if not available_cols:
print(f"No columns found for {group_name}")
continue
print(f"Running CA for {group_name}...")
data = df[ugs_cols + available_cols].dropna()
corr = data.corr(method='spearman').loc[ugs_cols, available_cols]
corr_shifted = corr + 1
# Initialize CA
ca = prince.CA(n_components=2, n_iter=3, copy=True, check_input=True, engine='scipy', random_state=42)
ca = ca.fit(corr_shifted)
# Extract column and row coordinates
row_coords = ca.row_coordinates(corr_shifted) # UGS
col_coords = ca.column_coordinates(corr_shifted) # Services
# Save to CSV
corr.to_csv(os.path.join(output_dir, f'CA_{group_name}_Correlation_Matrix.csv'))
row_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_UGS_Coords.csv'))
col_coords.to_csv(os.path.join(output_dir, f'CA_{group_name}_Services_Coords.csv'))
# Plot Biplot
fig, ax = plt.subplots(figsize=(10, 8))
# Plot UGS points
ax.scatter(row_coords[0], row_coords[1], c='blue', label='UGS Types', s=80, marker='o', edgecolors='black')
# Plot Services points
color = colors[group_name]
ax.scatter(col_coords[0], col_coords[1], c=color, marker='s', s=80, edgecolors='black', label=group_name)
# Labels
for i, txt in enumerate(ugs_cols):
ax.annotate(txt, (row_coords.iloc[i, 0], row_coords.iloc[i, 1]),
xytext=(-10, 10), textcoords='offset points',
color='blue', fontweight='bold', fontsize=12, ha='right', va='bottom')
for i, txt in enumerate(available_cols):
ax.annotate(txt, (col_coords.iloc[i, 0], col_coords.iloc[i, 1]),
xytext=(10, -10), textcoords='offset points',
color=color, fontsize=12, fontweight='bold', ha='left', va='top')
ax.axhline(0, color='grey', linestyle='--', linewidth=1)
ax.axvline(0, color='grey', linestyle='--', linewidth=1)
ax.set_title(f'CA Biplot: UGS vs {group_name}', fontsize=16)
ax.set_xlabel('Component 0', fontsize=12)
ax.set_ylabel('Component 1', fontsize=12)
ax.legend(loc='best', fontsize=12)
plt.grid(True, linestyle=':', alpha=0.6)
plt.savefig(os.path.join(output_dir, f'CA_plot_{group_name}.png'), dpi=300, bbox_inches='tight')
plt.close()
print(f"Finished {group_name}.")
@@ -1,25 +1,25 @@
,Beta (B),P-value,Odds Ratio EXP(B),Significance
const,-13.3839,0.0078,0.0,**
Frequency,0.7183,0.0098,2.051,**
Career_6,4.1539,0.0105,63.6799,*
Rooftop,1.7017,0.0123,5.4833,*
Income,0.0045,0.0229,1.0045,*
Transportation_4,4.6473,0.0284,104.3069,*
MEAN CES,1.8488,0.0288,6.3523,*
Career_4,2.8862,0.0334,17.9242,*
Career_3,3.4267,0.0368,30.7762,*
Garden,-1.4026,0.0446,0.246,*
MEAN DES,-0.4982,0.0586,0.6076,
Transportation_3,2.1343,0.0739,8.4512,
Literacy,0.5519,0.0906,1.7365,
Gender_2,1.1137,0.1406,3.0455,
Nature,-0.6135,0.2466,0.5415,
MEAN RES,-0.6547,0.2723,0.5196,
Residential,0.6836,0.391,1.981,
Time,-0.1722,0.5721,0.8418,
Distance,-0.161,0.6202,0.8513,
Transportation_2,0.6629,0.6341,1.9404,
Recreation,-0.1187,0.7927,0.8881,
Career_2,0.347,0.8054,1.4148,
Park,-0.1024,0.8656,0.9027,
Agriculture,-0.0665,0.8894,0.9356,
,Beta (B),S.E.,P-value,Odds Ratio EXP(B),Significance
const,-13.3839,5.0267,0.0078,0.0,**
Frequency,0.7183,0.278,0.0098,2.051,**
Career_6,4.1539,1.6229,0.0105,63.6799,*
Rooftop,1.7017,0.6798,0.0123,5.4833,*
Income,0.0045,0.002,0.0229,1.0045,*
Transportation_4,4.6473,2.1208,0.0284,104.3069,*
MEAN CES,1.8488,0.8458,0.0288,6.3523,*
Career_4,2.8862,1.3569,0.0334,17.9242,*
Career_3,3.4267,1.6413,0.0368,30.7762,*
Garden,-1.4026,0.6984,0.0446,0.246,*
MEAN DES,-0.4982,0.2635,0.0586,0.6076,
Transportation_3,2.1343,1.1942,0.0739,8.4512,
Literacy,0.5519,0.3261,0.0906,1.7365,
Gender_2,1.1137,0.7557,0.1406,3.0455,
Nature,-0.6135,0.5294,0.2466,0.5415,
MEAN RES,-0.6547,0.5964,0.2723,0.5196,
Residential,0.6836,0.7969,0.391,1.981,
Time,-0.1722,0.3048,0.5721,0.8418,
Distance,-0.161,0.3249,0.6202,0.8513,
Transportation_2,0.6629,1.3928,0.6341,1.9404,
Recreation,-0.1187,0.4515,0.7927,0.8881,
Career_2,0.347,1.4082,0.8054,1.4148,
Park,-0.1024,0.6051,0.8656,0.9027,
Agriculture,-0.0665,0.4784,0.8894,0.9356,
1 Beta (B) S.E. P-value Odds Ratio EXP(B) Significance
2 const -13.3839 5.0267 0.0078 0.0 **
3 Frequency 0.7183 0.278 0.0098 2.051 **
4 Career_6 4.1539 1.6229 0.0105 63.6799 *
5 Rooftop 1.7017 0.6798 0.0123 5.4833 *
6 Income 0.0045 0.002 0.0229 1.0045 *
7 Transportation_4 4.6473 2.1208 0.0284 104.3069 *
8 MEAN CES 1.8488 0.8458 0.0288 6.3523 *
9 Career_4 2.8862 1.3569 0.0334 17.9242 *
10 Career_3 3.4267 1.6413 0.0368 30.7762 *
11 Garden -1.4026 0.6984 0.0446 0.246 *
12 MEAN DES -0.4982 0.2635 0.0586 0.6076
13 Transportation_3 2.1343 1.1942 0.0739 8.4512
14 Literacy 0.5519 0.3261 0.0906 1.7365
15 Gender_2 1.1137 0.7557 0.1406 3.0455
16 Nature -0.6135 0.5294 0.2466 0.5415
17 MEAN RES -0.6547 0.5964 0.2723 0.5196
18 Residential 0.6836 0.7969 0.391 1.981
19 Time -0.1722 0.3048 0.5721 0.8418
20 Distance -0.161 0.3249 0.6202 0.8513
21 Transportation_2 0.6629 1.3928 0.6341 1.9404
22 Recreation -0.1187 0.4515 0.7927 0.8881
23 Career_2 0.347 1.4082 0.8054 1.4148
24 Park -0.1024 0.6051 0.8656 0.9027
25 Agriculture -0.0665 0.4784 0.8894 0.9356
@@ -0,0 +1,75 @@
import sys
sys.stdout.reconfigure(encoding='utf-8')
import pandas as pd
import numpy as np
import os
base_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\4_Logistic_Regression'
file_donation = os.path.join(base_dir, 'Logistic_Results_Donation_Firth.csv')
file_decision = os.path.join(base_dir, 'Logistic_Results_Decision_Final.csv')
def process_results(filepath, model_name="M₁"):
df = pd.read_csv(filepath, index_col=0)
# Check if we have S.E. or std err
se_col = 'S.E.' if 'S.E.' in df.columns else ('std err' if 'std err' in df.columns else None)
if not se_col:
# For some files it might be 'SE'
if 'SE' in df.columns: se_col = 'SE'
# Recreate necessary metrics for JASP format
estimate = df['Beta (B)']
se = df[se_col] if se_col else 0
odds_ratio = df['Odds Ratio EXP(B)']
p_val = df['P-value']
# Calculate Z and Wald
z_stat = estimate / se
wald = z_stat ** 2
lower = np.exp(estimate - 1.96 * se)
upper = np.exp(estimate + 1.96 * se)
# Rename index to match JASP
# JASP formats: e.g. "Gender_2" -> "Gender (2)", "const" -> "(Intercept)"
def map_name(name):
if name == 'const': return '(Intercept)'
if '_' in name:
parts = name.split('_')
return f"{parts[0]} ({parts[1]})"
return name
new_index = [map_name(str(i)) for i in df.index]
# Build the target dataframe
out_df = pd.DataFrame({
'Model': [model_name] + [np.nan] * (len(df) - 1),
'': new_index,
'Estimate': estimate.values,
'Standard Error': se.values if se_col else [np.nan]*len(df),
'Odds Ratio': odds_ratio.values,
'z': z_stat.values,
'Wald Statistic': wald.values,
'df': [1] * len(df),
'p': p_val.values,
'Lower bound': lower.values,
'Upper bound': upper.values
})
# Format p-values < .00001
out_df['p'] = out_df['p'].apply(lambda x: '< .00001' if pd.notnull(x) and x < 0.00001 else (round(x, 5) if pd.notnull(x) else x))
out_df = out_df.round(5)
return out_df
out_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Project_Code_and_Results\4_Logistic_Regression\KẾT_QUẢ_FINAL_JASP_LAYOUT.xlsx'
with pd.ExcelWriter(out_path, engine='openpyxl') as writer:
if os.path.exists(file_donation):
df_don = process_results(file_donation, "M₁")
df_don.to_excel(writer, sheet_name='Donation_New', index=False)
if os.path.exists(file_decision):
df_dec = process_results(file_decision, "M₁")
df_dec.to_excel(writer, sheet_name='Decision_New', index=False)
print(f"Exported successfully to {out_path}")
@@ -69,6 +69,7 @@ model_dec = sm.Logit(y_dec, X)
res_dec = model_dec.fit(method='newton', maxiter=1000, disp=False)
sum_dec = pd.DataFrame({
'Beta (B)': res_dec.params,
'S.E.': res_dec.bse,
'P-value': res_dec.pvalues,
'Odds Ratio EXP(B)': np.exp(res_dec.params)
}).round(4)
@@ -0,0 +1,102 @@
import sys
sys.stdout.reconfigure(encoding='utf-8')
import pandas as pd
import numpy as np
import openpyxl
from openpyxl.utils.dataframe import dataframe_to_rows
from openpyxl.styles import Font, Alignment
import os
import shutil
base_dir = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8'
orig_file = os.path.join(base_dir, 'KẾT QUẢ PHÂN TÍCH.xlsx')
final_file = os.path.join(base_dir, 'Project_Code_and_Results', 'KẾT_QUẢ_PHÂN_TÍCH_MASTER.xlsx')
# Copy original to new file
shutil.copy(orig_file, final_file)
# Load workbook
wb = openpyxl.load_workbook(final_file)
# Helper to process CSV into JASP layout dataframe
def get_jasp_df(filepath, model_name="M₁"):
df = pd.read_csv(filepath, index_col=0)
se_col = 'S.E.' if 'S.E.' in df.columns else ('std err' if 'std err' in df.columns else 'SE')
estimate = df['Beta (B)']
se = df[se_col]
odds_ratio = df['Odds Ratio EXP(B)']
p_val = df['P-value']
z_stat = estimate / se
wald = z_stat ** 2
lower = np.exp(estimate - 1.96 * se)
upper = np.exp(estimate + 1.96 * se)
def map_name(name):
if name == 'const': return '(Intercept)'
if '_' in name:
parts = name.split('_')
return f"{parts[0]} ({parts[1]})"
return name
new_index = [map_name(str(i)) for i in df.index]
out_df = pd.DataFrame({
'Model': [model_name] + [np.nan] * (len(df) - 1),
'': new_index,
'Estimate': estimate.values,
'Standard Error': se.values,
'Odds Ratio': odds_ratio.values,
'z': z_stat.values,
'Wald Statistic': wald.values,
'df': [1] * len(df),
'p': p_val.values,
'Lower bound': lower.values,
'Upper bound': upper.values
})
# Format
out_df['p'] = out_df['p'].apply(lambda x: '< .00001' if pd.notnull(x) and x < 0.00001 else (round(x, 5) if pd.notnull(x) else x))
out_df = out_df.round(5)
return out_df
# Replace a sheet's content
def replace_sheet(sheet_name, csv_path):
idx = wb.sheetnames.index(sheet_name)
del wb[sheet_name]
ws = wb.create_sheet(sheet_name, idx)
# Write Title
ws.cell(row=1, column=1, value="Logistic Regression (Updated with Firth/Optimized)").font = Font(bold=True)
ws.cell(row=3, column=1, value="Coefficients").font = Font(bold=True)
# Headers
df_jasp = get_jasp_df(csv_path)
headers = list(df_jasp.columns)
for c_idx, col_name in enumerate(headers, 1):
cell = ws.cell(row=4, column=c_idx, value=col_name)
cell.font = Font(bold=True)
cell.alignment = Alignment(horizontal='center')
ws.cell(row=3, column=10, value="95% Confidence interval").font = Font(bold=True)
ws.cell(row=3, column=10).alignment = Alignment(horizontal='center')
# Write Data
for r_idx, row in enumerate(dataframe_to_rows(df_jasp, index=False, header=False), 5):
for c_idx, value in enumerate(row, 1):
# Check for NaN float
if isinstance(value, float) and np.isnan(value):
ws.cell(row=r_idx, column=c_idx, value="")
else:
ws.cell(row=r_idx, column=c_idx, value=value)
path_don = os.path.join(base_dir, 'Project_Code_and_Results', '4_Logistic_Regression', 'Logistic_Results_Donation_Firth.csv')
path_dec = os.path.join(base_dir, 'Project_Code_and_Results', '4_Logistic_Regression', 'Logistic_Results_Decision_Final.csv')
replace_sheet('Donation', path_don)
replace_sheet('Decision', path_dec)
wb.save(final_file)
print("Tạo thành công MASTER file!")
@@ -0,0 +1,22 @@
import sys
sys.stdout.reconfigure(encoding='utf-8')
import pandas as pd
import os
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\KẾT QUẢ PHÂN TÍCH.xlsx'
if not os.path.exists(file_path):
print("File not found")
sys.exit()
print(f"Reading file: {file_path}")
xl = pd.ExcelFile(file_path)
print(f"Sheet names: {xl.sheet_names}\n")
for sheet in xl.sheet_names:
print(f"--- Sheet: {sheet} ---")
df = xl.parse(sheet)
print(f"Shape: {df.shape}")
print(f"Columns: {list(df.columns)}")
print(f"First few rows:\n{df.head(3)}\n")
@@ -0,0 +1,11 @@
import sys
sys.stdout.reconfigure(encoding='utf-8')
import pandas as pd
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\KẾT QUẢ PHÂN TÍCH.xlsx'
df_donation = pd.read_excel(file_path, sheet_name='Donation')
df_decision = pd.read_excel(file_path, sheet_name='Decision')
print("--- Donation Sheet Layout ---")
print(df_donation.head(20).to_string())