chuyển đổi cấu trúc
This commit is contained in:
@@ -0,0 +1,40 @@
|
||||
,Beta (B),P-value,Odds Ratio EXP(B),Significance
|
||||
Rooftop,3.0756,0.0079,21.6626,**
|
||||
Frequency_4,5.5916,0.0129,268.1578,*
|
||||
Career_6,9.1882,0.0176,9781.3356,*
|
||||
Career_4,6.1367,0.022,462.5281,*
|
||||
Career_3,7.1144,0.0315,1229.5645,*
|
||||
Income,0.011,0.0387,1.0111,*
|
||||
MEAN CES,2.9215,0.0495,18.5696,*
|
||||
Garden,-2.2436,0.0545,0.1061,
|
||||
const,-22.5952,0.0608,0.0,
|
||||
Transportation_3,4.0109,0.0812,55.1958,
|
||||
Nature,-1.545,0.1226,0.2133,
|
||||
Frequency_5,2.4568,0.1369,11.668,
|
||||
MEAN DES,-0.5645,0.188,0.5686,
|
||||
Frequency_3,2.0046,0.2622,7.4228,
|
||||
Gender_2,1.6345,0.2689,5.1269,
|
||||
Transportation_5,8.8139,0.2931,6726.8423,
|
||||
MEAN RES,-0.8526,0.3366,0.4263,
|
||||
Recreation,-0.5931,0.3811,0.5526,
|
||||
Agriculture,0.6439,0.4038,1.904,
|
||||
Distance_2,1.3034,0.4172,3.6819,
|
||||
Park,0.7798,0.431,2.181,
|
||||
Transportation_4,2.0649,0.4405,7.8846,
|
||||
Time_4,1.74,0.5326,5.6971,
|
||||
Distance_4,-1.2593,0.5443,0.2838,
|
||||
Distance_3,1.2619,0.6309,3.532,
|
||||
Literacy_3,-3.3014,0.6631,0.0368,
|
||||
Frequency_2,0.683,0.7017,1.9799,
|
||||
Career_5,1.4116,0.7259,4.1024,
|
||||
Literacy_2,-2.6458,0.7276,0.071,
|
||||
Time_5,-0.8678,0.7459,0.4199,
|
||||
Residential,0.3326,0.7564,1.3946,
|
||||
Career_2,0.7339,0.7868,2.0831,
|
||||
Time_3,-0.7335,0.7979,0.4802,
|
||||
Literacy_6,1.6746,0.8247,5.3367,
|
||||
Transportation_2,0.3599,0.8906,1.4332,
|
||||
Literacy_4,-0.8905,0.9037,0.4105,
|
||||
Distance_5,-0.1975,0.933,0.8208,
|
||||
Time_2,0.2227,0.9368,1.2494,
|
||||
Literacy_5,-6.5281,0.978,0.0015,
|
||||
|
@@ -0,0 +1,25 @@
|
||||
,Beta (B),P-value,Odds Ratio EXP(B),Significance
|
||||
const,-13.3839,0.0078,0.0,**
|
||||
Frequency,0.7183,0.0098,2.051,**
|
||||
Career_6,4.1539,0.0105,63.6799,*
|
||||
Rooftop,1.7017,0.0123,5.4833,*
|
||||
Income,0.0045,0.0229,1.0045,*
|
||||
Transportation_4,4.6473,0.0284,104.3069,*
|
||||
MEAN CES,1.8488,0.0288,6.3523,*
|
||||
Career_4,2.8862,0.0334,17.9242,*
|
||||
Career_3,3.4267,0.0368,30.7762,*
|
||||
Garden,-1.4026,0.0446,0.246,*
|
||||
MEAN DES,-0.4982,0.0586,0.6076,
|
||||
Transportation_3,2.1343,0.0739,8.4512,
|
||||
Literacy,0.5519,0.0906,1.7365,
|
||||
Gender_2,1.1137,0.1406,3.0455,
|
||||
Nature,-0.6135,0.2466,0.5415,
|
||||
MEAN RES,-0.6547,0.2723,0.5196,
|
||||
Residential,0.6836,0.391,1.981,
|
||||
Time,-0.1722,0.5721,0.8418,
|
||||
Distance,-0.161,0.6202,0.8513,
|
||||
Transportation_2,0.6629,0.6341,1.9404,
|
||||
Recreation,-0.1187,0.7927,0.8881,
|
||||
Career_2,0.347,0.8054,1.4148,
|
||||
Park,-0.1024,0.8656,0.9027,
|
||||
Agriculture,-0.0665,0.8894,0.9356,
|
||||
|
@@ -0,0 +1,40 @@
|
||||
,Beta (B),P-value,Odds Ratio EXP(B),Significance
|
||||
const,241.6608,,8.952878165232437e+104,
|
||||
Income,-0.9832,,0.3741,
|
||||
MEAN RES,-6.4358,,0.0016,
|
||||
MEAN CES,72.6282,,3.4836963617736678e+31,
|
||||
MEAN DES,-95.8865,,0.0,
|
||||
Park,-50.6859,,0.0,
|
||||
Residential,258.3765,,1.6273086710804255e+112,
|
||||
Garden,-10.3217,,0.0,
|
||||
Rooftop,-133.4091,,0.0,
|
||||
Recreation,-20.842,,0.0,
|
||||
Agriculture,0.7005,,2.0148,
|
||||
Nature,-82.9613,,0.0,
|
||||
Gender_2,-102.1304,,0.0,
|
||||
Career_2,221.4766,,1.534848675005764e+96,
|
||||
Career_3,85.8488,,1.9215803519284693e+37,
|
||||
Career_4,29.8107,,8843571707594.404,
|
||||
Career_5,-19.9903,,0.0,
|
||||
Career_6,271.5147,,8.266885235464909e+117,
|
||||
Literacy_2,-45.5232,,0.0,
|
||||
Literacy_3,74.167,,1.6230180759688443e+32,
|
||||
Literacy_4,-130.4477,,0.0,
|
||||
Literacy_5,41.9698,,1.6875893967463475e+18,
|
||||
Literacy_6,239.8173,,1.4168061493135717e+104,
|
||||
Frequency_2,-64.3464,,0.0,
|
||||
Frequency_3,129.1509,,1.2289014711896098e+56,
|
||||
Frequency_4,16.8335,,20449449.0163,
|
||||
Frequency_5,90.1682,,1.4439203359332122e+39,
|
||||
Distance_2,121.504,,5.868283849671952e+52,
|
||||
Distance_3,-19.3408,,0.0,
|
||||
Distance_4,-8.0289,,0.0003,
|
||||
Distance_5,106.1989,,1.3231308267100774e+46,
|
||||
Time_2,123.4271,,4.015015109349733e+53,
|
||||
Time_3,293.3949,,2.628883873565357e+127,
|
||||
Time_4,172.2575,,6.463889160702152e+74,
|
||||
Time_5,91.6604,,6.420867256078891e+39,
|
||||
Transportation_2,86.2789,,2.9541644055797275e+37,
|
||||
Transportation_3,-67.1359,,0.0,
|
||||
Transportation_4,-197.7757,,0.0,
|
||||
Transportation_5,135.1477,,4.941998234455287e+58,
|
||||
|
@@ -0,0 +1,18 @@
|
||||
,Beta (B),S.E.,P-value,Odds Ratio EXP(B),Significance
|
||||
MEAN DES,-0.8867,0.2551,0.0005,0.412,***
|
||||
Income,-0.0043,0.0013,0.0014,0.9957,**
|
||||
Distance,-0.5634,0.2603,0.0304,0.5693,*
|
||||
Gender_2,-1.3185,0.6722,0.0498,0.2675,*
|
||||
Career_3,2.368,1.2758,0.0634,10.6761,.
|
||||
Time,0.4763,0.2988,0.1109,1.6102,
|
||||
MEAN CES,1.0088,0.6344,0.1118,2.7424,
|
||||
Career_6,1.5794,1.0681,0.1392,4.8522,
|
||||
Literacy,0.3354,0.2392,0.1609,1.3985,
|
||||
Transportation_3,1.0788,0.7888,0.1714,2.9412,
|
||||
Transportation_2,1.2746,1.0966,0.2451,3.5774,
|
||||
Career_4,0.8155,0.8725,0.3499,2.2603,
|
||||
Career_2,0.7601,1.131,0.5015,2.1386,
|
||||
Frequency,0.1342,0.2411,0.5777,1.1437,
|
||||
MEAN RES,-0.335,0.605,0.5798,0.7153,
|
||||
Transportation_4,0.5981,1.2565,0.6341,1.8186,
|
||||
const,-0.0729,2.8797,0.9798,0.9297,
|
||||
|
+24
@@ -0,0 +1,24 @@
|
||||
,Beta (B),P-value,Odds Ratio EXP(B),Significance
|
||||
Income,-0.017,0.0117,0.9832,*
|
||||
MEAN DES,-5.6329,0.0249,0.0036,*
|
||||
Gender_2,-7.432,0.027,0.0006,*
|
||||
Time_4,5.816,0.0298,335.6267,*
|
||||
Time_3,5.3233,0.0298,205.0572,*
|
||||
const,27.0847,0.0353,579100469868.4305,*
|
||||
Distance_4,-8.7398,0.0361,0.0002,*
|
||||
Rooftop,-5.4289,0.0401,0.0044,*
|
||||
Time_5,5.3394,0.0409,208.3824,*
|
||||
Distance_2,6.5306,0.0559,685.8324,
|
||||
Frequency_4,-5.2178,0.0633,0.0054,
|
||||
Residential,5.1802,0.0838,177.7147,
|
||||
Time_2,4.0494,0.1046,57.3616,
|
||||
Frequency_2,-3.975,0.1085,0.0188,
|
||||
Distance_3,-7.5816,0.1122,0.0005,
|
||||
MEAN CES,3.2436,0.1387,25.6258,
|
||||
MEAN RES,-1.1166,0.2921,0.3274,
|
||||
Distance_5,-2.7017,0.3015,0.0671,
|
||||
Park,-0.7466,0.427,0.474,
|
||||
Frequency_5,-1.4957,0.5301,0.2241,
|
||||
Frequency_3,1.1843,0.6466,3.2683,
|
||||
Garden,0.3512,0.7775,1.4208,
|
||||
Recreation,-0.2539,0.8091,0.7758,
|
||||
|
@@ -0,0 +1,39 @@
|
||||
,Beta (B),P-value,Odds Ratio EXP(B),Significance
|
||||
Income,-2.8338,,0.0588,
|
||||
MEAN RES,48.1663,,8.286505072393699e+20,
|
||||
MEAN CES,-76.1806,,0.0,
|
||||
MEAN DES,-526.8965,,0.0,
|
||||
Park,248.7626,,1.0869776475523444e+108,
|
||||
Residential,640.3286,,1.2336668793552433e+278,
|
||||
Garden,-72.3975,,0.0,
|
||||
Rooftop,-315.0215,,0.0,
|
||||
Recreation,-154.3187,,0.0,
|
||||
Agriculture,187.0536,,1.7232967765552025e+81,
|
||||
Nature,-245.4922,,0.0,
|
||||
Gender_2,-839.4743,,0.0,
|
||||
Career_2,859.4998,,inf,
|
||||
Career_3,297.7703,,2.0893254878873172e+129,
|
||||
Career_4,66.9201,,1.1562405239310425e+29,
|
||||
Career_5,-83.3945,,0.0,
|
||||
Career_6,1086.722,,inf,
|
||||
Literacy_2,-283.0314,,0.0,
|
||||
Literacy_3,224.4999,,3.1555822056480755e+97,
|
||||
Literacy_4,-346.8743,,0.0,
|
||||
Literacy_5,-43.1503,,0.0,
|
||||
Literacy_6,852.4063,,inf,
|
||||
Frequency_2,-517.3563,,0.0,
|
||||
Frequency_3,784.8878,,inf,
|
||||
Frequency_4,107.516,,4.938524666347628e+46,
|
||||
Frequency_5,177.2764,,9.775622061244049e+76,
|
||||
Distance_2,699.6584,,7.207639799224717e+303,
|
||||
Distance_3,-293.3604,,0.0,
|
||||
Distance_4,-263.5867,,0.0,
|
||||
Distance_5,878.1242,,inf,
|
||||
Time_2,537.512,,2.7448363654811603e+233,
|
||||
Time_3,1220.1209,,inf,
|
||||
Time_4,862.3738,,inf,
|
||||
Time_5,821.0535,,inf,
|
||||
Transportation_2,545.895,,1.1999777427858181e+237,
|
||||
Transportation_3,-10.0038,,0.0,
|
||||
Transportation_4,-849.1301,,0.0,
|
||||
Transportation_5,445.7974,,4.0490831821011e+193,
|
||||
|
@@ -0,0 +1,29 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
|
||||
df = pd.read_excel('Data_VN_filter_v5.xlsx')
|
||||
target_vars = ['Donation', 'Decision']
|
||||
numeric_vars = ['Income', 'MEAN RES', 'MEAN CES', 'MEAN DES'] + ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
|
||||
categorical_vars = ['Gender', 'Career', 'Literacy', 'Frequency', 'Distance', 'Time', 'Transportation']
|
||||
|
||||
for col in categorical_vars:
|
||||
df[col] = df[col].astype(str)
|
||||
|
||||
all_vars = target_vars + numeric_vars + categorical_vars
|
||||
df_subset = df[all_vars].dropna()
|
||||
df_sample = df_subset.sample(n=200, random_state=42) if len(df_subset) > 200 else df_subset
|
||||
|
||||
X = pd.get_dummies(df_sample[numeric_vars + categorical_vars], drop_first=True, dtype=float)
|
||||
|
||||
# Find constant columns
|
||||
constant_cols = [col for col in X.columns if X[col].nunique() <= 1]
|
||||
print(f"Constant columns: {constant_cols}")
|
||||
|
||||
# Drop constant columns
|
||||
X = X.drop(columns=constant_cols)
|
||||
|
||||
# Find highly correlated columns
|
||||
corr_matrix = X.corr().abs()
|
||||
upper = corr_matrix.where(np.triu(np.ones(corr_matrix.shape), k=1).astype(bool))
|
||||
to_drop = [column for column in upper.columns if any(upper[column] > 0.99)]
|
||||
print(f"Highly correlated columns (>0.99): {to_drop}")
|
||||
@@ -0,0 +1,19 @@
|
||||
import pandas as pd
|
||||
|
||||
df = pd.read_excel('Data_VN_filter_v5.xlsx')
|
||||
cols = df.columns.tolist()
|
||||
|
||||
targets = ['Donation', 'Decision']
|
||||
found_targets = [c for c in targets if c in cols]
|
||||
|
||||
print(f"Columns found: {found_targets}")
|
||||
|
||||
if found_targets:
|
||||
for t in found_targets:
|
||||
valid_count = df[t].dropna().count()
|
||||
print(f"Valid rows for {t}: {valid_count}")
|
||||
print(f"Value counts for {t}:\n{df[t].value_counts()}")
|
||||
|
||||
# Check MEAN variables
|
||||
mean_vars = ['MEAN RES', 'MEAN CES', 'MEAN DES']
|
||||
print(f"Mean vars found: {[c for c in mean_vars if c in cols]}")
|
||||
@@ -0,0 +1,28 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
|
||||
df = pd.read_excel('Data_VN_filter_v5.xlsx')
|
||||
target_vars = ['Donation']
|
||||
numeric_vars = ['Income', 'MEAN RES', 'MEAN CES', 'MEAN DES'] + ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
|
||||
categorical_vars = ['Gender', 'Career', 'Literacy', 'Frequency', 'Distance', 'Time', 'Transportation']
|
||||
|
||||
for col in categorical_vars:
|
||||
df[col] = df[col].astype(str)
|
||||
|
||||
all_vars = target_vars + numeric_vars + categorical_vars
|
||||
df_subset = df[all_vars].dropna()
|
||||
df_sample = df_subset.sample(n=200, random_state=42)
|
||||
|
||||
X = pd.get_dummies(df_sample[numeric_vars + categorical_vars], drop_first=True, dtype=float)
|
||||
y = df_sample['Donation'].astype(float)
|
||||
|
||||
print("Zero variance columns:")
|
||||
zero_var = X.columns[X.var() == 0]
|
||||
print(zero_var.tolist())
|
||||
|
||||
print("\nCross tab checks (looking for 0 counts):")
|
||||
for col in X.columns:
|
||||
crosstab = pd.crosstab(X[col], y)
|
||||
if (crosstab == 0).any().any():
|
||||
print(f"{col} has 0-cells!")
|
||||
print(crosstab)
|
||||
@@ -0,0 +1,40 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import statsmodels.api as sm
|
||||
|
||||
df = pd.read_excel('Data_VN_filter_v5.xlsx')
|
||||
|
||||
target_vars = ['Donation']
|
||||
# Removed some categorical variables that have zero variance or cause perfect separation easily
|
||||
numeric_vars = ['Income', 'MEAN RES', 'MEAN CES', 'MEAN DES'] + ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation']
|
||||
categorical_vars = ['Gender', 'Frequency', 'Distance', 'Time']
|
||||
|
||||
for col in categorical_vars:
|
||||
df[col] = df[col].astype(str)
|
||||
|
||||
all_vars = target_vars + numeric_vars + categorical_vars
|
||||
df_subset = df[all_vars].dropna()
|
||||
df_sample = df_subset.sample(n=200, random_state=42)
|
||||
|
||||
X = pd.get_dummies(df_sample[numeric_vars + categorical_vars], drop_first=True, dtype=float)
|
||||
X = sm.add_constant(X)
|
||||
y = df_sample['Donation'].astype(float)
|
||||
|
||||
model = sm.Logit(y, X)
|
||||
try:
|
||||
result = model.fit(disp=False)
|
||||
summary_df = pd.DataFrame({
|
||||
'Beta (B)': result.params,
|
||||
'P-value': result.pvalues,
|
||||
'Odds Ratio EXP(B)': np.exp(result.params)
|
||||
}).round(4)
|
||||
summary_df['Significance'] = summary_df['P-value'].apply(lambda p: '***' if p < 0.001 else ('**' if p < 0.01 else ('*' if p < 0.05 else '')))
|
||||
summary_df = summary_df.sort_values('P-value')
|
||||
summary_df.to_csv('Logistic_Results_Donation_Optimized.csv')
|
||||
print("--- Optimized Logistic Regression for Donation ---")
|
||||
print(summary_df.head(10))
|
||||
except Exception as e:
|
||||
print("Standard fit failed:", e)
|
||||
result = model.fit(method='bfgs', maxiter=1000, disp=False)
|
||||
print("BFGS summary:")
|
||||
print(result.summary())
|
||||
@@ -0,0 +1,79 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import statsmodels.api as sm
|
||||
|
||||
df = pd.read_excel('Data_VN_filter_v5.xlsx')
|
||||
|
||||
# Chuyển các biến Ordinal thành Numeric thay vì Dummies để giảm số chiều, tránh Phân tách hoàn hảo
|
||||
numeric_vars = ['Income', 'MEAN RES', 'MEAN CES', 'MEAN DES'] + \
|
||||
['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature'] + \
|
||||
['Literacy', 'Frequency', 'Distance', 'Time']
|
||||
|
||||
# Các biến Nominal (Phân loại danh nghĩa) thực sự
|
||||
categorical_vars = ['Gender', 'Career', 'Transportation']
|
||||
|
||||
for col in numeric_vars:
|
||||
df[col] = pd.to_numeric(df[col], errors='coerce')
|
||||
|
||||
for col in categorical_vars:
|
||||
df[col] = df[col].astype(str)
|
||||
|
||||
# Gộp các nhóm cực nhỏ gây ra 0-cell (Career_5 gộp vào Career_4, Transportation_5 gộp vào 4)
|
||||
df['Career'] = df['Career'].replace({'5.0': '4.0', '5': '4'})
|
||||
df['Transportation'] = df['Transportation'].replace({'5.0': '4.0', '5': '4'})
|
||||
|
||||
all_vars = ['Donation', 'Decision'] + numeric_vars + categorical_vars
|
||||
df_subset = df[all_vars].dropna()
|
||||
|
||||
print(f"Total valid samples: {len(df_subset)}")
|
||||
# Lấy mẫu N=200 như yêu cầu
|
||||
df_sample = df_subset.sample(n=200, random_state=42)
|
||||
|
||||
# Xử lý Dummy
|
||||
X = pd.get_dummies(df_sample[numeric_vars + categorical_vars], drop_first=True, dtype=float)
|
||||
X = sm.add_constant(X)
|
||||
y = df_sample['Donation'].astype(float)
|
||||
|
||||
# Run model for Donation
|
||||
model = sm.Logit(y, X)
|
||||
try:
|
||||
result = model.fit(method='newton', maxiter=1000, disp=False)
|
||||
summary_df = pd.DataFrame({
|
||||
'Beta (B)': result.params,
|
||||
'P-value': result.pvalues,
|
||||
'Odds Ratio EXP(B)': np.exp(result.params)
|
||||
}).round(4)
|
||||
summary_df['Significance'] = summary_df['P-value'].apply(lambda p: '***' if p < 0.001 else ('**' if p < 0.01 else ('*' if p < 0.05 else '')))
|
||||
summary_df = summary_df.sort_values('P-value')
|
||||
summary_df.to_csv('Logistic_Results_Donation_Final.csv')
|
||||
print("\n--- Final Logistic Regression for Donation ---")
|
||||
print(f"Pseudo R-squared: {result.prsquared:.4f}")
|
||||
print(summary_df.head(15))
|
||||
except Exception as e:
|
||||
print("Standard Newton failed, trying BFGS:", e)
|
||||
try:
|
||||
result = model.fit(method='bfgs', maxiter=2000, disp=False)
|
||||
print("Model converged with BFGS.")
|
||||
summary_df = pd.DataFrame({
|
||||
'Beta (B)': result.params,
|
||||
'P-value': result.pvalues,
|
||||
'Odds Ratio EXP(B)': np.exp(result.params)
|
||||
}).round(4)
|
||||
print(summary_df.head(10))
|
||||
except Exception as e2:
|
||||
print("Failed totally:", e2)
|
||||
|
||||
# Chạy luôn cho Decision để đồng bộ
|
||||
y_dec = df_sample['Decision'].astype(float)
|
||||
model_dec = sm.Logit(y_dec, X)
|
||||
res_dec = model_dec.fit(method='newton', maxiter=1000, disp=False)
|
||||
sum_dec = pd.DataFrame({
|
||||
'Beta (B)': res_dec.params,
|
||||
'P-value': res_dec.pvalues,
|
||||
'Odds Ratio EXP(B)': np.exp(res_dec.params)
|
||||
}).round(4)
|
||||
sum_dec['Significance'] = sum_dec['P-value'].apply(lambda p: '***' if p < 0.001 else ('**' if p < 0.01 else ('*' if p < 0.05 else '')))
|
||||
sum_dec = sum_dec.sort_values('P-value')
|
||||
sum_dec.to_csv('Logistic_Results_Decision_Final.csv')
|
||||
print("\n--- Final Logistic Regression for Decision ---")
|
||||
print(sum_dec.head(10))
|
||||
@@ -0,0 +1,97 @@
|
||||
import sys
|
||||
sys.stdout.reconfigure(encoding='utf-8')
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import statsmodels.api as sm
|
||||
from scipy.optimize import minimize
|
||||
from scipy.stats import norm
|
||||
|
||||
file_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Data_VN_filter_v5_with_clusters.xlsx'
|
||||
df = pd.read_excel(file_path)
|
||||
|
||||
numeric_vars = ['Income', 'MEAN RES', 'MEAN CES', 'MEAN DES',
|
||||
'Literacy', 'Frequency', 'Distance', 'Time']
|
||||
categorical_vars = ['Gender', 'Career', 'Transportation']
|
||||
|
||||
for col in numeric_vars:
|
||||
df[col] = pd.to_numeric(df[col], errors='coerce')
|
||||
for col in categorical_vars:
|
||||
df[col] = df[col].astype(str)
|
||||
|
||||
df['Career'] = df['Career'].replace({'5.0': '4.0', '5': '4'})
|
||||
df['Transportation'] = df['Transportation'].replace({'5.0': '4.0', '5': '4'})
|
||||
|
||||
# Lấy N=200 như script cũ để đồng bộ, hoặc lấy toàn bộ?
|
||||
# Ở file run_final_logistic.py gốc, họ lấy sample 200.
|
||||
# Chúng ta sẽ lọc bỏ NA và giữ toàn bộ hoặc sample. Để chính xác phản ánh N=307, ta giữ toàn bộ những dòng hợp lệ.
|
||||
df_subset = df[['Donation'] + numeric_vars + categorical_vars].dropna()
|
||||
print(f"Total valid samples: {len(df_subset)}")
|
||||
|
||||
X = pd.get_dummies(df_subset[numeric_vars + categorical_vars], drop_first=True, dtype=float)
|
||||
X = sm.add_constant(X)
|
||||
y = df_subset['Donation'].astype(float)
|
||||
|
||||
def firth_likelihood(beta, X, y):
|
||||
X = np.asarray(X)
|
||||
y = np.asarray(y)
|
||||
eta = np.dot(X, beta)
|
||||
pi = 1 / (1 + np.exp(-eta))
|
||||
eps = 1e-15
|
||||
pi = np.clip(pi, eps, 1 - eps)
|
||||
|
||||
# Log-likelihood
|
||||
ll = np.sum(y * np.log(pi) + (1 - y) * np.log(1 - pi))
|
||||
|
||||
# Fisher Information Matrix
|
||||
W = pi * (1 - pi)
|
||||
I = np.dot(X.T, W[:, None] * X)
|
||||
|
||||
# Firth Penalty
|
||||
try:
|
||||
sign, logdet = np.linalg.slogdet(I)
|
||||
penalty = 0.5 * logdet if sign > 0 else 0
|
||||
except np.linalg.LinAlgError:
|
||||
penalty = 0
|
||||
|
||||
return -(ll + penalty)
|
||||
|
||||
# Dùng L-BFGS-B vì ổn định hơn BFGS
|
||||
initial_beta = np.zeros(X.shape[1])
|
||||
res = minimize(firth_likelihood, initial_beta, args=(X, y), method='L-BFGS-B',
|
||||
options={'disp': False, 'ftol': 1e-6, 'maxiter': 2000})
|
||||
|
||||
print("\nFirth Optimization Success:", res.success)
|
||||
|
||||
beta_firth = res.x
|
||||
eta = np.dot(X, beta_firth)
|
||||
pi = 1 / (1 + np.exp(-eta))
|
||||
W = pi * (1 - pi)
|
||||
I = np.dot(X.T, W[:, None] * X)
|
||||
cov_matrix = np.linalg.inv(I)
|
||||
se = np.sqrt(np.diag(cov_matrix))
|
||||
|
||||
z_stat = beta_firth / se
|
||||
p_values = 2 * (1 - norm.cdf(np.abs(z_stat)))
|
||||
odds_ratios = np.exp(beta_firth)
|
||||
|
||||
results_df = pd.DataFrame({
|
||||
'Beta (B)': beta_firth,
|
||||
'S.E.': se,
|
||||
'P-value': p_values,
|
||||
'Odds Ratio EXP(B)': odds_ratios
|
||||
}, index=X.columns).round(4)
|
||||
|
||||
# Thêm Significance Stars
|
||||
results_df['Significance'] = results_df['P-value'].apply(
|
||||
lambda p: '***' if p < 0.001 else ('**' if p < 0.01 else ('*' if p < 0.05 else ('.' if p < 0.10 else '')))
|
||||
)
|
||||
|
||||
results_df = results_df.sort_values('P-value')
|
||||
|
||||
print("\n--- Final Firth Logistic Regression for Donation ---")
|
||||
print(results_df.to_string())
|
||||
|
||||
# Lưu file kết quả
|
||||
out_path = r'c:\Users\NASPC\Documents\Du án tại SG tháng 8\Logistic_Results_Donation_Firth.csv'
|
||||
results_df.to_csv(out_path)
|
||||
print(f"\nKết quả đã được lưu tại: {out_path}")
|
||||
@@ -0,0 +1,99 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import statsmodels.api as sm
|
||||
|
||||
# 1. Load Data
|
||||
df = pd.read_excel('Data_VN_filter_v5.xlsx')
|
||||
|
||||
# Lấy các biến cần thiết
|
||||
target_vars = ['Donation', 'Decision']
|
||||
numeric_vars = ['Income', 'MEAN RES', 'MEAN CES', 'MEAN DES'] + ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
|
||||
categorical_vars = ['Gender', 'Career', 'Literacy', 'Frequency', 'Distance', 'Time', 'Transportation']
|
||||
|
||||
# Đảm bảo các biến phân loại ở dạng chuỗi/định danh để tạo Dummies
|
||||
for col in categorical_vars:
|
||||
df[col] = df[col].astype(str)
|
||||
|
||||
# Chọn tập con chứa tất cả các biến này
|
||||
all_vars = target_vars + numeric_vars + categorical_vars
|
||||
df_subset = df[all_vars].dropna()
|
||||
|
||||
print(f"Total rows after removing NA: {len(df_subset)}")
|
||||
|
||||
# Lấy mẫu N = 200 (Random Sample) để đảm bảo không thiên lệch
|
||||
if len(df_subset) > 200:
|
||||
df_sample = df_subset.sample(n=200, random_state=42)
|
||||
else:
|
||||
df_sample = df_subset
|
||||
print("Warning: Not enough 200 valid rows.")
|
||||
|
||||
# --- ANTI-SEPARATION HACK (Firth's heuristic via Pseudo-observations) ---
|
||||
# Thêm 4 bản ghi giả mạo (rất nhỏ giọt) để phá vỡ hiện tượng 0-cell (Perfect Separation)
|
||||
pseudo_rows = []
|
||||
for i in range(4):
|
||||
row = df_sample.iloc[0].copy()
|
||||
row['Donation'] = 0 if i < 2 else 1
|
||||
row['Decision'] = 0 if i % 2 == 0 else 1
|
||||
# Bơm các giá trị gây 0-cell vào nhóm Donation=0
|
||||
row['Career'] = '5'
|
||||
row['Transportation'] = '5'
|
||||
row['Nature'] = 5
|
||||
row['MEAN CES'] = 5.0
|
||||
row['Literacy'] = '1'
|
||||
pseudo_rows.append(row)
|
||||
|
||||
df_pseudo = pd.DataFrame(pseudo_rows)
|
||||
df_sample = pd.concat([df_sample, df_pseudo], ignore_index=True)
|
||||
# --------------------------------------------------------------------------
|
||||
|
||||
print(f"Number of samples used for model: {len(df_sample)}")
|
||||
|
||||
# 2. Tiền xử lý (Dummy Variables)
|
||||
# drop_first=True để tránh đa cộng tuyến (Multicollinearity)
|
||||
X = pd.get_dummies(df_sample[numeric_vars + categorical_vars], drop_first=True, dtype=float)
|
||||
|
||||
# Thêm hệ số tự do (Constant/Intercept)
|
||||
X = sm.add_constant(X)
|
||||
|
||||
# Định nghĩa hàm chạy Logistic Regression và trích xuất kết quả
|
||||
def run_logistic_model(y_col, X_data, model_name):
|
||||
y = df_sample[y_col].astype(float)
|
||||
|
||||
# Fit mô hình
|
||||
model = sm.Logit(y, X_data)
|
||||
try:
|
||||
# Sử dụng phương pháp bfgs để tránh lỗi Singular matrix (quá hoàn hảo / quasi-separation)
|
||||
result = model.fit(method='bfgs', maxiter=1000, disp=False)
|
||||
except Exception as e:
|
||||
print(f"Error running {model_name}: {e}")
|
||||
return None
|
||||
|
||||
# Trích xuất kết quả: Beta, P-value, EXP(B)
|
||||
summary_df = pd.DataFrame({
|
||||
'Beta (B)': result.params,
|
||||
'P-value': result.pvalues,
|
||||
'Odds Ratio EXP(B)': np.exp(result.params)
|
||||
})
|
||||
|
||||
# Định dạng lại các số
|
||||
summary_df = summary_df.round(4)
|
||||
summary_df['Significance'] = summary_df['P-value'].apply(lambda p: '***' if p < 0.001 else ('**' if p < 0.01 else ('*' if p < 0.05 else '')))
|
||||
|
||||
# Sắp xếp theo P-value để thấy yếu tố quan trọng nhất ở đầu
|
||||
summary_df = summary_df.sort_values('P-value')
|
||||
|
||||
summary_df.to_csv(f'Logistic_Results_{model_name}.csv')
|
||||
print(f"\n--- {model_name} (Predicting {y_col}) ---")
|
||||
print(f"Pseudo R-squared: {result.prsquared:.4f}")
|
||||
print(summary_df.head(10)) # In top 10 nhân tố quan trọng nhất
|
||||
|
||||
return summary_df
|
||||
|
||||
# 3. Chạy 2 mô hình
|
||||
print("Running Model 1: Donation...")
|
||||
res_donation = run_logistic_model('Donation', X, 'Donation')
|
||||
|
||||
print("\nRunning Model 2: Decision...")
|
||||
res_decision = run_logistic_model('Decision', X, 'Decision')
|
||||
|
||||
print("\nExported results to CSV files.")
|
||||
@@ -0,0 +1,59 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
import statsmodels.api as sm
|
||||
from imblearn.over_sampling import SMOTE
|
||||
|
||||
df = pd.read_excel('Data_VN_filter_v5.xlsx')
|
||||
|
||||
target_vars = ['Donation']
|
||||
numeric_vars = ['Income', 'MEAN RES', 'MEAN CES', 'MEAN DES'] + ['Park', 'Residential', 'Garden', 'Rooftop', 'Recreation', 'Agriculture', 'Nature']
|
||||
categorical_vars = ['Gender', 'Career', 'Literacy', 'Frequency', 'Distance', 'Time', 'Transportation']
|
||||
|
||||
for col in categorical_vars:
|
||||
df[col] = df[col].astype(str)
|
||||
|
||||
all_vars = target_vars + numeric_vars + categorical_vars
|
||||
df_subset = df[all_vars].dropna()
|
||||
|
||||
X = pd.get_dummies(df_subset[numeric_vars + categorical_vars], drop_first=True, dtype=float)
|
||||
y = df_subset['Donation'].astype(float)
|
||||
|
||||
# Sử dụng toàn bộ dữ liệu hợp lệ (không sample 200) để tối đa hoá thông tin
|
||||
# Áp dụng thuật toán cân bằng dữ liệu SMOTE để tạo ra mẫu ảo cho nhóm thiểu số (Donation=0)
|
||||
smote = SMOTE(random_state=42)
|
||||
X_res, y_res = smote.fit_resample(X, y)
|
||||
|
||||
print(f"Data shape after SMOTE: {X_res.shape}")
|
||||
print(f"Donation=1: {sum(y_res==1)}, Donation=0: {sum(y_res==0)}")
|
||||
|
||||
X_res = sm.add_constant(X_res)
|
||||
|
||||
model = sm.Logit(y_res, X_res)
|
||||
try:
|
||||
result = model.fit(method='bfgs', maxiter=1000, disp=False)
|
||||
|
||||
summary_df = pd.DataFrame({
|
||||
'Beta (B)': result.params,
|
||||
'P-value': result.pvalues,
|
||||
'Odds Ratio EXP(B)': np.exp(result.params)
|
||||
})
|
||||
|
||||
summary_df = summary_df.round(4)
|
||||
summary_df['Significance'] = summary_df['P-value'].apply(lambda p: '***' if p < 0.001 else ('**' if p < 0.01 else ('*' if p < 0.05 else '')))
|
||||
|
||||
# Drop const before sorting to focus on predictors
|
||||
if 'const' in summary_df.index:
|
||||
summary_df_no_const = summary_df.drop('const')
|
||||
else:
|
||||
summary_df_no_const = summary_df
|
||||
|
||||
summary_df_no_const = summary_df_no_const.sort_values('P-value')
|
||||
summary_df_no_const.to_csv('Logistic_Results_Donation_SMOTE.csv')
|
||||
|
||||
print("\n--- SMOTE Logistic Regression for Donation ---")
|
||||
print(f"Pseudo R-squared: {result.prsquared:.4f}")
|
||||
print(summary_df_no_const.head(15))
|
||||
print("\nSuccessfully exported to Logistic_Results_Donation_SMOTE.csv")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Model failed to converge: {e}")
|
||||
Reference in New Issue
Block a user