Repository navigation
Expand file tree
/
Copy pathTrAdaboostreg.py
More file actions
161 lines (137 loc) · 6.15 KB
/
Copy pathTrAdaboostreg.py
File metadata and controls
161 lines (137 loc) · 6.15 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
# code by chenchiwei
# -*- coding: UTF-8 -*-
import numpy as np
from sklearn import tree
from sklearn import svm
import math
from sklearn.metrics import r2_score
from sklearn.tree import DecisionTreeRegressor
def stable_cumsum(arr, axis=None, rtol=1e-05, atol=1e-08):
out = np.cumsum(arr, axis=axis, dtype=np.float64)
return out
# H 测试样本分类结果
# TrainS 原训练样本 np数组
# TrainA 辅助训练样本
# LabelS 原训练样本标签
# LabelA 辅助训练样本标签
# Test 测试样本
# N 迭代次数
def tradaboost(trans_S, trans_A, label_S, label_A, test, y_target_test,N,islog):
trans_data = np.concatenate((trans_A, trans_S), axis=0)
trans_label = np.concatenate((label_A, label_S), axis=0)
row_A = trans_A.shape[0]
row_S = trans_S.shape[0]
row_T = test.shape[0]
test_data = np.concatenate((trans_data, test), axis=0)
# 初始化权重
weights_A = np.ones([row_A, 1])/row_A
weights_S = np.ones([row_S, 1])/row_S
weights = np.concatenate((weights_A, weights_S), axis=0)
bata = 1 / (1 + np.sqrt(2 * np.log(row_A/N)))
# 存储每次迭代的标签和bata值
bata_T = np.zeros([1, N])
result_label = np.ones([row_A + row_S + row_T, N])
predict = np.zeros([row_T])
#print ('params initial finished.')
for i in range(N):
#将权重向量归一化
P = calculate_P(weights)
result_label[:, i] = train_classify(trans_data, trans_label,
test_data, weights,test,y_target_test,islog)
temp0 = np.abs(result_label[:row_A + row_S, i] - trans_label)
error_max0 = temp0.max()
temp = np.abs(result_label[row_A:row_A + row_S, i] - label_S)
error_max = temp.max()
if error_max0==0.0 or error_max==0.0:
N=i;
break
temp2 = np.abs(result_label[:row_A, i] - label_A)
error_max2 = temp2.max()
error_rate = 0.0
for j in range(row_A, row_A + row_S):
error_rate += (weights[j] * ((abs(result_label[j, i] - trans_label[j])/error_max0)))
error_rate = error_rate / sum(weights[row_A:])
if error_rate >= 0.5:
error_rate = 0.499;
if error_rate == 0:
error_rate=0.001
bata_T[0, i] = error_rate / (1 - error_rate)
for j in range(row_S):
weights[row_A + j] = weights[row_A + j] * np.power(bata_T[0, i], -(
np.abs(result_label[row_A + j, i] - label_S[j]) / error_max0))
# 调整辅域样本权重
for j in range(row_A):
if islog:
if (abs(result_label[j, i] - label_A[j]) >0):#0.02872
weights[j] = weights[j] * np.power(bata, np.abs((result_label[j, i] - label_A[j])/error_max0))
else:
weights[j] = weights[j] * np.power(bata, np.abs((result_label[j, i] - label_A[j]) / error_max0))
# bata_T[0,:]=bata_T[0,:]/np.sum(bata_T[0,:])
#
predictions=result_label[row_A + row_S:,int(np.ceil(N / 2)):N]
# Sort the predictions
sorted_idx = np.argsort(predictions, axis=1)
# Find index of median prediction for each sample
bata_T = np.log(1/bata_T[0, int(np.ceil(N / 2)):N])
#bata_T = bata_T[0, int(np.ceil(N / 2)):N]
bata_T[:] = bata_T[:] / np.sum(bata_T[:])
weight_cdf = stable_cumsum(bata_T[sorted_idx], axis=1)
median_or_above = weight_cdf >= 0.5 * weight_cdf[:, -1][:, np.newaxis]
median_idx = median_or_above.argmax(axis=1)
median_estimators = sorted_idx[np.arange(test.shape[0]), median_idx]
# Return median predictions
return predictions[np.arange(test.shape[0]), median_estimators]
#
# for i in range(row_T):
# # 跳过训练数据的标签
# # predict[i]=np.median(
# # result_label[row_A + row_S + i, :] * np.log(1 / bata_T[0, :]))
# # predict[i] = np.sum(
# # result_label[row_A + row_S + i, :] * (1-bata_T[0, :]))
#
# predict[i] = weighted_median(result_label[row_A + row_S + i,int(np.ceil(N / 2)):N],
# np.log(1 / bata_T[0,int(np.ceil(N / 2)):N]))
# return predict
def calculate_P(weights):
total = np.sum(weights)
return weights/total
def train_classify(trans_data, trans_label, test_data, P,test,y_target_test,islog,):
# if islog:
# clf = svm.SVR(C=100)
# else:
clf = DecisionTreeRegressor(max_depth=3)
clf.fit(trans_data, trans_label, sample_weight=P[:, 0])
return clf.predict(test_data)
def weighted_median(values, weights):
''' compute the weighted median of values list. The
weighted median is computed as follows:
1- sort both lists (values and weights) based on values.
2- select the 0.5 point from the weights and return the corresponding values as results
e.g. values = [1, 3, 0] and weights=[0.1, 0.3, 0.6] assuming weights are probabilities.
sorted values = [0, 1, 3] and corresponding sorted weights = [0.6, 0.1, 0.3] the 0.5 point on
weight corresponds to the first item which is 0. so the weighted median is 0.'''
#convert the weights into probabilities
sum_weights = sum(weights)
weights = np.array([(w*1.0)/sum_weights for w in weights])
#sort values and weights based on values
values = np.array(values)
sorted_indices = np.argsort(values)
values_sorted = values[sorted_indices]
weights_sorted = weights[sorted_indices]
#select the median point
it = np.nditer(weights_sorted, flags=['f_index'])
accumulative_probability = 0
median_index = -1
while not it.finished:
accumulative_probability += it[0]
if accumulative_probability > 0.5:
median_index = it.index
return values_sorted[median_index]
elif accumulative_probability == 0.5:
median_index = it.index
it.iternext()
next_median_index = it.index
return np.mean(values_sorted[[median_index, next_median_index]])
it.iternext()
return values_sorted[median_index]
from sklearn.ensemble import AdaBoostRegressor