Eskair commited on
Commit
6e47529
·
0 Parent(s):

deploy clean space v2

Browse files
.gitattributes ADDED
@@ -0,0 +1,309 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ src/data/prepared/test_fix_profile/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
37
+ src/data/prepared/test_fix_profile/page_images/page_002.png filter=lfs diff=lfs merge=lfs -text
38
+ src/data/prepared/test_fix_profile/page_images/page_003.png filter=lfs diff=lfs merge=lfs -text
39
+ src/data/prepared/test_fix_profile/page_images/page_004.png filter=lfs diff=lfs merge=lfs -text
40
+ src/data/prepared/test_fix_profile/page_images/page_005.png filter=lfs diff=lfs merge=lfs -text
41
+ src/data/prepared/test_fix_profile/page_images/page_006.png filter=lfs diff=lfs merge=lfs -text
42
+ src/data/prepared/test_fix_profile/page_images/page_007.png filter=lfs diff=lfs merge=lfs -text
43
+ src/data/prepared/test_fix_profile/page_images/page_008.png filter=lfs diff=lfs merge=lfs -text
44
+ src/data/prepared/test_fix_profile/page_images/page_009.png filter=lfs diff=lfs merge=lfs -text
45
+ src/data/prepared/test_fix_profile/page_images/page_010.png filter=lfs diff=lfs merge=lfs -text
46
+ src/data/prepared/test_fix_profile/page_images/page_011.png filter=lfs diff=lfs merge=lfs -text
47
+ src/data/prepared/test_fix_profile/page_images/page_012.png filter=lfs diff=lfs merge=lfs -text
48
+ src/data/prepared/test_fix_profile/page_images/page_013.png filter=lfs diff=lfs merge=lfs -text
49
+ src/data/prepared/test_fix_profile/page_images/page_014.png filter=lfs diff=lfs merge=lfs -text
50
+ src/data/prepared/test_fix_profile/page_images/page_015.png filter=lfs diff=lfs merge=lfs -text
51
+ src/data/prepared/test_fix_profile/page_images/page_016.png filter=lfs diff=lfs merge=lfs -text
52
+ src/data/prepared/test_fix_profile/page_images/page_017.png filter=lfs diff=lfs merge=lfs -text
53
+ src/data/prepared/test_fix_profile/page_images/page_018.png filter=lfs diff=lfs merge=lfs -text
54
+ src/data/prepared/test_fix_profile/page_images/page_019.png filter=lfs diff=lfs merge=lfs -text
55
+ src/data/prepared/test_fix_profile/page_images/page_020.png filter=lfs diff=lfs merge=lfs -text
56
+ src/data/prepared/test_fix_profile/page_images/page_021.png filter=lfs diff=lfs merge=lfs -text
57
+ src/data/prepared/test_fix_profile/page_images/page_022.png filter=lfs diff=lfs merge=lfs -text
58
+ src/data/prepared/test_fix_profile/page_images/page_023.png filter=lfs diff=lfs merge=lfs -text
59
+ src/data/prepared/test_fix_profile/page_images/page_024.png filter=lfs diff=lfs merge=lfs -text
60
+ src/data/prepared/test_fix_profile/page_images/page_025.png filter=lfs diff=lfs merge=lfs -text
61
+ src/data/prepared/test_fix_profile/page_images/page_026.png filter=lfs diff=lfs merge=lfs -text
62
+ src/data/prepared/test_fix_profile/page_images/page_027.png filter=lfs diff=lfs merge=lfs -text
63
+ src/data/prepared/test_fix_profile/page_images/page_028.png filter=lfs diff=lfs merge=lfs -text
64
+ src/data/prepared/test_fix_profile/page_images/page_029.png filter=lfs diff=lfs merge=lfs -text
65
+ src/data/prepared/test_fix_profile/page_images/page_030.png filter=lfs diff=lfs merge=lfs -text
66
+ src/data/prepared/test_fix_profile/page_images/page_031.png filter=lfs diff=lfs merge=lfs -text
67
+ src/data/prepared/test_fix_profile/page_images/page_032.png filter=lfs diff=lfs merge=lfs -text
68
+ src/data/prepared/test_fix_profile/page_images/page_033.png filter=lfs diff=lfs merge=lfs -text
69
+ src/data/prepared/test_fix_profile/page_images/page_034.png filter=lfs diff=lfs merge=lfs -text
70
+ src/data/prepared/test_fix_profile/page_images/page_035.png filter=lfs diff=lfs merge=lfs -text
71
+ src/data/prepared/test_fix_profile/page_images/page_036.png filter=lfs diff=lfs merge=lfs -text
72
+ src/data/prepared/test_fix_profile/page_images/page_037.png filter=lfs diff=lfs merge=lfs -text
73
+ src/data/prepared/test_fix_profile/page_images/page_038.png filter=lfs diff=lfs merge=lfs -text
74
+ src/data/prepared/test_fix_profile/page_images/page_039.png filter=lfs diff=lfs merge=lfs -text
75
+ src/data/prepared/test_ocr/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
76
+ src/data/prepared/test_ocr/page_images/page_002.png filter=lfs diff=lfs merge=lfs -text
77
+ src/data/prepared/test_ocr/page_images/page_003.png filter=lfs diff=lfs merge=lfs -text
78
+ src/data/prepared/test_ocr/page_images/page_004.png filter=lfs diff=lfs merge=lfs -text
79
+ src/data/prepared/test_ocr/page_images/page_005.png filter=lfs diff=lfs merge=lfs -text
80
+ src/data/prepared/test_ocr/page_images/page_006.png filter=lfs diff=lfs merge=lfs -text
81
+ src/data/prepared/test_ocr/page_images/page_007.png filter=lfs diff=lfs merge=lfs -text
82
+ src/data/prepared/test_ocr/page_images/page_008.png filter=lfs diff=lfs merge=lfs -text
83
+ src/data/prepared/test_ocr/page_images/page_009.png filter=lfs diff=lfs merge=lfs -text
84
+ src/data/prepared/test_ocr/page_images/page_010.png filter=lfs diff=lfs merge=lfs -text
85
+ src/data/prepared/test_ocr/page_images/page_011.png filter=lfs diff=lfs merge=lfs -text
86
+ src/data/prepared/test_ocr/page_images/page_012.png filter=lfs diff=lfs merge=lfs -text
87
+ src/data/prepared/test_ocr/page_images/page_013.png filter=lfs diff=lfs merge=lfs -text
88
+ src/data/prepared/test_ocr/page_images/page_014.png filter=lfs diff=lfs merge=lfs -text
89
+ src/data/prepared/test_ocr/page_images/page_015.png filter=lfs diff=lfs merge=lfs -text
90
+ src/data/prepared/test_ocr/page_images/page_016.png filter=lfs diff=lfs merge=lfs -text
91
+ src/data/prepared/test_ocr/page_images/page_017.png filter=lfs diff=lfs merge=lfs -text
92
+ src/data/prepared/test_ocr/page_images/page_018.png filter=lfs diff=lfs merge=lfs -text
93
+ src/data/prepared/test_ocr/page_images/page_019.png filter=lfs diff=lfs merge=lfs -text
94
+ src/data/prepared/test_ocr/page_images/page_020.png filter=lfs diff=lfs merge=lfs -text
95
+ src/data/prepared/test_ocr/page_images/page_021.png filter=lfs diff=lfs merge=lfs -text
96
+ src/data/prepared/test_ocr/page_images/page_022.png filter=lfs diff=lfs merge=lfs -text
97
+ src/data/prepared/test_ocr/page_images/page_023.png filter=lfs diff=lfs merge=lfs -text
98
+ src/data/prepared/test_ocr/page_images/page_024.png filter=lfs diff=lfs merge=lfs -text
99
+ src/data/prepared/test_ocr/page_images/page_025.png filter=lfs diff=lfs merge=lfs -text
100
+ src/data/prepared/test_ocr/page_images/page_026.png filter=lfs diff=lfs merge=lfs -text
101
+ src/data/prepared/test_ocr/page_images/page_027.png filter=lfs diff=lfs merge=lfs -text
102
+ src/data/prepared/test_ocr/page_images/page_028.png filter=lfs diff=lfs merge=lfs -text
103
+ src/data/prepared/test_ocr/page_images/page_029.png filter=lfs diff=lfs merge=lfs -text
104
+ src/data/prepared/test_ocr/page_images/page_030.png filter=lfs diff=lfs merge=lfs -text
105
+ src/data/prepared/test_ocr/page_images/page_031.png filter=lfs diff=lfs merge=lfs -text
106
+ src/data/prepared/test_ocr/page_images/page_032.png filter=lfs diff=lfs merge=lfs -text
107
+ src/data/prepared/test_ocr/page_images/page_033.png filter=lfs diff=lfs merge=lfs -text
108
+ src/data/prepared/test_ocr/page_images/page_034.png filter=lfs diff=lfs merge=lfs -text
109
+ src/data/prepared/test_ocr/page_images/page_035.png filter=lfs diff=lfs merge=lfs -text
110
+ src/data/prepared/test_ocr/page_images/page_036.png filter=lfs diff=lfs merge=lfs -text
111
+ src/data/prepared/test_ocr/page_images/page_037.png filter=lfs diff=lfs merge=lfs -text
112
+ src/data/prepared/test_ocr/page_images/page_038.png filter=lfs diff=lfs merge=lfs -text
113
+ src/data/prepared/test_ocr/page_images/page_039.png filter=lfs diff=lfs merge=lfs -text
114
+ src/data/prepared/test_ocr2/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
115
+ src/data/prepared/test_ocr2/page_images/page_002.png filter=lfs diff=lfs merge=lfs -text
116
+ src/data/prepared/test_ocr2/page_images/page_003.png filter=lfs diff=lfs merge=lfs -text
117
+ src/data/prepared/test_ocr2/page_images/page_004.png filter=lfs diff=lfs merge=lfs -text
118
+ src/data/prepared/test_ocr2/page_images/page_005.png filter=lfs diff=lfs merge=lfs -text
119
+ src/data/prepared/test_ocr2/page_images/page_006.png filter=lfs diff=lfs merge=lfs -text
120
+ src/data/prepared/test_ocr2/page_images/page_007.png filter=lfs diff=lfs merge=lfs -text
121
+ src/data/prepared/test_ocr2/page_images/page_008.png filter=lfs diff=lfs merge=lfs -text
122
+ src/data/prepared/test_ocr2/page_images/page_009.png filter=lfs diff=lfs merge=lfs -text
123
+ src/data/prepared/test_ocr2/page_images/page_010.png filter=lfs diff=lfs merge=lfs -text
124
+ src/data/prepared/test_ocr2/page_images/page_011.png filter=lfs diff=lfs merge=lfs -text
125
+ src/data/prepared/test_ocr2/page_images/page_012.png filter=lfs diff=lfs merge=lfs -text
126
+ src/data/prepared/test_ocr2/page_images/page_013.png filter=lfs diff=lfs merge=lfs -text
127
+ src/data/prepared/test_ocr2/page_images/page_014.png filter=lfs diff=lfs merge=lfs -text
128
+ src/data/prepared/test_ocr2/page_images/page_015.png filter=lfs diff=lfs merge=lfs -text
129
+ src/data/prepared/test_ocr2/page_images/page_016.png filter=lfs diff=lfs merge=lfs -text
130
+ src/data/prepared/test_ocr2/page_images/page_017.png filter=lfs diff=lfs merge=lfs -text
131
+ src/data/prepared/test_ocr2/page_images/page_018.png filter=lfs diff=lfs merge=lfs -text
132
+ src/data/prepared/test_ocr2/page_images/page_019.png filter=lfs diff=lfs merge=lfs -text
133
+ src/data/prepared/test_ocr2/page_images/page_020.png filter=lfs diff=lfs merge=lfs -text
134
+ src/data/prepared/test_ocr2/page_images/page_021.png filter=lfs diff=lfs merge=lfs -text
135
+ src/data/prepared/test_ocr2/page_images/page_022.png filter=lfs diff=lfs merge=lfs -text
136
+ src/data/prepared/test_ocr2/page_images/page_023.png filter=lfs diff=lfs merge=lfs -text
137
+ src/data/prepared/test_ocr2/page_images/page_024.png filter=lfs diff=lfs merge=lfs -text
138
+ src/data/prepared/test_ocr2/page_images/page_025.png filter=lfs diff=lfs merge=lfs -text
139
+ src/data/prepared/test_ocr2/page_images/page_026.png filter=lfs diff=lfs merge=lfs -text
140
+ src/data/prepared/test_ocr2/page_images/page_027.png filter=lfs diff=lfs merge=lfs -text
141
+ src/data/prepared/test_ocr2/page_images/page_028.png filter=lfs diff=lfs merge=lfs -text
142
+ src/data/prepared/test_ocr2/page_images/page_029.png filter=lfs diff=lfs merge=lfs -text
143
+ src/data/prepared/test_ocr2/page_images/page_030.png filter=lfs diff=lfs merge=lfs -text
144
+ src/data/prepared/test_ocr2/page_images/page_031.png filter=lfs diff=lfs merge=lfs -text
145
+ src/data/prepared/test_ocr2/page_images/page_032.png filter=lfs diff=lfs merge=lfs -text
146
+ src/data/prepared/test_ocr2/page_images/page_033.png filter=lfs diff=lfs merge=lfs -text
147
+ src/data/prepared/test_ocr2/page_images/page_034.png filter=lfs diff=lfs merge=lfs -text
148
+ src/data/prepared/test_ocr2/page_images/page_035.png filter=lfs diff=lfs merge=lfs -text
149
+ src/data/prepared/test_ocr2/page_images/page_036.png filter=lfs diff=lfs merge=lfs -text
150
+ src/data/prepared/test_ocr2/page_images/page_037.png filter=lfs diff=lfs merge=lfs -text
151
+ src/data/prepared/test_ocr2/page_images/page_038.png filter=lfs diff=lfs merge=lfs -text
152
+ src/data/prepared/test_ocr2/page_images/page_039.png filter=lfs diff=lfs merge=lfs -text
153
+ src/data/prepared/test_openai/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
154
+ src/data/prepared/test_openai/page_images/page_002.png filter=lfs diff=lfs merge=lfs -text
155
+ src/data/prepared/test_openai/page_images/page_003.png filter=lfs diff=lfs merge=lfs -text
156
+ src/data/prepared/test_openai/page_images/page_004.png filter=lfs diff=lfs merge=lfs -text
157
+ src/data/prepared/test_openai/page_images/page_005.png filter=lfs diff=lfs merge=lfs -text
158
+ src/data/prepared/test_openai/page_images/page_006.png filter=lfs diff=lfs merge=lfs -text
159
+ src/data/prepared/test_openai/page_images/page_007.png filter=lfs diff=lfs merge=lfs -text
160
+ src/data/prepared/test_openai/page_images/page_008.png filter=lfs diff=lfs merge=lfs -text
161
+ src/data/prepared/test_openai/page_images/page_009.png filter=lfs diff=lfs merge=lfs -text
162
+ src/data/prepared/test_openai/page_images/page_010.png filter=lfs diff=lfs merge=lfs -text
163
+ src/data/prepared/test_openai/page_images/page_011.png filter=lfs diff=lfs merge=lfs -text
164
+ src/data/prepared/test_openai/page_images/page_012.png filter=lfs diff=lfs merge=lfs -text
165
+ src/data/prepared/test_openai/page_images/page_013.png filter=lfs diff=lfs merge=lfs -text
166
+ src/data/prepared/test_openai/page_images/page_014.png filter=lfs diff=lfs merge=lfs -text
167
+ src/data/prepared/test_openai/page_images/page_015.png filter=lfs diff=lfs merge=lfs -text
168
+ src/data/prepared/test_openai/page_images/page_016.png filter=lfs diff=lfs merge=lfs -text
169
+ src/data/prepared/test_openai/page_images/page_017.png filter=lfs diff=lfs merge=lfs -text
170
+ src/data/prepared/test_openai/page_images/page_018.png filter=lfs diff=lfs merge=lfs -text
171
+ src/data/prepared/test_openai/page_images/page_019.png filter=lfs diff=lfs merge=lfs -text
172
+ src/data/prepared/test_openai/page_images/page_020.png filter=lfs diff=lfs merge=lfs -text
173
+ src/data/prepared/test_openai/page_images/page_021.png filter=lfs diff=lfs merge=lfs -text
174
+ src/data/prepared/test_openai/page_images/page_022.png filter=lfs diff=lfs merge=lfs -text
175
+ src/data/prepared/test_openai/page_images/page_023.png filter=lfs diff=lfs merge=lfs -text
176
+ src/data/prepared/test_openai/page_images/page_024.png filter=lfs diff=lfs merge=lfs -text
177
+ src/data/prepared/test_openai/page_images/page_025.png filter=lfs diff=lfs merge=lfs -text
178
+ src/data/prepared/test_openai/page_images/page_026.png filter=lfs diff=lfs merge=lfs -text
179
+ src/data/prepared/test_openai/page_images/page_027.png filter=lfs diff=lfs merge=lfs -text
180
+ src/data/prepared/test_openai/page_images/page_028.png filter=lfs diff=lfs merge=lfs -text
181
+ src/data/prepared/test_openai/page_images/page_029.png filter=lfs diff=lfs merge=lfs -text
182
+ src/data/prepared/test_openai/page_images/page_030.png filter=lfs diff=lfs merge=lfs -text
183
+ src/data/prepared/test_openai/page_images/page_031.png filter=lfs diff=lfs merge=lfs -text
184
+ src/data/prepared/test_openai/page_images/page_032.png filter=lfs diff=lfs merge=lfs -text
185
+ src/data/prepared/test_openai/page_images/page_033.png filter=lfs diff=lfs merge=lfs -text
186
+ src/data/prepared/test_openai/page_images/page_034.png filter=lfs diff=lfs merge=lfs -text
187
+ src/data/prepared/test_openai/page_images/page_035.png filter=lfs diff=lfs merge=lfs -text
188
+ src/data/prepared/test_openai/page_images/page_036.png filter=lfs diff=lfs merge=lfs -text
189
+ src/data/prepared/test_openai/page_images/page_037.png filter=lfs diff=lfs merge=lfs -text
190
+ src/data/prepared/test_openai/page_images/page_038.png filter=lfs diff=lfs merge=lfs -text
191
+ src/data/prepared/test_openai/page_images/page_039.png filter=lfs diff=lfs merge=lfs -text
192
+ src/data/prepared/test_openai2/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
193
+ src/data/prepared/test_openai2/page_images/page_002.png filter=lfs diff=lfs merge=lfs -text
194
+ src/data/prepared/test_openai2/page_images/page_003.png filter=lfs diff=lfs merge=lfs -text
195
+ src/data/prepared/test_openai2/page_images/page_004.png filter=lfs diff=lfs merge=lfs -text
196
+ src/data/prepared/test_openai2/page_images/page_005.png filter=lfs diff=lfs merge=lfs -text
197
+ src/data/prepared/test_openai2/page_images/page_006.png filter=lfs diff=lfs merge=lfs -text
198
+ src/data/prepared/test_openai2/page_images/page_007.png filter=lfs diff=lfs merge=lfs -text
199
+ src/data/prepared/test_openai2/page_images/page_008.png filter=lfs diff=lfs merge=lfs -text
200
+ src/data/prepared/test_openai2/page_images/page_009.png filter=lfs diff=lfs merge=lfs -text
201
+ src/data/prepared/test_openai2/page_images/page_010.png filter=lfs diff=lfs merge=lfs -text
202
+ src/data/prepared/test_openai2/page_images/page_011.png filter=lfs diff=lfs merge=lfs -text
203
+ src/data/prepared/test_openai2/page_images/page_012.png filter=lfs diff=lfs merge=lfs -text
204
+ src/data/prepared/test_openai2/page_images/page_013.png filter=lfs diff=lfs merge=lfs -text
205
+ src/data/prepared/test_openai2/page_images/page_014.png filter=lfs diff=lfs merge=lfs -text
206
+ src/data/prepared/test_openai2/page_images/page_015.png filter=lfs diff=lfs merge=lfs -text
207
+ src/data/prepared/test_openai2/page_images/page_016.png filter=lfs diff=lfs merge=lfs -text
208
+ src/data/prepared/test_openai2/page_images/page_017.png filter=lfs diff=lfs merge=lfs -text
209
+ src/data/prepared/test_openai2/page_images/page_018.png filter=lfs diff=lfs merge=lfs -text
210
+ src/data/prepared/test_openai2/page_images/page_019.png filter=lfs diff=lfs merge=lfs -text
211
+ src/data/prepared/test_openai2/page_images/page_020.png filter=lfs diff=lfs merge=lfs -text
212
+ src/data/prepared/test_openai2/page_images/page_021.png filter=lfs diff=lfs merge=lfs -text
213
+ src/data/prepared/test_openai2/page_images/page_022.png filter=lfs diff=lfs merge=lfs -text
214
+ src/data/prepared/test_openai2/page_images/page_023.png filter=lfs diff=lfs merge=lfs -text
215
+ src/data/prepared/test_openai2/page_images/page_024.png filter=lfs diff=lfs merge=lfs -text
216
+ src/data/prepared/test_openai2/page_images/page_025.png filter=lfs diff=lfs merge=lfs -text
217
+ src/data/prepared/test_openai2/page_images/page_026.png filter=lfs diff=lfs merge=lfs -text
218
+ src/data/prepared/test_openai2/page_images/page_027.png filter=lfs diff=lfs merge=lfs -text
219
+ src/data/prepared/test_openai2/page_images/page_028.png filter=lfs diff=lfs merge=lfs -text
220
+ src/data/prepared/test_openai2/page_images/page_029.png filter=lfs diff=lfs merge=lfs -text
221
+ src/data/prepared/test_openai2/page_images/page_030.png filter=lfs diff=lfs merge=lfs -text
222
+ src/data/prepared/test_openai2/page_images/page_031.png filter=lfs diff=lfs merge=lfs -text
223
+ src/data/prepared/test_openai2/page_images/page_032.png filter=lfs diff=lfs merge=lfs -text
224
+ src/data/prepared/test_openai2/page_images/page_033.png filter=lfs diff=lfs merge=lfs -text
225
+ src/data/prepared/test_openai2/page_images/page_034.png filter=lfs diff=lfs merge=lfs -text
226
+ src/data/prepared/test_openai2/page_images/page_035.png filter=lfs diff=lfs merge=lfs -text
227
+ src/data/prepared/test_openai2/page_images/page_036.png filter=lfs diff=lfs merge=lfs -text
228
+ src/data/prepared/test_openai2/page_images/page_037.png filter=lfs diff=lfs merge=lfs -text
229
+ src/data/prepared/test_openai2/page_images/page_038.png filter=lfs diff=lfs merge=lfs -text
230
+ src/data/prepared/test_openai2/page_images/page_039.png filter=lfs diff=lfs merge=lfs -text
231
+ src/data/prepared/test_page10/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
232
+ src/data/prepared/test_suffix/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
233
+ src/data/prepared/test_suffix/page_images/page_002.png filter=lfs diff=lfs merge=lfs -text
234
+ src/data/prepared/test_suffix/page_images/page_003.png filter=lfs diff=lfs merge=lfs -text
235
+ src/data/prepared/test_suffix/page_images/page_004.png filter=lfs diff=lfs merge=lfs -text
236
+ src/data/prepared/test_suffix/page_images/page_005.png filter=lfs diff=lfs merge=lfs -text
237
+ src/data/prepared/test_suffix/page_images/page_006.png filter=lfs diff=lfs merge=lfs -text
238
+ src/data/prepared/test_suffix/page_images/page_007.png filter=lfs diff=lfs merge=lfs -text
239
+ src/data/prepared/test_suffix/page_images/page_008.png filter=lfs diff=lfs merge=lfs -text
240
+ src/data/prepared/test_suffix/page_images/page_009.png filter=lfs diff=lfs merge=lfs -text
241
+ src/data/prepared/test_suffix/page_images/page_010.png filter=lfs diff=lfs merge=lfs -text
242
+ src/data/prepared/test_suffix/page_images/page_011.png filter=lfs diff=lfs merge=lfs -text
243
+ src/data/prepared/test_suffix/page_images/page_012.png filter=lfs diff=lfs merge=lfs -text
244
+ src/data/prepared/test_suffix/page_images/page_013.png filter=lfs diff=lfs merge=lfs -text
245
+ src/data/prepared/test_suffix/page_images/page_014.png filter=lfs diff=lfs merge=lfs -text
246
+ src/data/prepared/test_suffix/page_images/page_015.png filter=lfs diff=lfs merge=lfs -text
247
+ src/data/prepared/test_suffix/page_images/page_016.png filter=lfs diff=lfs merge=lfs -text
248
+ src/data/prepared/test_suffix/page_images/page_017.png filter=lfs diff=lfs merge=lfs -text
249
+ src/data/prepared/test_suffix/page_images/page_018.png filter=lfs diff=lfs merge=lfs -text
250
+ src/data/prepared/test_suffix/page_images/page_019.png filter=lfs diff=lfs merge=lfs -text
251
+ src/data/prepared/test_suffix/page_images/page_020.png filter=lfs diff=lfs merge=lfs -text
252
+ src/data/prepared/test_suffix/page_images/page_021.png filter=lfs diff=lfs merge=lfs -text
253
+ src/data/prepared/test_suffix/page_images/page_022.png filter=lfs diff=lfs merge=lfs -text
254
+ src/data/prepared/test_suffix/page_images/page_023.png filter=lfs diff=lfs merge=lfs -text
255
+ src/data/prepared/test_suffix/page_images/page_024.png filter=lfs diff=lfs merge=lfs -text
256
+ src/data/prepared/test_suffix/page_images/page_025.png filter=lfs diff=lfs merge=lfs -text
257
+ src/data/prepared/test_suffix/page_images/page_026.png filter=lfs diff=lfs merge=lfs -text
258
+ src/data/prepared/test_suffix/page_images/page_027.png filter=lfs diff=lfs merge=lfs -text
259
+ src/data/prepared/test_suffix/page_images/page_028.png filter=lfs diff=lfs merge=lfs -text
260
+ src/data/prepared/test_suffix/page_images/page_029.png filter=lfs diff=lfs merge=lfs -text
261
+ src/data/prepared/test_suffix/page_images/page_030.png filter=lfs diff=lfs merge=lfs -text
262
+ src/data/prepared/test_suffix/page_images/page_031.png filter=lfs diff=lfs merge=lfs -text
263
+ src/data/prepared/test_suffix/page_images/page_032.png filter=lfs diff=lfs merge=lfs -text
264
+ src/data/prepared/test_suffix/page_images/page_033.png filter=lfs diff=lfs merge=lfs -text
265
+ src/data/prepared/test_suffix/page_images/page_034.png filter=lfs diff=lfs merge=lfs -text
266
+ src/data/prepared/test_suffix/page_images/page_035.png filter=lfs diff=lfs merge=lfs -text
267
+ src/data/prepared/test_suffix/page_images/page_036.png filter=lfs diff=lfs merge=lfs -text
268
+ src/data/prepared/test_suffix/page_images/page_037.png filter=lfs diff=lfs merge=lfs -text
269
+ src/data/prepared/test_suffix/page_images/page_038.png filter=lfs diff=lfs merge=lfs -text
270
+ src/data/prepared/test_suffix/page_images/page_039.png filter=lfs diff=lfs merge=lfs -text
271
+ src/data/prepared/test1/page_images/page_001.png filter=lfs diff=lfs merge=lfs -text
272
+ src/data/prepared/test1/page_images/page_002.png filter=lfs diff=lfs merge=lfs -text
273
+ src/data/prepared/test1/page_images/page_003.png filter=lfs diff=lfs merge=lfs -text
274
+ src/data/prepared/test1/page_images/page_004.png filter=lfs diff=lfs merge=lfs -text
275
+ src/data/prepared/test1/page_images/page_005.png filter=lfs diff=lfs merge=lfs -text
276
+ src/data/prepared/test1/page_images/page_006.png filter=lfs diff=lfs merge=lfs -text
277
+ src/data/prepared/test1/page_images/page_007.png filter=lfs diff=lfs merge=lfs -text
278
+ src/data/prepared/test1/page_images/page_008.png filter=lfs diff=lfs merge=lfs -text
279
+ src/data/prepared/test1/page_images/page_009.png filter=lfs diff=lfs merge=lfs -text
280
+ src/data/prepared/test1/page_images/page_010.png filter=lfs diff=lfs merge=lfs -text
281
+ src/data/prepared/test1/page_images/page_011.png filter=lfs diff=lfs merge=lfs -text
282
+ src/data/prepared/test1/page_images/page_012.png filter=lfs diff=lfs merge=lfs -text
283
+ src/data/prepared/test1/page_images/page_013.png filter=lfs diff=lfs merge=lfs -text
284
+ src/data/prepared/test1/page_images/page_014.png filter=lfs diff=lfs merge=lfs -text
285
+ src/data/prepared/test1/page_images/page_015.png filter=lfs diff=lfs merge=lfs -text
286
+ src/data/prepared/test1/page_images/page_016.png filter=lfs diff=lfs merge=lfs -text
287
+ src/data/prepared/test1/page_images/page_017.png filter=lfs diff=lfs merge=lfs -text
288
+ src/data/prepared/test1/page_images/page_018.png filter=lfs diff=lfs merge=lfs -text
289
+ src/data/prepared/test1/page_images/page_019.png filter=lfs diff=lfs merge=lfs -text
290
+ src/data/prepared/test1/page_images/page_020.png filter=lfs diff=lfs merge=lfs -text
291
+ src/data/prepared/test1/page_images/page_021.png filter=lfs diff=lfs merge=lfs -text
292
+ src/data/prepared/test1/page_images/page_022.png filter=lfs diff=lfs merge=lfs -text
293
+ src/data/prepared/test1/page_images/page_023.png filter=lfs diff=lfs merge=lfs -text
294
+ src/data/prepared/test1/page_images/page_024.png filter=lfs diff=lfs merge=lfs -text
295
+ src/data/prepared/test1/page_images/page_025.png filter=lfs diff=lfs merge=lfs -text
296
+ src/data/prepared/test1/page_images/page_026.png filter=lfs diff=lfs merge=lfs -text
297
+ src/data/prepared/test1/page_images/page_027.png filter=lfs diff=lfs merge=lfs -text
298
+ src/data/prepared/test1/page_images/page_028.png filter=lfs diff=lfs merge=lfs -text
299
+ src/data/prepared/test1/page_images/page_029.png filter=lfs diff=lfs merge=lfs -text
300
+ src/data/prepared/test1/page_images/page_030.png filter=lfs diff=lfs merge=lfs -text
301
+ src/data/prepared/test1/page_images/page_031.png filter=lfs diff=lfs merge=lfs -text
302
+ src/data/prepared/test1/page_images/page_032.png filter=lfs diff=lfs merge=lfs -text
303
+ src/data/prepared/test1/page_images/page_033.png filter=lfs diff=lfs merge=lfs -text
304
+ src/data/prepared/test1/page_images/page_034.png filter=lfs diff=lfs merge=lfs -text
305
+ src/data/prepared/test1/page_images/page_035.png filter=lfs diff=lfs merge=lfs -text
306
+ src/data/prepared/test1/page_images/page_036.png filter=lfs diff=lfs merge=lfs -text
307
+ src/data/prepared/test1/page_images/page_037.png filter=lfs diff=lfs merge=lfs -text
308
+ src/data/prepared/test1/page_images/page_038.png filter=lfs diff=lfs merge=lfs -text
309
+ src/data/prepared/test1/page_images/page_039.png filter=lfs diff=lfs merge=lfs -text
.gitignore ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # 忽略数据和生成文件
2
+ src/data/
3
+ *.pdf
4
+ *.png
5
+ *.jpg
6
+
7
+ # Python
8
+ __pycache__/
9
+ *.pyc
10
+
11
+ # 临时报告
12
+ report_*.md
README.md ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ title: Yangtez Beta
3
+ emoji: ⚡
4
+ colorFrom: gray
5
+ colorTo: blue
6
+ sdk: gradio
7
+ sdk_version: 6.11.0
8
+ app_file: app.py
9
+ pinned: false
10
+ license: mit
11
+ short_description: fund porposal reviewer
12
+ ---
13
+
14
+ Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
app.py ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ import subprocess
3
+ import os
4
+ import uuid
5
+ import shutil
6
+
7
+ def run_pipeline(file):
8
+
9
+ try:
10
+ # 1️⃣ 生成唯一 ID(核心)
11
+ proposal_id = f"web_{uuid.uuid4().hex[:8]}"
12
+
13
+ # 2️⃣ 创建独立目录
14
+ work_dir = os.path.join("tmp", proposal_id)
15
+ os.makedirs(work_dir, exist_ok=True)
16
+
17
+ # 3️⃣ 保存上传文件(避免覆盖)
18
+ file_path = os.path.join(work_dir, os.path.basename(file.name))
19
+ shutil.copy(file.name, file_path)
20
+
21
+ # 4️⃣ 调用后端(安全方式)
22
+ result = subprocess.run(
23
+ ["python", "run_review.py", "--file", file_path, "--proposal_id", proposal_id],
24
+ check=True,
25
+ timeout=600,
26
+ capture_output=True,
27
+ text=True
28
+ )
29
+
30
+ print("STDOUT:\n", result.stdout)
31
+ print("STDERR:\n", result.stderr)
32
+
33
+ # 5️⃣ 找输出报告
34
+ report_path = os.path.join(
35
+ "src", "data", "runs", proposal_id, f"{proposal_id}_review_report.md"
36
+ )
37
+
38
+ if os.path.exists(report_path):
39
+ with open(report_path, "r", encoding="utf-8") as f:
40
+ return f.read()
41
+ else:
42
+ return "❌ Report not found."
43
+
44
+ except subprocess.TimeoutExpired:
45
+ return "⏱️ Error: Task timeout (too long)."
46
+
47
+ except Exception as e:
48
+ return f"❌ Error: {str(e)}"
49
+
50
+
51
+ # UI
52
+ demo = gr.Interface(
53
+ fn=run_pipeline,
54
+ inputs=gr.File(label="Upload PDF"),
55
+ outputs=gr.Textbox(label="Review Result"),
56
+ title="Yangtze AI Reviewer"
57
+ )
58
+
59
+ if __name__ == "__main__":
60
+ demo.launch()
main ADDED
File without changes
packages.txt ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ poppler-utils
2
+ tesseract-ocr
requirements.txt ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ pdfplumber
2
+ langchain
3
+ langchain-community
4
+ langchain-huggingface
5
+ sentence-transformers
6
+ chromadb
7
+ numpy
8
+ pandas
9
+ tqdm
10
+
11
+ pdf2image
12
+ pdfplumber
13
+ pytesseract
14
+ Pillow
15
+ python-docx
16
+
17
+ python-dotenv
18
+ openai
19
+ requests
20
+ beautifulsoup4
21
+ lxml
22
+ pydantic
23
+
24
+ # optional stronger multimodal/layout/table stack
25
+ transformers
26
+ torch
27
+ accelerate
28
+ # paddleocr
29
+ torchvision
30
+ timm
run_review.py ADDED
@@ -0,0 +1,808 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ from __future__ import annotations
4
+
5
+ """
6
+ End-to-end Yangtze review runner.
7
+
8
+ Pipeline:
9
+ PDF -> prepared JSON -> domain profile -> metric check -> specialized prompts
10
+ -> review JSON -> markdown report
11
+ """
12
+
13
+ import argparse
14
+ import json
15
+ import os
16
+ import re
17
+ from collections import defaultdict
18
+ from dataclasses import dataclass
19
+ from pathlib import Path
20
+ from typing import Any, Dict, List, Sequence, Tuple
21
+
22
+ from dotenv import load_dotenv
23
+ from openai import OpenAI
24
+
25
+ from src.tools.domain_profiler import profile_with_optional_llm
26
+ from src.prompting.domain_adaptive import (
27
+ QUESTION_SEARCH_HINTS,
28
+ REVIEW_TASKS,
29
+ build_specialized_question,
30
+ )
31
+ from src.tools.prepare_proposal_text import prepare_text
32
+ from src.tools.metric_checker import build_metric_report, build_metric_prompt_suffix
33
+
34
+ load_dotenv()
35
+
36
+ BASE_DIR = Path(__file__).resolve().parent
37
+ DATA_DIR = BASE_DIR / "src" / "data"
38
+ RUNS_DIR = DATA_DIR / "runs"
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class EvidenceSnippet:
43
+ page_index: int
44
+ text: str
45
+ score: float
46
+
47
+
48
+ def _clean_text(text: str) -> str:
49
+ return re.sub(r"\s+", " ", text or "").strip()
50
+
51
+
52
+ def _safe_name(name: str) -> str:
53
+ return re.sub(r"[^a-zA-Z0-9_\-]+", "_", name).strip("_") or "proposal"
54
+
55
+
56
+ def load_pages(proposal_id: str) -> List[Dict[str, Any]]:
57
+ path = DATA_DIR / "prepared" / proposal_id / "pages.json"
58
+ return json.loads(path.read_text(encoding="utf-8"))
59
+
60
+
61
+ TASK_KEYWORDS = {
62
+ "problem": ["背景", "需求", "意义", "motivation", "pain point", "problem", "行业背景"],
63
+ "objectives": ["目标", "scope", "deliverable", "计划", "路径", "objective"],
64
+ "methods": ["方法", "技术路线", "样机", "验证", "test", "prototype", "method"],
65
+ "evidence": ["样机", "测试", "专利", "数据", "成果", "validation", "evidence"],
66
+ "feasibility": ["交付", "timeline", "roadmap", "风险", "量产", "feasibility"],
67
+ "innovation": ["唯一", "领先", "创新", "novel", "differentiation", "突破"],
68
+ "risks": ["风险", "瓶颈", "dependency", "cost", "寿命", "竞争", "risk"],
69
+ "team": ["团队", "依托单位", "博士", "研究员", "roles", "team"],
70
+ "outcomes": ["市场", "订单", "收入", "profit", "impact", "evaluation", "融资"],
71
+ }
72
+
73
+
74
+ def score_page_for_task(task_id: str, page_text: str, profile: Dict[str, Any]) -> float:
75
+ text = _clean_text(page_text)
76
+ if not text:
77
+ return 0.0
78
+
79
+ lower = text.lower()
80
+ score = 0.0
81
+
82
+ for term in TASK_KEYWORDS.get(task_id, []):
83
+ if term.lower() in lower:
84
+ score += 2.0
85
+
86
+ dim_map = {
87
+ "problem": profile["evaluation_focus"]["problem"],
88
+ "objectives": profile["evaluation_focus"]["objectives"],
89
+ "methods": profile["methods"],
90
+ "evidence": profile["methods"],
91
+ "feasibility": profile["evaluation_focus"]["feasibility"] + profile["risks"],
92
+ "innovation": profile["evaluation_focus"]["innovation"],
93
+ "risks": profile["risks"],
94
+ "team": profile["evaluation_focus"]["team"],
95
+ "outcomes": profile["evaluation_focus"]["outcomes"],
96
+ }
97
+
98
+ for term in dim_map.get(task_id, []):
99
+ if term.lower() in lower:
100
+ score += 1.2
101
+
102
+ numbers = re.findall(r"\d+(?:\.\d+)?", text)
103
+ if task_id in {"evidence", "feasibility", "outcomes"}:
104
+ score += min(2.0, len(numbers) * 0.15)
105
+
106
+ if task_id == "team" and any(x in text for x in ["博士", "研究员", "教授", "工程师", "expert"]):
107
+ score += 2.0
108
+
109
+ if task_id == "innovation" and any(x in text for x in ["唯一", "领先", "首台", "first", "unique"]):
110
+ score += 1.5
111
+
112
+ return score
113
+
114
+
115
+ def select_evidence(
116
+ task_id: str,
117
+ pages: Sequence[Dict[str, Any]],
118
+ profile: Dict[str, Any],
119
+ top_k: int = 4,
120
+ ) -> List[EvidenceSnippet]:
121
+ scored = []
122
+ for page in pages:
123
+ score = score_page_for_task(task_id, page.get("text", ""), profile)
124
+ if score > 0:
125
+ scored.append(
126
+ EvidenceSnippet(
127
+ page_index=int(page["page_index"]),
128
+ text=_clean_text(page["text"])[:420],
129
+ score=score,
130
+ )
131
+ )
132
+ scored.sort(key=lambda x: (x.score, -x.page_index), reverse=True)
133
+ return scored[:top_k]
134
+
135
+
136
+ def feature_flags(pages: Sequence[Dict[str, Any]], full_text: str) -> Dict[str, bool]:
137
+ return {
138
+ "has_team": bool(re.search(r"团队|leader|博士|教授|研究员", full_text)),
139
+ "has_timeline": bool(re.search(r"20\d{2}|timeline|roadmap|交付|研发路径", full_text)),
140
+ "has_budget": bool(re.search(r"融资|预算|资金|万元|million|cash|profit", full_text)),
141
+ "has_market": bool(re.search(r"市场|customer|订单|demand|销量|sales", full_text)),
142
+ "has_prototype": bool(re.search(r"样机|prototype|测试|验证|patent|专利", full_text)),
143
+ "has_risk_signal": bool(re.search(r"风险|瓶颈|challenge|uncertainty|dependency|��争", full_text)),
144
+ "has_metrics": bool(len(re.findall(r"\d+(?:\.\d+)?", full_text)) >= 10),
145
+ }
146
+
147
+
148
+ def score_task(task_id: str, evidences: Sequence[EvidenceSnippet], flags: Dict[str, bool]) -> Tuple[float, float]:
149
+ base = 5.0
150
+ evidence_strength = min(2.0, sum(min(ev.score, 3.0) for ev in evidences) / 6.0)
151
+ score = base + evidence_strength
152
+ conf = 0.45 + min(0.45, len(evidences) * 0.08)
153
+
154
+ if task_id == "team" and flags["has_team"]:
155
+ score += 1.0
156
+ conf += 0.06
157
+ if task_id == "outcomes" and flags["has_market"] and flags["has_budget"]:
158
+ score += 0.6
159
+ if task_id == "feasibility" and flags["has_timeline"]:
160
+ score += 0.6
161
+ if task_id in {"methods", "evidence"} and flags["has_prototype"]:
162
+ score += 0.8
163
+ if task_id == "risks" and flags["has_risk_signal"]:
164
+ score += 0.6
165
+ if task_id == "innovation" and flags["has_metrics"]:
166
+ score += 0.2
167
+
168
+ if task_id == "outcomes" and not flags["has_market"]:
169
+ score -= 1.1
170
+ if task_id == "feasibility" and not flags["has_timeline"]:
171
+ score -= 1.0
172
+ if task_id == "team" and not flags["has_team"]:
173
+ score -= 1.2
174
+ if task_id in {"methods", "evidence"} and not flags["has_prototype"]:
175
+ score -= 1.2
176
+ if task_id == "risks" and not flags["has_risk_signal"]:
177
+ score -= 0.8
178
+
179
+ score = max(1.0, min(9.5, round(score, 1)))
180
+ conf = max(0.2, min(0.95, round(conf, 2)))
181
+ return score, conf
182
+
183
+
184
+ TASK_TEMPLATES = {
185
+ "problem": (
186
+ "The proposal appears to target a concrete problem area with visible context and motivation.",
187
+ "The motivation would be stronger with clearer articulation of urgency, constraints, and why this is the right solution path now.",
188
+ ),
189
+ "objectives": (
190
+ "The objectives are broadly understandable and tied to an identifiable work direction.",
191
+ "The proposal would benefit from sharper success criteria, scope boundaries, and milestone-level deliverables.",
192
+ ),
193
+ "methods": (
194
+ "The technical or methodological path is visible and internally coherent at a high level.",
195
+ "The document still leaves gaps around assumptions, validation thresholds, and detailed implementation logic.",
196
+ ),
197
+ "evidence": (
198
+ "The proposal includes tangible supporting signals such as prototypes, tests, prior work, or enabling resources.",
199
+ "The evidence base remains incomplete for a decisive judgment because end-to-end validation and independent proof points are limited.",
200
+ ),
201
+ "feasibility": (
202
+ "The execution pathway is plausible and partially supported by roadmap and resource signals.",
203
+ "Feasibility remains exposed to delivery, integration, scaling, and dependency risks that are not fully closed.",
204
+ ),
205
+ "innovation": (
206
+ "The proposal makes a credible differentiation claim relative to standard alternatives.",
207
+ "The novelty case would be stronger with side-by-side benchmarking against the best current alternatives.",
208
+ ),
209
+ "risks": (
210
+ "The proposal contains signals of awareness around technical and operational constraints.",
211
+ "Risk treatment is not yet fully mature: several major failure modes are implied rather than explicitly mitigated.",
212
+ ),
213
+ "team": (
214
+ "The team profile suggests relevant expertise and some execution capability.",
215
+ "The commercial, manufacturing, or cross-functional governance setup is less developed than the technical story.",
216
+ ),
217
+ "outcomes": (
218
+ "The expected outcomes and impact direction are understandable and potentially meaningful.",
219
+ "Commercial or practical outcomes still rely on assumptions that need firmer validation and customer conversion evidence.",
220
+ ),
221
+ }
222
+
223
+
224
+ def build_task_assessment(
225
+ task_id: str,
226
+ task_title: str,
227
+ prompt_text: str,
228
+ evidences: Sequence[EvidenceSnippet],
229
+ score: float,
230
+ confidence: float,
231
+ metric_report: Dict[str, Any],
232
+ ) -> Dict[str, Any]:
233
+ pages = [ev.page_index for ev in evidences]
234
+ snippets = [ev.text[:120] for ev in evidences[:2] if ev.text]
235
+
236
+ if snippets:
237
+ evidence_text = " | ".join(snippets)
238
+ judgment = f"Based on evidence: {evidence_text}"
239
+ else:
240
+ judgment = "Limited direct evidence found in the document."
241
+
242
+ strengths: List[str] = []
243
+ weaknesses: List[str] = []
244
+
245
+ if evidences:
246
+ strengths.append(f"Evidence located on pages {', '.join(map(str, pages[:3]))}.")
247
+ if snippets:
248
+ strengths.append("The proposal includes concrete supporting details rather than only high-level claims.")
249
+
250
+ if score >= 7.5:
251
+ strengths.append("This dimension is relatively strong compared to others.")
252
+
253
+ # ❗减少模板味,增加针对性
254
+ if score <= 6.0:
255
+ weaknesses.append("Evidence is insufficient or not clearly connected to claims.")
256
+
257
+ if task_id == "outcomes":
258
+ weaknesses.append("Outcome claims are not tightly linked to measurable metrics.")
259
+ elif task_id == "feasibility":
260
+ weaknesses.append("Execution feasibility lacks detailed validation or timeline support.")
261
+ elif task_id == "risks":
262
+ weaknesses.append("Risk mitigation strategies are not sufficiently detailed.")
263
+ elif task_id == "team":
264
+ weaknesses.append("Team capability is not strongly evidenced by concrete roles or experience.")
265
+ else:
266
+ weaknesses.append("This aspect could be made more explicit and evidence-supported.")
267
+
268
+ evidence_level = metric_report.get("metric_signals", {}).get("evidence_level", "weak")
269
+
270
+ if evidence_level == "weak":
271
+ weaknesses.append("Quantitative evidence is weak, reducing confidence.")
272
+ elif evidence_level == "moderate":
273
+ weaknesses.append("Quantitative support exists but is not fully convincing.")
274
+
275
+ missing: List[str] = []
276
+
277
+ metric_missing = metric_report.get("metric_signals", {}).get("missing_metric_signals", [])
278
+ missing.extend(metric_missing[:3])
279
+
280
+ missing.extend([
281
+ "Explicit measurable success criteria",
282
+ "Clear assumptions or boundary conditions",
283
+ ])
284
+
285
+ # 去重
286
+ seen = set()
287
+ dedup_missing = []
288
+ for item in missing:
289
+ key = item.lower().strip()
290
+ if key and key not in seen:
291
+ seen.add(key)
292
+ dedup_missing.append(item)
293
+
294
+ return {
295
+ "task_id": task_id,
296
+ "title": task_title,
297
+ "prompt": prompt_text,
298
+ "judgment": judgment,
299
+ "strengths": strengths[:3],
300
+ "weaknesses": weaknesses[:3],
301
+ "missing_information": dedup_missing[:5],
302
+ "confidence": confidence,
303
+ "score_10": score,
304
+ "evidence": [ev.__dict__ for ev in evidences],
305
+ }
306
+
307
+
308
+ def aggregate_dimension_scores(task_results: Sequence[Dict[str, Any]]) -> Dict[str, float]:
309
+ dim_scores: Dict[str, List[float]] = defaultdict(list)
310
+ for task, result in zip(REVIEW_TASKS, task_results):
311
+ dim_scores[task.dimension].append(float(result["score_10"]))
312
+ return {dim: round(sum(vals) / len(vals), 2) for dim, vals in dim_scores.items()}
313
+
314
+
315
+ def compute_final_verdict(task_results: Sequence[Dict[str, Any]]) -> Tuple[float, float, str]:
316
+ scores = [float(r["score_10"]) for r in task_results]
317
+ confs = [float(r["confidence"]) for r in task_results]
318
+ overall = round(sum(scores) / len(scores), 2) if scores else 0.0
319
+ confidence = round(sum(confs) / len(confs), 2) if confs else 0.0
320
+
321
+ if overall >= 7.8 and confidence >= 0.7:
322
+ verdict = "SUPPORT"
323
+ elif overall >= 6.0:
324
+ verdict = "HOLD"
325
+ else:
326
+ verdict = "CONCERN"
327
+
328
+ return overall, confidence, verdict
329
+
330
+
331
+ def generate_final_review(result, proposal_id="Unknown", domain="Unknown"):
332
+ from datetime import datetime
333
+
334
+ def score_to_level(score):
335
+ if score >= 0.8:
336
+ return "较高"
337
+ elif score >= 0.6:
338
+ return "中等偏上"
339
+ elif score >= 0.4:
340
+ return "中等"
341
+ else:
342
+ return "较低"
343
+
344
+ def generate_overall_comment(result):
345
+ score = result.get("overall_score", 0)
346
+
347
+ obj = result.get("objectives", {}).get("summary", "")
348
+ strat = result.get("strategy", {}).get("summary", "")
349
+ innov = result.get("innovation", {}).get("summary", "")
350
+ feas = result.get("feasibility", {}).get("summary", "")
351
+
352
+ # 去掉太原始/证据型文本
353
+ def clean(x):
354
+ if not x:
355
+ return ""
356
+ x = x.replace("Based on evidence:", "")
357
+ return x.strip()
358
+
359
+ obj = clean(obj)
360
+ strat = clean(strat)
361
+ innov = clean(innov)
362
+ feas = clean(feas)
363
+
364
+ # 分数 -> 总体判断
365
+ if score >= 0.8:
366
+ level = "整体表现较好"
367
+ elif score >= 0.6:
368
+ level = "整体表现中等偏上"
369
+ elif score >= 0.4:
370
+ level = "整体表现一般"
371
+ else:
372
+ level = "整体表现较弱"
373
+
374
+ # 动态拼接(关键)
375
+ parts = []
376
+
377
+ if obj:
378
+ parts.append(f"在研究目标方面,{obj}")
379
+ if strat:
380
+ parts.append(f"在技术路线方面,{strat}")
381
+ if innov:
382
+ parts.append(f"在创新性方面,{innov}")
383
+ if feas:
384
+ parts.append(f"在可行性方面,{feas}")
385
+
386
+ parts.append(f"总体来看,项目{level},但仍存在进一步优化空间。")
387
+
388
+ return "。".join(parts) + "。"
389
+
390
+ def summarize_dimension(text, dim_name):
391
+ if not text:
392
+ return "该部分描述较为有限"
393
+
394
+ if dim_name == "team":
395
+ return "团队具备一定研究基础,但人员结构与分工说明仍不够充分"
396
+ elif dim_name == "objectives":
397
+ return "研究目标较为明确,但可量化程度有待提升"
398
+ elif dim_name == "strategy":
399
+ return "技术路线整体较为清晰,但部分关键环节需进一步细化"
400
+ elif dim_name == "innovation":
401
+ return "项目具有一���创新性,但创新点仍需进一步突出"
402
+ elif dim_name == "feasibility":
403
+ return "项目具备一定实施基础,但可行性论证仍需加强"
404
+ else:
405
+ return "该部分内容较为完整,但仍有提升空间"
406
+
407
+ def generate_strengths():
408
+ return [
409
+ "项目整体结构较为完整,涵盖研究目标、技术路线与实施方案",
410
+ "技术路线具有一定逻辑性,具备基础可行性",
411
+ "在相关领域具有一定探索性,体现出一定创新潜力"
412
+ ]
413
+
414
+ def generate_weaknesses():
415
+ return [
416
+ "研究目标缺乏明确的量化指标,难以进行效果评估",
417
+ "技术路线部分关键环节描述不够具体,存在一定不确定性",
418
+ "创新点与现有研究的差异性阐述不够充分",
419
+ "可行性分析偏弱,缺乏实验或数据支撑",
420
+ "风险识别与应对机制尚不完善"
421
+ ]
422
+
423
+ def generate_suggestions():
424
+ return [
425
+ "建议细化研究目标,增加可量化评价指标",
426
+ "建议完善技术路线描述,补充关键步骤说明",
427
+ "建议加强与现有研究的对比分析,突出创新点",
428
+ "建议补充实验验证或数据支撑,提高方案可信度",
429
+ "建议完善风险评估与应对策略"
430
+ ]
431
+
432
+ overall_score = result.get("overall_score", 0)
433
+ confidence = result.get("confidence", 0)
434
+ verdict = result.get("verdict", "HOLD")
435
+
436
+ team = result.get("team", {})
437
+ obj = result.get("objectives", {})
438
+ strat = result.get("strategy", {})
439
+ innov = result.get("innovation", {})
440
+ feas = result.get("feasibility", {})
441
+
442
+ strengths = generate_strengths()
443
+ weaknesses = generate_weaknesses()
444
+ suggestions = generate_suggestions()
445
+
446
+ now = datetime.now().strftime("%Y-%m-%d")
447
+
448
+ report = f"""
449
+ # 科研项目评审报告
450
+
451
+ ## 一、项目基本信息
452
+ 项目名称:{proposal_id}
453
+ 申报领域:{domain}
454
+ 评审时间:{now}
455
+
456
+ ## 二、总体评价
457
+ {generate_overall_comment(result)}
458
+
459
+ ## 三、总体评分
460
+ 综合评分:{round(overall_score * 10, 2)} / 10
461
+ 评审置信度:{round(confidence, 3)}
462
+ 评审结论:{verdict}
463
+
464
+ ## 四、分项评分
465
+
466
+ | 评审维度 | 得分 | 评价 |
467
+ |----------|------|------|
468
+ | 研究团队 | {team.get("score", 0):.3f} | {summarize_dimension("", "team")} |
469
+ | 研究目标 | {obj.get("score", 0):.3f} | {summarize_dimension("", "objectives")} |
470
+ | 技术路线 | {strat.get("score", 0):.3f} | {summarize_dimension("", "strategy")} |
471
+ | 创新性 | {innov.get("score", 0):.3f} | {summarize_dimension("", "innovation")} |
472
+ | 可行性 | {feas.get("score", 0):.3f} | {summarize_dimension("", "feasibility")} |
473
+
474
+ ## 五、主要优点
475
+ {chr(10).join([f"{i+1}. {s}" for i, s in enumerate(strengths)]) if strengths else "暂无明显优点总结"}
476
+
477
+ ## 六、主要问题
478
+ {chr(10).join([f"{i+1}. {w}" for i, w in enumerate(weaknesses)]) if weaknesses else "暂无明显问题"}
479
+
480
+ ## 七、修改建议
481
+ {chr(10).join([f"{i+1}. {s}" for i, s in enumerate(suggestions)]) if suggestions else "暂无具体建议"}
482
+
483
+ ## 八、最终结论
484
+ 综合以上分析,本项目整体达到{score_to_level(overall_score)}水平,建议:{verdict}。
485
+ """
486
+
487
+ return report
488
+
489
+
490
+ def llm_generate_review(result, proposal_id, domain):
491
+ client = OpenAI()
492
+
493
+ prompt = f"""
494
+ 你是一位国家自然科学基金评审专家,请基于以下结构化评审结果,撰写一份正式、规范、严谨的科研项目评审报告。
495
+
496
+ 【严格要求】
497
+ 1. 必须使用中文
498
+ 2. 必须使用正式评审报告语气(严谨、客观、有判断)
499
+ 3. 不允许出现英文
500
+ 4. 不允许出现“根据数据/系统/模型”等字样
501
+ 5. 不允许输出JSON或解释过程
502
+ 6. 必须使用标准科研评审结构(如下)
503
+ 7. 最终结论必须与“评审结论”字段严格一致,不得出现前后矛盾。
504
+ 8. 当评审结论为 HOLD 时,最终结论应表述为“建议暂缓资助”或“建议补充完善后再评估”,不得写为“建议资助”。
505
+
506
+ 【报告结构(必须严格遵守)】
507
+
508
+ # 科研项目评审报告
509
+
510
+ ## 一、项目基本信息
511
+ 项目名称:{proposal_id}
512
+ 申报领域:{domain}
513
+
514
+ ## 二、总体评价
515
+ (必须为一段完整文字,总结目标、方法、创新、可行性,并给出整体判断)
516
+
517
+ ## 三、总体评分
518
+ - 综合评分:X / 10(注意:必须转为10分制)
519
+ - 评审置信度:X.XX
520
+ - 评审结论:{result.get("verdict")}
521
+
522
+ ## 四、分项评分
523
+ (必须为表格形式)
524
+
525
+ | 评审维度 | 得分(0-1) | 评价 |
526
+ |----------|------------|------|
527
+ | 研究团队 | {result.get("team", {}).get("score", 0)} | (一句评价) |
528
+ | 研究目标 | {result.get("objectives", {}).get("score", 0)} | (一句评价) |
529
+ | 技术路线 | {result.get("strategy", {}).get("score", 0)} | (一句评价) |
530
+ | 创新性 | {result.get("innovation", {}).get("score", 0)} | (一句评价) |
531
+ | 可行性 | {result.get("feasibility", {}).get("score", 0)} | (一句评价) |
532
+
533
+ ## 五、主要优点
534
+ (3–5条,必须是专家评价语气)
535
+
536
+ ## 六、主要问题
537
+ (3–5条,必须具体、有判断)
538
+
539
+ ## 七、修改建议
540
+ (必须与问题一一对应)
541
+
542
+ ## 八、最终结论
543
+ (必须是正式评审结论语言,如:建议资助 / 有条件资助 / 暂缓资助 / 不建议资助)
544
+
545
+ 【输入数据】
546
+ 综合评分(0-1):{result.get("overall_score")}
547
+ 置信度:{result.get("confidence")}
548
+
549
+ Team: {result.get("team")}
550
+ Objectives: {result.get("objectives")}
551
+ Strategy: {result.get("strategy")}
552
+ Innovation: {result.get("innovation")}
553
+ Feasibility: {result.get("feasibility")}
554
+ """
555
+
556
+ response = client.chat.completions.create(
557
+ model="gpt-4o-mini",
558
+ messages=[{"role": "user", "content": prompt}],
559
+ temperature=0.3
560
+ )
561
+
562
+ return response.choices[0].message.content
563
+
564
+
565
+ def generate_markdown_report(
566
+ proposal_id: str,
567
+ file_path: Path,
568
+ profile: Dict[str, Any],
569
+ task_results: Sequence[Dict[str, Any]],
570
+ dim_scores: Dict[str, float],
571
+ overall: float,
572
+ confidence: float,
573
+ verdict: str,
574
+ profiler_mode: str,
575
+ ) -> str:
576
+ lines: List[str] = []
577
+ lines.append(f"# Yangtze Review Report — {proposal_id}")
578
+ lines.append("")
579
+ lines.append(f"- Source file: `{file_path.name}`")
580
+ lines.append(f"- Profiler mode: `{profiler_mode}`")
581
+ lines.append(f"- Overall score: **{overall:.2f} / 10**")
582
+ lines.append(f"- Confidence: **{confidence:.2f}**")
583
+ lines.append(f"- Verdict: **{verdict}**")
584
+ lines.append("")
585
+ lines.append("## Domain Profile")
586
+ lines.append("")
587
+ lines.append(f"- Primary domain: **{profile['domain']['primary']}**")
588
+ if profile["domain"]["secondary"]:
589
+ lines.append(f"- Secondary tags: {', '.join(profile['domain']['secondary'])}")
590
+ lines.append(f"- Methods: {', '.join(profile['methods']) or 'N/A'}")
591
+ lines.append(f"- Risks: {', '.join(profile['risks']) or 'N/A'}")
592
+ lines.append(f"- Terminology: {', '.join(profile['terminology']) or 'N/A'}")
593
+ lines.append("")
594
+ lines.append("## Dimension Scores")
595
+ lines.append("")
596
+ for dim in ["team", "objectives", "strategy", "innovation", "feasibility"]:
597
+ value = dim_scores.get(dim)
598
+ if value is not None:
599
+ lines.append(f"- **{dim}**: {value:.2f} / 10")
600
+ lines.append("")
601
+
602
+ for result in task_results:
603
+ lines.append(f"## {result['title']}")
604
+ lines.append("")
605
+ lines.append(f"**Score:** {result['score_10']:.1f} / 10 ")
606
+ lines.append(f"**Confidence:** {result['confidence']:.2f}")
607
+ lines.append("")
608
+ lines.append(f"**Prompt used**: {result['prompt']}")
609
+ lines.append("")
610
+ lines.append(result["judgment"])
611
+ lines.append("")
612
+ lines.append("**Strengths**")
613
+ for item in result["strengths"]:
614
+ lines.append(f"- {item}")
615
+ lines.append("")
616
+ lines.append("**Weaknesses**")
617
+ for item in result["weaknesses"]:
618
+ lines.append(f"- {item}")
619
+ lines.append("")
620
+ lines.append("**Missing information**")
621
+ for item in result["missing_information"]:
622
+ lines.append(f"- {item}")
623
+ lines.append("")
624
+ lines.append("**Evidence excerpts**")
625
+ if result["evidence"]:
626
+ for ev in result["evidence"]:
627
+ snippet = _clean_text(ev["text"])
628
+ lines.append(f"- Page {ev['page_index']}: {snippet}")
629
+ else:
630
+ lines.append("- No strong evidence matched this task.")
631
+ lines.append("")
632
+
633
+ strongest = sorted(task_results, key=lambda r: r["score_10"], reverse=True)[:3]
634
+ weakest = sorted(task_results, key=lambda r: r["score_10"])[:3]
635
+
636
+ lines.append("## Executive Summary")
637
+ lines.append("")
638
+ lines.append("### Stronger areas")
639
+ for item in strongest:
640
+ lines.append(f"- {item['title']} ({item['score_10']:.1f}/10)")
641
+ lines.append("")
642
+ lines.append("### Weaker areas")
643
+ for item in weakest:
644
+ lines.append(f"- {item['title']} ({item['score_10']:.1f}/10)")
645
+ lines.append("")
646
+
647
+ if verdict == "SUPPORT":
648
+ rec = "The proposal is supportable as presented, though normal diligence is still recommended."
649
+ elif verdict == "HOLD":
650
+ rec = "The proposal has credible potential but should move forward only with targeted clarification, validation, and execution-risk reduction."
651
+ else:
652
+ rec = "The proposal requires substantial strengthening before it would be ready for a positive funding or approval decision."
653
+
654
+ lines.append(f"**Recommendation:** {rec}")
655
+ lines.append("")
656
+ return "\n".join(lines)
657
+
658
+
659
+ def run_review(file_path: Path, proposal_id: str | None = None, use_ocr: bool = True) -> Dict[str, Any]:
660
+ file_path = file_path.resolve()
661
+ proposal_id = proposal_id or _safe_name(file_path.stem)
662
+ RUNS_DIR.mkdir(parents=True, exist_ok=True)
663
+
664
+ prep = prepare_text(file_path=file_path, proposal_id=proposal_id, use_ocr=use_ocr)
665
+ print("=== DEBUG PAGE 3 TEXT ===")
666
+ print(prep.get("reconstructed_full_text", "")[:2000])
667
+ pages = load_pages(proposal_id)
668
+ full_text_path = Path(prep["full_text_path"])
669
+ full_text = full_text_path.read_text(encoding="utf-8", errors="ignore")
670
+
671
+ run_dir = RUNS_DIR / proposal_id
672
+ run_dir.mkdir(parents=True, exist_ok=True)
673
+
674
+ profile, profiler_mode = profile_with_optional_llm(full_text, pages)
675
+ (run_dir / "domain_profile.json").write_text(
676
+ json.dumps(profile, ensure_ascii=False, indent=2),
677
+ encoding="utf-8",
678
+ )
679
+
680
+ metric_report = build_metric_report(full_text, pages)
681
+ (run_dir / "metric_report.json").write_text(
682
+ json.dumps(metric_report, ensure_ascii=False, indent=2),
683
+ encoding="utf-8",
684
+ )
685
+
686
+ metric_prompt_suffix = build_metric_prompt_suffix(metric_report)
687
+
688
+ prompt_items = []
689
+ task_results = []
690
+ flags = feature_flags(pages, full_text)
691
+
692
+ for task in REVIEW_TASKS:
693
+ base_prompt_text = build_specialized_question(task, profile)
694
+ prompt_text = f"{base_prompt_text} {metric_prompt_suffix}"
695
+ evidences = select_evidence(task.task_id, pages, profile, top_k=4)
696
+ score, conf = score_task(task.task_id, evidences, flags)
697
+
698
+ result = build_task_assessment(
699
+ task.task_id,
700
+ task.title,
701
+ prompt_text,
702
+ evidences,
703
+ score,
704
+ conf,
705
+ metric_report,
706
+ )
707
+
708
+ prompt_items.append({
709
+ "task_id": task.task_id,
710
+ "template_id": task.template_id,
711
+ "dimension": task.dimension,
712
+ "title": task.title,
713
+ "prompt": prompt_text,
714
+ "search_hints": QUESTION_SEARCH_HINTS.get(task.dimension, []),
715
+ })
716
+ task_results.append(result)
717
+
718
+ (run_dir / "prompts.json").write_text(
719
+ json.dumps(prompt_items, ensure_ascii=False, indent=2),
720
+ encoding="utf-8",
721
+ )
722
+ (run_dir / "task_results.json").write_text(
723
+ json.dumps(task_results, ensure_ascii=False, indent=2),
724
+ encoding="utf-8",
725
+ )
726
+
727
+ dim_scores = aggregate_dimension_scores(task_results)
728
+ overall, confidence, verdict = compute_final_verdict(task_results)
729
+
730
+ review_json = {
731
+ "proposal_id": proposal_id,
732
+ "source_file": str(file_path),
733
+ "profiler_mode": profiler_mode,
734
+ "overall_score_10": overall,
735
+ "confidence": confidence,
736
+ "verdict": verdict,
737
+ "dimension_scores": dim_scores,
738
+ "domain_profile": profile,
739
+ "metric_report": metric_report,
740
+ "tasks": task_results,
741
+ }
742
+ (run_dir / "review.json").write_text(
743
+ json.dumps(review_json, ensure_ascii=False, indent=2),
744
+ encoding="utf-8",
745
+ )
746
+
747
+ def _dim_bucket(dim: str) -> Dict[str, Any]:
748
+ dim_tasks = [t for t in task_results if any(rt.dimension == dim and rt.task_id == t["task_id"] for rt in REVIEW_TASKS)]
749
+ score_10 = dim_scores.get(dim, 0.0)
750
+ summary = ""
751
+ if dim_tasks:
752
+ top = sorted(dim_tasks, key=lambda x: x.get("score_10", 0), reverse=True)[0]
753
+ summary = top.get("judgment", "")[:120]
754
+ return {
755
+ "score": score_10 / 10.0,
756
+ "summary": summary,
757
+ "strengths": [s for t in dim_tasks for s in t.get("strengths", [])],
758
+ "weaknesses": [w for t in dim_tasks for w in t.get("weaknesses", [])],
759
+ }
760
+
761
+ result = {
762
+ "overall_score": overall / 10.0,
763
+ "confidence": confidence,
764
+ "verdict": verdict,
765
+ "team": _dim_bucket("team"),
766
+ "objectives": _dim_bucket("objectives"),
767
+ "strategy": _dim_bucket("strategy"),
768
+ "innovation": _dim_bucket("innovation"),
769
+ "feasibility": _dim_bucket("feasibility"),
770
+ }
771
+ domain = profile.get("domain", {}).get("primary", "Unknown")
772
+ final_report = llm_generate_review(result, proposal_id, domain)
773
+ print(final_report)
774
+
775
+ report_path = run_dir / f"{proposal_id}_review_report.md"
776
+ report_path.write_text(final_report, encoding="utf-8")
777
+
778
+ with open(f"report_{proposal_id}.md", "w", encoding="utf-8") as f:
779
+ f.write(final_report)
780
+
781
+ return {
782
+ "proposal_id": proposal_id,
783
+ "run_dir": str(run_dir),
784
+ "report_path": str(report_path),
785
+ "review_json_path": str(run_dir / "review.json"),
786
+ "overall_score_10": overall,
787
+ "confidence": confidence,
788
+ "verdict": verdict,
789
+ "profiler_mode": profiler_mode,
790
+ }
791
+
792
+
793
+ def main() -> None:
794
+ ap = argparse.ArgumentParser(description="Run the Yangtze domain-adaptive review pipeline end to end.")
795
+ ap.add_argument("--file", required=True, help="Path to PDF/DOCX/TXT/MD proposal file")
796
+ ap.add_argument("--proposal_id", default="", help="Optional proposal ID for output directories")
797
+ ap.add_argument("--no_ocr", action="store_true", help="Disable OCR fallback during PDF text preparation")
798
+ args = ap.parse_args()
799
+
800
+ result = run_review(
801
+ Path(args.file),
802
+ proposal_id=args.proposal_id.strip() or None,
803
+ use_ocr=not args.no_ocr,
804
+ )
805
+
806
+
807
+ if __name__ == "__main__":
808
+ main()
src/__init__.py ADDED
File without changes
src/api/models.py ADDED
File without changes
src/api/server.py ADDED
File without changes
src/backend/__init__.py ADDED
File without changes
src/backend/retrievers/__init__.py ADDED
File without changes
src/backend/retrievers/web_search.py ADDED
@@ -0,0 +1,662 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ 模块:Web Search(v2025.12 ProClean · TrustLayer + MustHave + Diagnostics + ReRank v1)
4
+ 保持签名:simple_search(query, max_results=8, dimension="general", hints=None, source="LLM")
5
+ 输出不变:src/data/evidence/{proposal_id}/{dimension}_combined.json
6
+
7
+ 新增要点(对相关性友好,且默认安全):
8
+ - Query 智能拼接:若用户 query 已含 site:/时间窗/负向词,则不重复注入;team 维度更少误杀
9
+ - 轻量 ReRank(BM25-lite + 关键短语/实体加权 + 信息密度 + 来源置信度)
10
+ - 文本近重复抑制(n-gram Jaccard,默认阈值 0.92)
11
+ - 结果有序写入 combined(更相关的排前面,但仍保存全部)
12
+ - Cache 命中不再覆盖 combined(之前已修)
13
+ - 多线程 host 配额加锁(之前已修)
14
+ """
15
+ import os, re, time, json, hashlib, warnings, math
16
+ import requests, trafilatura
17
+ from bs4 import BeautifulSoup, XMLParsedAsHTMLWarning
18
+ from urllib.parse import urlparse, urlunparse, parse_qsl, urlencode
19
+ from concurrent.futures import ThreadPoolExecutor, as_completed
20
+ from collections import Counter, defaultdict
21
+ from pathlib import Path
22
+ import threading
23
+
24
+ warnings.filterwarnings("ignore", category=XMLParsedAsHTMLWarning)
25
+
26
+ # ===== 外部依赖(可选)=====
27
+ try:
28
+ from tavily import TavilyClient
29
+ except ImportError:
30
+ TavilyClient = None
31
+ try:
32
+ from duckduckgo_search import DDGS
33
+ except ImportError:
34
+ DDGS = None
35
+
36
+ # ===== LLM (仅用于小摘要,非必须)=====
37
+ try:
38
+ from backend.utils.model_selector import get_llm_client
39
+ _llm_info = get_llm_client()
40
+ LLM_CLIENT = _llm_info["client"]
41
+ LLM_MODEL = _llm_info["model_name"]
42
+ LLM_PROVIDER = _llm_info["provider"]
43
+ print(f"💬 WebSearch 使用 {LLM_PROVIDER.upper()} 模型:{LLM_MODEL}")
44
+ except Exception:
45
+ LLM_CLIENT, LLM_MODEL = None, None
46
+
47
+ # ===== 环境变量 =====
48
+ TAVILY_KEY = os.getenv("TAVILY_API_KEY")
49
+ GOOGLE_KEY = os.getenv("GOOGLE_API_KEY")
50
+ GOOGLE_CX = os.getenv("GOOGLE_CX")
51
+ client_tavily = TavilyClient(api_key=TAVILY_KEY) if (TavilyClient and TAVILY_KEY) else None
52
+
53
+ # 可调控开关(全部向后兼容)
54
+ MUST_HAVE_STRICT = os.getenv("WEB_MUST_HAVE_STRICT", "1") != "0"
55
+ RERANK_ENABLE = os.getenv("WEB_RERANK_ENABLE", "1") != "0" # 新增:开启轻量重排
56
+ DEDUP_ENABLE = os.getenv("WEB_DEDUP_ENABLE", "1") != "0" # 新增:开启文本近重复抑制
57
+ DEDUP_JACCARD = float(os.getenv("WEB_DEDUP_JACCARD", "0.92")) # 新增:n-gram Jaccard 阈值
58
+
59
+ # ===== 全局参数 =====
60
+ MIN_LEN_BASE = 140
61
+ MAX_LEN = 2600
62
+ MAX_WORKERS = 8
63
+ RETRY = 2
64
+ TIMEOUT = 12
65
+ HOST_QUOTA = 3
66
+ EARLY_ACADEMIC_MIN = {"strategy": 4, "objectives": 4, "feasibility": 4, "innovation": 3, "team": 3}
67
+
68
+ LANG_ALLOW = {"en", "zh", "zh-cn", "zh-tw", "und"}
69
+ BLOCK_HOSTS = (
70
+ "facebook.", "twitter.", "x.com", "reddit.", "zhihu.", "bilibili.",
71
+ "medium.com", "pinterest.", "wechat.", "weibo.", "quora.", "csdn.",
72
+ "scribd.com", "moomoo.com", "sol-war.ru", "xmind.com", "islandenergy.je",
73
+ "glassdoor.", "indeed.", "join.com", "job", "careers", "recruit", "press" # (已移除 "news")
74
+ )
75
+
76
+ ACADEMIC_DOMAINS = [
77
+ "arxiv.org","openreview.net","aclweb.org","ieeexplore.ieee.org","dl.acm.org",
78
+ "springer.com","nature.com","sciencedirect.com","wiley.com","tandfonline.com",
79
+ "mdpi.com","frontiersin.org","osf.io","zenodo.org","doi.org"
80
+ ]
81
+ INSTITUTIONAL_HINTS = (".edu", ".ac.", "university", "hospital")
82
+
83
+ MUST_HAVE_BY_DIM = {
84
+ "innovation": ["novel", "innovation", "benchmark", "state of the art", "comparison"],
85
+ "strategy": ["method", "workflow", "implementation", "architecture", "validation"],
86
+ "objectives": ["objective", "scope", "deliverable", "evaluation", "metric"],
87
+ "feasibility":["resource", "timeline", "risk", "dependency", "budget"],
88
+ "team": ["team", "expertise", "affiliation", "leadership", "governance"]
89
+ }
90
+ MIN_LEN_BY_DIM = {"team": 80}
91
+
92
+ NEGATIVE_BY_DIM = {
93
+ "strategy": "-stock -finance -press -news -forum -blog -marketing",
94
+ "feasibility":"-press -news -forum -blog -marketing -stock -finance",
95
+ "objectives": "-news -press -blog -forum",
96
+ "innovation": "-press -news -blog -forum",
97
+ "team": "-jobs -career -recruit -hiring -admissions" # 放宽 team(不再 -press/-news)
98
+ }
99
+
100
+ _HTTP = requests.Session()
101
+ _HTTP.headers.update({"User-Agent": "Mozilla/5.0 (compatible; RAG6View/2025; +https://rag6view.local/agent)"})
102
+
103
+
104
+ # ===== 基础打分 =====
105
+ def _clamp(x, lo=0.3, hi=1.0):
106
+ try: return max(lo, min(hi, float(x)))
107
+ except Exception: return lo
108
+
109
+ def source_confidence(domain: str) -> float:
110
+ if not domain: return 0.55
111
+ d = domain.lower()
112
+ high = ["pubmed","pmc","nih","who","nature","fda","ema","sciencedirect","springer","clinicaltrials","nejm","thelancet","bmj","cell"]
113
+ medium = ["arxiv","biorxiv","medrxiv","researchsquare"]
114
+ low = ["news","press","blog","medium.com"]
115
+ if any(x in d for x in high): return 0.98
116
+ if any(x in d for x in medium): return 0.80
117
+ if any(x in d for x in low): return 0.50
118
+ return 0.62
119
+
120
+ def info_density_score(text: str) -> float:
121
+ n_year = len(re.findall(r"\b(19|20)\d{2}\b", text))
122
+ n_unit = len(re.findall(r"\b(AI|ML|benchmark|dataset|trial|phase\s?(I|II|III)|accuracy|F1|AUC|API|prototype|workflow|framework)\b", text, re.I))
123
+ return _clamp((n_year + n_unit) / 24.0, 0.0, 1.0)
124
+
125
+
126
+ # ===== URL/域名工具 =====
127
+ UTM_KEYS = {"utm_source","utm_medium","utm_campaign","utm_term","utm_content","utm_id","gclid","fbclid","msclkid"}
128
+ def normalize_url(u: str) -> str:
129
+ try:
130
+ p = urlparse(u)
131
+ scheme = p.scheme or "https"
132
+ q = [(k,v) for k,v in parse_qsl(p.query, keep_blank_values=True) if k.lower() not in UTM_KEYS]
133
+ new_q = urlencode(q, doseq=True)
134
+ path = p.path or "/"
135
+ if len(path) > 1 and path.endswith("/"): path = path[:-1]
136
+ return urlunparse((scheme, p.netloc.lower(), path, "", new_q, ""))
137
+ except Exception:
138
+ return u
139
+
140
+ def is_whitelisted(host: str) -> bool:
141
+ h = (host or "").lower()
142
+ return any(h.endswith(d) for d in ACADEMIC_DOMAINS) or any(x in h for x in INSTITUTIONAL_HINTS)
143
+
144
+ def _is_homepage(url: str) -> bool:
145
+ return bool(re.match(r"^https?://[^/]+/?$", url or ""))
146
+
147
+ def rough_lang(text: str) -> str:
148
+ if not text: return "en"
149
+ chinese = len([ch for ch in text[:1200] if "\u4e00" <= ch <= "\u9fff"])
150
+ ratio = chinese / max(1, len(text[:1200]))
151
+ return "zh" if ratio > 0.02 else "en"
152
+
153
+
154
+ # ===== 标题 & HEAD =====
155
+ def fetch_title(html: str) -> str:
156
+ try:
157
+ soup = BeautifulSoup(html, "html.parser")
158
+ t = soup.title.get_text(" ", strip=True) if soup.title else ""
159
+ return re.sub(r"\s+", " ", t)[:200]
160
+ except Exception:
161
+ return ""
162
+
163
+ def head_content_type(url: str) -> str:
164
+ try:
165
+ r = _HTTP.head(url, timeout=8, allow_redirects=True)
166
+ return r.headers.get("Content-Type", "").lower()
167
+ except Exception:
168
+ return ""
169
+
170
+
171
+ # ===== 正文抽取 =====
172
+ _BAD_HINTS = ["cookie", "privacy", "subscribe", "登录", "订阅", "广告", "forbidden", "please enable javascript"]
173
+
174
+ def _extract_pdf_with_trafilatura(url: str) -> str:
175
+ try:
176
+ html2 = trafilatura.fetch_url(url)
177
+ tx = trafilatura.extract(html2) if html2 else ""
178
+ return (tx or "").strip()
179
+ except Exception:
180
+ return ""
181
+
182
+ def fetch_clean_text(url: str, dimension: str, title_hint: str = "", host: str = ""):
183
+ # team 放宽高校/医院 PDF
184
+ ctype = head_content_type(url)
185
+ is_pdf = "pdf" in ctype or url.lower().endswith(".pdf")
186
+ allow_pdf = False
187
+ if is_pdf:
188
+ if dimension in ("strategy","objectives","feasibility") and is_whitelisted(host):
189
+ allow_pdf = True
190
+ if dimension == "team" and (".edu" in host or "university" in host or "hospital" in host):
191
+ allow_pdf = True
192
+ if not allow_pdf:
193
+ return None
194
+
195
+ if is_pdf:
196
+ text_pdf = _extract_pdf_with_trafilatura(url)
197
+ if not text_pdf:
198
+ return None
199
+ text_pdf = re.sub(r"\s+", " ", text_pdf)
200
+ return text_pdf[:MAX_LEN]
201
+
202
+ html = None
203
+ for r in range(RETRY + 1):
204
+ try:
205
+ resp = _HTTP.get(url, timeout=TIMEOUT)
206
+ if resp.status_code == 200:
207
+ html = resp.text
208
+ break
209
+ except Exception:
210
+ time.sleep(0.35 * (2 ** r))
211
+ if not html:
212
+ return None
213
+
214
+ text = ""
215
+ try:
216
+ soup = BeautifulSoup(html, "html.parser")
217
+ main = soup.find(["article","main"]) or soup
218
+ text = " ".join(p.get_text(" ", strip=True) for p in main.find_all(["p","li"]))
219
+ except Exception:
220
+ text = ""
221
+
222
+ if len(text) < 180:
223
+ try:
224
+ html2 = trafilatura.fetch_url(url)
225
+ text2 = trafilatura.extract(html2) if html2 else ""
226
+ if text2 and len(text2) > len(text):
227
+ text = text2
228
+ except Exception:
229
+ pass
230
+
231
+ if not text:
232
+ return None
233
+
234
+ text = re.sub(r"\s+", " ", text).strip()
235
+ low = text.lower()
236
+ if any(b in low for b in _BAD_HINTS):
237
+ return None
238
+
239
+ min_len = MIN_LEN_BY_DIM.get(dimension, MIN_LEN_BASE)
240
+ if dimension == "team" and any(k in (title_hint or "").lower() for k in ["team","lab","group","department","faculty","profile"]):
241
+ min_len = min(min_len, 60)
242
+
243
+ lang = rough_lang(text)
244
+ if lang not in LANG_ALLOW: return None
245
+ if len(text) < min_len: return None
246
+ return text[:MAX_LEN]
247
+
248
+
249
+ # ===== 基础搜索器 =====
250
+ def google_search(q, n=8):
251
+ if not GOOGLE_KEY or not GOOGLE_CX: return []
252
+ try:
253
+ r = _HTTP.get("https://www.googleapis.com/customsearch/v1",
254
+ params={"key": GOOGLE_KEY, "cx": GOOGLE_CX, "q": q, "num": n}, timeout=TIMEOUT)
255
+ if r.status_code == 200:
256
+ items = r.json().get("items", []) or []
257
+ return [i.get("link") for i in items if i.get("link")]
258
+ except Exception as e:
259
+ print(f"⚠️ Google 搜索失败: {e}")
260
+ return []
261
+
262
+ def tavily_search(q, n=8):
263
+ if not client_tavily: return []
264
+ try:
265
+ r = client_tavily.search(query=q, max_results=n)
266
+ return [x.get("url") for x in (r.get("results") or []) if x.get("url")]
267
+ except Exception as e:
268
+ if "limit" in str(e).lower():
269
+ print("⚠️ Tavily 限额已达,降级 DuckDuckGo")
270
+ return duckduckgo_search_fn(q, n)
271
+ print(f"⚠️ Tavily 搜索失败: {e}")
272
+ return []
273
+
274
+ def duckduckgo_search_fn(q, n=8):
275
+ if not DDGS: return []
276
+ try:
277
+ with DDGS() as d:
278
+ return [r.get("href") for r in d.text(q, max_results=n) if r.get("href")]
279
+ except Exception as e:
280
+ print(f"⚠️ DuckDuckGo 搜索失败: {e}")
281
+ return []
282
+
283
+
284
+ # ===== 学术回补关键词 =====
285
+ ACADEMIC_BACKFILL = {
286
+ "strategy": "method OR implementation OR architecture OR workflow",
287
+ "objectives": '"objective" OR scope OR deliverable OR evaluation',
288
+ "feasibility":'resource OR timeline OR risk OR dependency OR budget',
289
+ "innovation": 'novelty OR benchmark OR comparison OR patent',
290
+ "team": 'team OR expertise OR affiliation OR governance'
291
+ }
292
+
293
+
294
+ # ===== 原子写 =====
295
+ def atomic_write(path: Path, obj):
296
+ tmp = path.with_suffix(path.suffix + ".tmp")
297
+ tmp.write_text(json.dumps(obj, ensure_ascii=False, indent=2), encoding="utf-8")
298
+ tmp.replace(path)
299
+
300
+
301
+ # ===== 轻量相关性工具 =====
302
+ _WORD_SPLIT = re.compile(r"[^\w\-\./]+")
303
+
304
+ def _tokenize(s: str):
305
+ return [t.lower() for t in _WORD_SPLIT.split(s or "") if t]
306
+
307
+ def _bm25lite_score(text_tokens, query_tokens, k1=1.2, b=0.75):
308
+ if not text_tokens or not query_tokens: return 0.0
309
+ tf = Counter(text_tokens)
310
+ L = len(text_tokens)
311
+ avgL = 400.0 # 一个经验值:提取后正文常在 400~1500 tokens
312
+ score = 0.0
313
+ # 这里没有全语料 idf,只给查询词出现一个固定 idf_boost(出现=1,不出现=0)
314
+ for q in set(query_tokens):
315
+ f = tf.get(q, 0)
316
+ if f == 0:
317
+ continue
318
+ idf = 1.6 # 经验常数,保证出现即有显著贡献
319
+ denom = f + k1 * (1 - b + b * (L / avgL))
320
+ score += idf * (f * (k1 + 1)) / (denom if denom > 0 else 1.0)
321
+ # 归一
322
+ return _clamp(score / 12.0, 0.0, 1.0)
323
+
324
+ def _phrase_boost(text: str, phrases: list):
325
+ lo = text.lower()
326
+ hits = 0
327
+ for p in phrases or []:
328
+ p = str(p or "").lower().strip()
329
+ if not p: continue
330
+ if p in lo:
331
+ hits += 1
332
+ # 每命中一个短语 +0.12,上限 1.0
333
+ return _clamp(hits * 0.12, 0.0, 1.0)
334
+
335
+ def _relevance_score(item, query: str, hints, dimension: str):
336
+ text = item.get("text","") or ""
337
+ domain = item.get("domain","") or ""
338
+ toks_text = _tokenize(text)
339
+ toks_q = _tokenize(query)
340
+
341
+ # must-have、hints 合起来算短语 boost
342
+ musts = MUST_HAVE_BY_DIM.get(dimension.lower(), [])
343
+ phrases = (hints or [])[:6] + musts
344
+ bm25p = _bm25lite_score(toks_text, toks_q)
345
+ pboost = _phrase_boost(text, phrases)
346
+ dens = info_density_score(text)
347
+ src = source_confidence(domain)
348
+ wl = 0.07 if is_whitelisted(domain) else 0.0
349
+
350
+ # 线性融合:可通过环境变量微调
351
+ w_src = float(os.getenv("WEB_RR_W_SRC", "0.45"))
352
+ w_den = float(os.getenv("WEB_RR_W_DEN", "0.20"))
353
+ w_bm25 = float(os.getenv("WEB_RR_W_BM25","0.25"))
354
+ w_phra = float(os.getenv("WEB_RR_W_PHRA","0.10"))
355
+ score = (w_src*src + w_den*dens + w_bm25*bm25p + w_phra*pboost + wl)
356
+ return _clamp(score, 0.0, 1.0)
357
+
358
+ def _shingles(text: str, n=7):
359
+ toks = _tokenize(text)
360
+ return set(tuple(toks[i:i+n]) for i in range(0, max(0, len(toks)-n+1)))
361
+
362
+ def _jaccard(a: set, b: set):
363
+ if not a or not b: return 0.0
364
+ inter = len(a & b); union = len(a | b)
365
+ return (inter / union) if union else 0.0
366
+
367
+
368
+ # ===== 主函数(保持签名)=====
369
+ def simple_search(query, max_results=8, dimension="general", hints=None, source="LLM"):
370
+ raw_q = (query or "").strip()
371
+ query = re.sub(r"\s+", " ", raw_q)
372
+
373
+ # ---- Query 智能拼接(避免重复注入)----
374
+ qlower = query.lower()
375
+ has_site = " site:" in qlower or " site(" in qlower
376
+ has_year = re.search(r"\b(19|20)\d{2}\.\.(19|20)\d{2}\b", qlower) is not None
377
+ has_neg = any(tok in qlower for tok in [" -news", " -press", " -blog", " -forum", " -finance", " -jobs"])
378
+ neg = NEGATIVE_BY_DIM.get(dimension.lower(), "")
379
+ if has_neg: # 用户已带负向,则不重复注入
380
+ neg = ""
381
+ suffix_map = {
382
+ "team": "(team expertise OR affiliation OR governance OR profile)",
383
+ "objectives": "(objective OR scope OR deliverable OR evaluation)",
384
+ "strategy": "(method OR workflow OR implementation OR architecture)",
385
+ "innovation": "(novelty OR benchmark OR differentiation OR prior work)",
386
+ "feasibility": "(resource OR timeline OR risk OR dependency OR budget)"
387
+ }
388
+ suffix = suffix_map.get(dimension.lower(), "")
389
+
390
+ site_hint = ""
391
+ if (dimension.lower() in ("strategy","feasibility","objectives")) and not has_site:
392
+ site_hint = ""
393
+
394
+ # 若 query 已含时间窗则尊重
395
+ full_query = f"{query} {suffix} {neg} {site_hint}".strip()
396
+ full_query = re.sub(r"\s+", " ", full_query)
397
+ print(f"\n🔍 [{source}] 维度 {dimension} 搜索: {full_query}")
398
+
399
+ proposal_id = os.getenv("CURRENT_PROPOSAL_ID", "default")
400
+ results_dir = Path(f"src/data/evidence/{proposal_id}")
401
+ results_dir.mkdir(parents=True, exist_ok=True)
402
+
403
+ combined_path = results_dir / f"{dimension}_combined.json"
404
+ cache_file = results_dir / f"{dimension}_cache.json"
405
+ debug_path = results_dir / f"{dimension}_debug_stats.json"
406
+ diag_path = results_dir / f"{dimension}_diag.json"
407
+
408
+ diag_rows = []
409
+ dbg = defaultdict(int)
410
+
411
+ # === 缓存命中(不覆盖 combined;空缓存不落盘)===
412
+ cache = json.loads(cache_file.read_text(encoding="utf-8")) if cache_file.exists() else {}
413
+ cache_key = f"{dimension}::{hashlib.md5(full_query.encode('utf-8')).hexdigest()[:10]}"
414
+ if cache_key in cache:
415
+ print("🧠 命中缓存")
416
+ cached_items = cache[cache_key] or []
417
+ existing = []
418
+ if combined_path.exists():
419
+ try:
420
+ existing = json.loads(combined_path.read_text(encoding="utf-8"))
421
+ except Exception:
422
+ existing = []
423
+ seen = set(x.get("url") for x in existing if isinstance(x, dict))
424
+ merged = list(existing)
425
+ for item in cached_items:
426
+ u = (item or {}).get("url")
427
+ if u and u not in seen:
428
+ merged.append(item); seen.add(u)
429
+ if cached_items:
430
+ # 排序:若启用 rerank,则对 merged 重排(只改变顺序,不过滤)
431
+ if RERANK_ENABLE:
432
+ merged = _sort_by_relevance(merged, full_query, hints, dimension)
433
+ atomic_write(combined_path, merged)
434
+
435
+ texts = [i.get("text","") for i in cached_items if isinstance(i, dict)]
436
+ urls = [i.get("url","") for i in cached_items if isinstance(i, dict)]
437
+ return texts, urls
438
+
439
+ # === 三层检索 ===
440
+ urls = []
441
+ for idx, fn in enumerate((google_search, tavily_search, duckduckgo_search_fn), start=1):
442
+ try:
443
+ got = fn(full_query, max_results)
444
+ dbg[f"api_{fn.__name__}_urls"] += len(got)
445
+ urls.extend(got)
446
+ if len(urls) >= max_results: break
447
+ except Exception:
448
+ pass
449
+ time.sleep(0.25 * idx)
450
+ dbg["urls_found_by_api"] = len(urls)
451
+
452
+ # 归一化/去重/屏蔽
453
+ norm_urls, seen = [], set()
454
+ for u in urls:
455
+ if not u: continue
456
+ nu = normalize_url(u)
457
+ host = (urlparse(nu).hostname or "").lower()
458
+ if (not host) or any(b in host for b in BLOCK_HOSTS):
459
+ dbg["filtered_block_host"] += 1; continue
460
+ if nu not in seen:
461
+ norm_urls.append(nu); seen.add(nu)
462
+ dbg["urls_after_normalize"] = len(norm_urls)
463
+
464
+ # 过少则二次尝试:中文强化
465
+ if len(norm_urls) < max_results // 2:
466
+ q2 = f"{query} 项目 研究 方法 风险 2019..2025"
467
+ more = duckduckgo_search_fn(q2, max_results)
468
+ for u in more:
469
+ nu = normalize_url(u)
470
+ host = (urlparse(nu).hostname or "").lower()
471
+ if nu not in seen and host and not any(b in host for b in BLOCK_HOSTS):
472
+ norm_urls.append(nu); seen.add(nu)
473
+ dbg["urls_after_backoff_add"] = len(norm_urls)
474
+
475
+ if not norm_urls:
476
+ print("❌ 无搜索结果")
477
+ atomic_write(combined_path, []); atomic_write(debug_path, dict(dbg))
478
+ return [], []
479
+
480
+ # === 抓取正文 ===
481
+ extracted = []
482
+ host_seen = defaultdict(int)
483
+ _host_seen_lock = threading.Lock()
484
+
485
+ def fetch_one(u):
486
+ host = (urlparse(u).hostname or "").lower()
487
+ whitelisted = is_whitelisted(host)
488
+ if _is_homepage(u) and not whitelisted:
489
+ return None, ("filtered_homepage", u)
490
+ with _host_seen_lock:
491
+ if host_seen[host] >= HOST_QUOTA and not whitelisted:
492
+ return None, ("filtered_host_quota", u)
493
+
494
+ title = ""
495
+ try:
496
+ r = _HTTP.get(u, timeout=TIMEOUT)
497
+ if r.status_code != 200:
498
+ return None, ("fetch_non200", u)
499
+ title = fetch_title(r.text)
500
+ except Exception:
501
+ pass
502
+
503
+ text = fetch_clean_text(u, dimension=dimension, title_hint=title, host=host)
504
+ if not text:
505
+ return None, ("filtered_by_rules", u)
506
+
507
+ mhs_base = MUST_HAVE_BY_DIM.get(dimension.lower(), [])
508
+ hints_low = (hints or [])[:3] if isinstance(hints, list) else []
509
+ mhs = [tok for tok in (mhs_base + hints_low) if isinstance(tok, str) and tok.strip()]
510
+
511
+ if MUST_HAVE_STRICT and mhs and not whitelisted:
512
+ low = text.lower()
513
+ if not any(tok.lower() in low for tok in mhs):
514
+ # 宽松兜底:若文本信息密度很高(≥0.6),且标题/域名具备研究/学院线索,则��行
515
+ if info_density_score(text) >= 0.60 and any(key in (title.lower()+" "+host) for key in ["lab","faculty","research","university","hospital"]):
516
+ pass
517
+ else:
518
+ return None, ("filtered_must_have_miss", u)
519
+
520
+ with _host_seen_lock:
521
+ host_seen[host] += 1
522
+
523
+ conf_base = source_confidence(host)
524
+ conf = _clamp(conf_base + 0.25 * info_density_score(text) + (0.05 if whitelisted else 0.0))
525
+ text = text.encode("utf-8","ignore").decode("utf-8","ignore")
526
+ return {"url": u, "text": text, "domain": host, "confidence": conf, "len": len(text)}, None
527
+
528
+ with ThreadPoolExecutor(max_workers=MAX_WORKERS) as ex:
529
+ futures = [ex.submit(fetch_one, u) for u in norm_urls[:max_results * 2]]
530
+ for f in as_completed(futures):
531
+ r, reason = f.result()
532
+ if r:
533
+ extracted.append(r); dbg["fetch_ok"] += 1
534
+ else:
535
+ if reason:
536
+ dbg[reason[0]] += 1
537
+ if len(diag_rows) < 400:
538
+ diag_rows.append({"url": reason[1], "reason": reason[0]})
539
+
540
+ # 学术回补
541
+ academic_hits = [x for x in extracted if any(ad in x["domain"] for ad in ACADEMIC_DOMAINS)]
542
+ if len(academic_hits) < EARLY_ACADEMIC_MIN.get(dimension, 2):
543
+ print("🧠 学术来源过少,触发 Academic 回补...")
544
+ bf = ACADEMIC_BACKFILL.get(dimension, "")
545
+ scholar_query = f'{query} ({bf}) site:(' + " OR ".join(ACADEMIC_DOMAINS[:10]) + ") 2019..2025"
546
+ more = google_search(scholar_query, 10)
547
+ for u in more:
548
+ nu = normalize_url(u)
549
+ host = (urlparse(nu).hostname or "").lower()
550
+ if nu in {x["url"] for x in extracted}: continue
551
+ with _host_seen_lock:
552
+ over = host_seen[host] >= HOST_QUOTA and not is_whitelisted(host)
553
+ if over:
554
+ continue
555
+ t = fetch_clean_text(nu, dimension=dimension, host=host)
556
+ if t and len(t) >= MIN_LEN_BASE:
557
+ extracted.append({"url": nu, "text": t, "domain": host, "confidence": 0.99, "len": len(t)})
558
+ with _host_seen_lock:
559
+ host_seen[host] += 1
560
+ if len([x for x in extracted if any(ad in x["domain"] for ad in ACADEMIC_DOMAINS)]) >= EARLY_ACADEMIC_MIN.get(dimension, 2):
561
+ break
562
+
563
+ if not extracted:
564
+ print("❌ 未抓取到正文。")
565
+ atomic_write(combined_path, []); atomic_write(debug_path, dict(dbg))
566
+ if diag_rows: atomic_write(diag_path, diag_rows)
567
+ return [], []
568
+
569
+ # === (新增)近重复抑制 + 相关性排序 ===
570
+ extracted_proc = extracted
571
+ if DEDUP_ENABLE:
572
+ extracted_proc = []
573
+ shingle_bank = []
574
+ for item in extracted:
575
+ s = _shingles(item.get("text",""), n=7)
576
+ if not s:
577
+ extracted_proc.append(item); shingle_bank.append(s); continue
578
+ similar = False
579
+ for prev in shingle_bank:
580
+ if not prev: continue
581
+ if _jaccard(s, prev) >= DEDUP_JACCARD:
582
+ similar = True; break
583
+ if not similar:
584
+ extracted_proc.append(item); shingle_bank.append(s)
585
+
586
+ if RERANK_ENABLE:
587
+ extracted_proc = _sort_by_relevance(extracted_proc, full_query, hints, dimension)
588
+
589
+ # === 合并去重(按 URL)===
590
+ existing = []
591
+ if combined_path.exists():
592
+ try: existing = json.loads(combined_path.read_text(encoding="utf-8"))
593
+ except Exception: existing = []
594
+ union = existing + extracted_proc
595
+ seen_urls, merged = set(), []
596
+ for item in union:
597
+ u = item.get("url")
598
+ if u and u not in seen_urls:
599
+ merged.append(item); seen_urls.add(u)
600
+
601
+ # 重排 merged 顺序(不丢数据,只调整排序)
602
+ if RERANK_ENABLE:
603
+ merged = _sort_by_relevance(merged, full_query, hints, dimension)
604
+
605
+ # === 写 combined & cache ===
606
+ atomic_write(combined_path, merged)
607
+ cache[cache_key] = merged
608
+ atomic_write(cache_file, cache)
609
+
610
+ # === 统计 ===
611
+ academic_hits = [x for x in merged if any(ad in x["domain"] for ad in ACADEMIC_DOMAINS)]
612
+ avg_conf = round(sum(x["confidence"] for x in merged) / len(merged), 2) if merged else 0
613
+ ratio = round((len(academic_hits) / len(merged)), 2) if merged else 0.0
614
+ print(f"✅ 抓取完成:{len(merged)} 条 | 平均置信度 {avg_conf} | 学术比例 {ratio:.0%}")
615
+ print(f"🔝 Top 域名: {Counter([x['domain'] for x in merged]).most_common(5)}")
616
+
617
+ dbg["kept_total"] = len(merged); dbg["academic_hits"] = len(academic_hits)
618
+ atomic_write(debug_path, dict(dbg))
619
+ if diag_rows: atomic_write(diag_path, diag_rows)
620
+
621
+ # === 维度摘要(保持不变)===
622
+ try:
623
+ if merged and LLM_CLIENT:
624
+ joined = "\n\n".join(x["text"][:600] for x in merged[:3])
625
+ refs = "\n".join([f"[{i+1}] {x['url']}" for i, x in enumerate(merged[:3])])
626
+ prompt = (
627
+ f"请基于以下网页内容,总结维度 {dimension} 的关键信息(120~160字,学术语气,避免夸大/臆测;涉及年份/批准必须引用 [x])。\n\n"
628
+ f"{joined}\n\n参考:\n{refs}"
629
+ )
630
+ res = LLM_CLIENT.chat_completions.create( # 兼容某些 SDK 写法差异
631
+ model=LLM_MODEL, messages=[{"role": "user", "content": prompt}], temperature=0.3,
632
+ ) if hasattr(LLM_CLIENT, "chat_completions") else LLM_CLIENT.chat.completions.create(
633
+ model=LLM_MODEL, messages=[{"role": "user", "content": prompt}], temperature=0.3,
634
+ )
635
+ summary = (res.choices[0].message.content or "").strip()
636
+ atomic_write(results_dir / f"{dimension}_summary.json", {
637
+ "dimension": dimension, "summary": summary,
638
+ "urls": [x['url'] for x in merged], "avg_conf": avg_conf, "academic_ratio": ratio
639
+ })
640
+ except Exception as e:
641
+ print(f"⚠️ 摘要生成失败: {e}")
642
+
643
+ texts = [x["text"] for x in merged]
644
+ urls_extracted = [x["url"] for x in merged]
645
+ return texts, urls_extracted
646
+
647
+
648
+ # ===== 排序主函数(集中定义,便于切换算法)=====
649
+ def _sort_by_relevance(items, query, hints, dimension):
650
+ scored = []
651
+ for it in items:
652
+ s = _relevance_score(it, query, hints, dimension)
653
+ # 提升与维度 Must-Have 命中数(用于 tie-break)
654
+ mh = 0
655
+ low = (it.get("text","") or "").lower()
656
+ for tok in MUST_HAVE_BY_DIM.get(dimension.lower(), []):
657
+ if tok.lower() in low:
658
+ mh += 1
659
+ scored.append((s, mh, it))
660
+ # 先按得分,再按 must-have 命中数,最后轻微倾向更长文本
661
+ scored.sort(key=lambda x: (x[0], x[1], min(2600, x[2].get("len",0))), reverse=True)
662
+ return [it for _,__,it in scored]
src/backend/utils/__init__.py ADDED
File without changes
src/backend/utils/model_selector.py ADDED
@@ -0,0 +1,64 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ 通用模型选择器(v5.1 · 兼容新旧 Gemini SDK + 明确 Key 优先级)
4
+ - PROVIDER: openai / deepseek / gemini
5
+ - 返回: {"client": <client_instance>, "model_name": str, "provider": str}
6
+ """
7
+
8
+ import os
9
+ from dotenv import load_dotenv
10
+
11
+ load_dotenv()
12
+
13
+ def get_llm_client():
14
+ provider = (os.getenv("PROVIDER", "openai") or "openai").lower().strip()
15
+
16
+ # ========== OpenAI ==========
17
+ if provider in ("openai", "chatgpt"):
18
+ from openai import OpenAI
19
+ api_key = os.getenv("OPENAI_API_KEY")
20
+ if not api_key:
21
+ raise RuntimeError("OPENAI_API_KEY 未配置")
22
+ client = OpenAI(api_key=api_key)
23
+ model_name = os.getenv("OPENAI_MODEL", "gpt-4o-mini")
24
+ print(f"✅ 已加载 OpenAI 模型:{model_name}")
25
+ return {"client": client, "model_name": model_name, "provider": "openai"}
26
+
27
+ # ========== DeepSeek ==========
28
+ elif provider == "deepseek":
29
+ from openai import OpenAI
30
+ api_key = os.getenv("DEEPSEEK_API_KEY")
31
+ if not api_key:
32
+ raise RuntimeError("DEEPSEEK_API_KEY 未配置")
33
+ client = OpenAI(api_key=api_key, base_url="https://api.deepseek.com/v1")
34
+ model_name = os.getenv("DEEPSEEK_MODEL", "deepseek-chat")
35
+ print(f"✅ 已加载 DeepSeek 模型:{model_name}")
36
+ return {"client": client, "model_name": model_name, "provider": "deepseek"}
37
+
38
+ # ========== Gemini ==========
39
+ elif provider == "gemini":
40
+ # 统一优先 GEMINI_API_KEY;若仅有 GOOGLE_API_KEY 也做兜底
41
+ api_key = (os.getenv("GEMINI_API_KEY") or os.getenv("GOOGLE_API_KEY") or "").strip()
42
+ if not api_key:
43
+ raise RuntimeError("GEMINI_API_KEY 未配置(未找到 GOOGLE_API_KEY 兜底)")
44
+
45
+ # 兼容新旧 SDK:优先使用新 SDK(from google import genai),失败则回退到 google.generativeai
46
+ client = None
47
+ try:
48
+ from google import genai # 新 SDK
49
+ client = genai.Client(api_key=api_key)
50
+ using_new = True
51
+ except Exception:
52
+ import google.generativeai as genai # 旧 SDK
53
+ genai.configure(api_key=api_key)
54
+ # 旧 SDK 没有 Client 实例的语义,这里用模块本身作为“客户端”占位
55
+ client = genai
56
+ using_new = False
57
+
58
+ model_name = os.getenv("GEMINI_MODEL", "gemini-2.5-flash").strip()
59
+ suffix = "(新SDK)" if using_new else "(旧SDK)"
60
+ print(f"✅ 已加载 Gemini 模型:{model_name} {suffix}")
61
+ return {"client": client, "model_name": model_name, "provider": "gemini"}
62
+
63
+ else:
64
+ raise ValueError(f"❌ 未知 PROVIDER:{provider}(应为 openai / deepseek / gemini)")
src/main.py ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from dotenv import load_dotenv
2
+ load_dotenv()
3
+
4
+ import os
5
+ import json
6
+ import argparse
7
+
8
+ from backend.chains.base_chain import BaseChain
9
+ from backend.chains.orchestrator import run_all, save_full_report
10
+
11
+ def parse_args():
12
+ p = argparse.ArgumentParser(description="RAG-6View CLI")
13
+ p.add_argument("--mode", choices=["single", "all"], default="all",
14
+ help="single=只跑一个维度; all=并行跑所有维度")
15
+ p.add_argument("--dimension", default="team", help="在 single 模式下指定维度")
16
+ p.add_argument("--question", default=None, help="自定义问题(可选)")
17
+ p.add_argument("--max_workers", type=int, default=3, help="并行线程数")
18
+ return p.parse_args()
19
+
20
+ def run_single(dim: str, question: str = None):
21
+ print("🚀 Starting RAG Demo (Single)...")
22
+ chain = BaseChain(dim)
23
+ q = question or "Does the core team have strong research capability?"
24
+ result = chain.run(q)
25
+ os.makedirs("data/results", exist_ok=True)
26
+ out = "data/results/single_result.json"
27
+ with open(out, "w", encoding="utf-8") as f:
28
+ json.dump(result, f, indent=2, ensure_ascii=False)
29
+ print(f"\n✅ 单维度分析完成!结果已保存到: {out}")
30
+ print(json.dumps(result, indent=2, ensure_ascii=False))
31
+
32
+ def run_all_dims(max_workers: int):
33
+ print("🚀 Starting RAG Demo (ALL Dimensions, parallel)...")
34
+ report = run_all(max_workers=max_workers)
35
+ out_path = save_full_report(report)
36
+ print(f"\n✅ 全维度分析完成!报告已保存到: {out_path}")
37
+ print(json.dumps(report["summary"], indent=2, ensure_ascii=False))
38
+
39
+ def main():
40
+ args = parse_args()
41
+ if args.mode == "single":
42
+ run_single(args.dimension, args.question)
43
+ else:
44
+ run_all_dims(args.max_workers)
45
+
46
+ if __name__ == "__main__":
47
+ main()
src/profiling/__init__.py ADDED
File without changes
src/profiling/domain_profiler.py ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import os
7
+ from pathlib import Path
8
+ from typing import Any, Dict
9
+
10
+ from dotenv import load_dotenv
11
+ from openai import OpenAI
12
+
13
+ from src.prompting.domain_adaptive import PROFILER_SYSTEM_PROMPT, sanitize_domain_profile
14
+
15
+ load_dotenv()
16
+ BASE_DIR = Path(__file__).resolve().parents[2]
17
+ DATA_DIR = BASE_DIR / "src" / "data"
18
+ PREPARED_DIR = DATA_DIR / "prepared"
19
+ EXTRACTED_DIR = DATA_DIR / "extracted"
20
+ OPENAI_MODEL = os.getenv("OPENAI_MODEL", "gpt-4o-mini")
21
+
22
+
23
+ def detect_latest_pid() -> str:
24
+ if not PREPARED_DIR.exists():
25
+ return "unknown"
26
+ dirs = [d for d in PREPARED_DIR.iterdir() if d.is_dir()]
27
+ if not dirs:
28
+ return "unknown"
29
+ dirs.sort(key=lambda p: p.stat().st_mtime, reverse=True)
30
+ return dirs[0].name
31
+
32
+
33
+ def load_source_text(pid: str) -> str:
34
+ for candidate in [PREPARED_DIR / pid / "full_text.txt", EXTRACTED_DIR / pid / "dimensions_v2.json"]:
35
+ if candidate.exists() and candidate.suffix == '.txt':
36
+ return candidate.read_text(encoding='utf-8', errors='ignore')[:18000]
37
+ dim_path = EXTRACTED_DIR / pid / "dimensions_v2.json"
38
+ if dim_path.exists():
39
+ return dim_path.read_text(encoding='utf-8', errors='ignore')[:18000]
40
+ raise FileNotFoundError(f"No source text found for proposal_id={pid}")
41
+
42
+
43
+ def profile_with_llm(text: str) -> Dict[str, Any]:
44
+ client = OpenAI()
45
+ resp = client.chat.completions.create(
46
+ model=OPENAI_MODEL,
47
+ messages=[
48
+ {"role": "system", "content": PROFILER_SYSTEM_PROMPT},
49
+ {"role": "user", "content": text},
50
+ ],
51
+ response_format={"type": "json_object"},
52
+ temperature=0.0,
53
+ max_tokens=900,
54
+ )
55
+ raw = resp.choices[0].message.content
56
+ try:
57
+ data = json.loads(raw)
58
+ except Exception:
59
+ data = {}
60
+ return sanitize_domain_profile(data)
61
+
62
+
63
+ def run_domain_profiler(proposal_id: str) -> Path:
64
+ text = load_source_text(proposal_id)
65
+ profile = profile_with_llm(text)
66
+ out_dir = EXTRACTED_DIR / proposal_id
67
+ out_dir.mkdir(parents=True, exist_ok=True)
68
+ out_path = out_dir / 'domain_profile.json'
69
+ out_path.write_text(json.dumps(profile, ensure_ascii=False, indent=2), encoding='utf-8')
70
+ print(f"🧭 Domain profile saved -> {out_path}")
71
+ return out_path
72
+
73
+
74
+ def main():
75
+ ap = argparse.ArgumentParser(description='Generate domain_profile.json for the current proposal')
76
+ ap.add_argument('--proposal_id', type=str, default='')
77
+ args = ap.parse_args()
78
+ pid = args.proposal_id.strip() or detect_latest_pid()
79
+ run_domain_profiler(pid)
80
+
81
+
82
+ if __name__ == '__main__':
83
+ main()
src/prompting/__init__.py ADDED
File without changes
src/prompting/domain_adaptive.py ADDED
@@ -0,0 +1,354 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ from __future__ import annotations
3
+
4
+ GLOBAL_PROMPT_SUFFIX = """
5
+ If quantitative metrics are not explicitly provided, do not penalize the proposal by default.
6
+ Instead, infer quality from contextual evidence, domain knowledge, and implicit signals.
7
+ """.strip()
8
+
9
+ import json
10
+ from dataclasses import dataclass
11
+ from pathlib import Path
12
+ from typing import Any, Dict, List
13
+
14
+ DIM_ORDER = ["team", "objectives", "strategy", "innovation", "feasibility"]
15
+
16
+ UNIVERSAL_REVIEW_TEMPLATES: Dict[str, Dict[str, Any]] = {
17
+ "T1_problem": {
18
+ "title": "Problem / Need / Motivation",
19
+ "dimension": "objectives",
20
+ "question_template": (
21
+ "Assess whether the proposal addresses an important and well-motivated problem. "
22
+ "Pay special attention to: {evaluation_focus.problem}. "
23
+ "Respect domain terminology: {terminology}. "
24
+ + GLOBAL_PROMPT_SUFFIX
25
+ ),
26
+ },
27
+ "T2_objectives": {
28
+ "title": "Objectives / Scope",
29
+ "dimension": "objectives",
30
+ "question_template": (
31
+ "Assess whether the proposal objectives are clear, specific, and appropriately scoped. "
32
+ "Pay special attention to: {evaluation_focus.objectives}. "
33
+ "Respect domain terminology: {terminology}. "
34
+ + GLOBAL_PROMPT_SUFFIX
35
+ ),
36
+ },
37
+ "T3_methods": {
38
+ "title": "Methods / Technical Approach",
39
+ "dimension": "strategy",
40
+ "question_template": (
41
+ "Assess whether the proposed methods or technical approach are coherent and fit the objectives. "
42
+ "Method signals: {methods}. "
43
+ "Respect domain terminology: {terminology}. "
44
+ + GLOBAL_PROMPT_SUFFIX
45
+ ),
46
+ },
47
+ "T4_evidence": {
48
+ "title": "Evidence / Data / Resources",
49
+ "dimension": "feasibility",
50
+ "question_template": (
51
+ "Assess whether the proposal provides adequate evidence, data, resources, or enabling conditions to support the work. "
52
+ "Method signals: {methods}. "
53
+ "Respect domain terminology: {terminology}. "
54
+ + GLOBAL_PROMPT_SUFFIX
55
+ ),
56
+ },
57
+ "T5_feasibility": {
58
+ "title": "Feasibility / Execution",
59
+ "dimension": "feasibility",
60
+ "question_template": (
61
+ "Assess whether the work appears executable in practice. "
62
+ "Pay special attention to: {evaluation_focus.feasibility}. "
63
+ "Risk signals: {risks}. "
64
+ + GLOBAL_PROMPT_SUFFIX
65
+ ),
66
+ },
67
+ "T6_innovation": {
68
+ "title": "Innovation / Differentiation",
69
+ "dimension": "innovation",
70
+ "question_template": (
71
+ "Assess the novelty or differentiating contribution of the proposal. "
72
+ "Pay special attention to: {evaluation_focus.innovation}. "
73
+ "Respect domain terminology: {terminology}. "
74
+ + GLOBAL_PROMPT_SUFFIX
75
+ ),
76
+ },
77
+ "T7_risks": {
78
+ "title": "Risks / Failure Modes / Mitigation",
79
+ "dimension": "feasibility",
80
+ "question_template": (
81
+ "Identify the major risks and evaluate whether mitigation thinking is credible. "
82
+ "Risk signals: {risks}. "
83
+ "Method signals: {methods}. "
84
+ + GLOBAL_PROMPT_SUFFIX
85
+ ),
86
+ },
87
+ "T8_team": {
88
+ "title": "Team / Capability / Governance",
89
+ "dimension": "team",
90
+ "question_template": (
91
+ "Assess whether the team and governance setup are adequate for the proposed work. "
92
+ "Pay special attention to: {evaluation_focus.team}. "
93
+ "Respect domain terminology: {terminology}. "
94
+ + GLOBAL_PROMPT_SUFFIX
95
+ ),
96
+ },
97
+ "T9_outcomes": {
98
+ "title": "Outcomes / Impact / Evaluation",
99
+ "dimension": "objectives",
100
+ "question_template": (
101
+ "Assess whether the expected outcomes, impact claims, and evaluation logic are credible. "
102
+ "Pay special attention to: {evaluation_focus.outcomes}. "
103
+ "Respect domain terminology: {terminology}. "
104
+ + GLOBAL_PROMPT_SUFFIX
105
+ ),
106
+ },
107
+ }
108
+
109
+ ALLOWED_SLOTS = {
110
+ "evaluation_focus.problem",
111
+ "evaluation_focus.objectives",
112
+ "evaluation_focus.feasibility",
113
+ "evaluation_focus.innovation",
114
+ "evaluation_focus.team",
115
+ "evaluation_focus.outcomes",
116
+ "methods",
117
+ "risks",
118
+ "terminology",
119
+ }
120
+
121
+ TEMPLATE_SLOT_MAP = {
122
+ "T1_problem": ["evaluation_focus.problem", "terminology"],
123
+ "T2_objectives": ["evaluation_focus.objectives", "terminology"],
124
+ "T3_methods": ["methods", "terminology"],
125
+ "T4_evidence": ["methods", "terminology"],
126
+ "T5_feasibility": ["evaluation_focus.feasibility", "risks"],
127
+ "T6_innovation": ["evaluation_focus.innovation", "terminology"],
128
+ "T7_risks": ["risks", "methods"],
129
+ "T8_team": ["evaluation_focus.team", "terminology"],
130
+ "T9_outcomes": ["evaluation_focus.outcomes", "terminology"],
131
+ }
132
+
133
+
134
+ @dataclass(frozen=True)
135
+ class ReviewTask:
136
+ task_id: str
137
+ template_id: str
138
+ dimension: str
139
+ title: str
140
+
141
+
142
+ REVIEW_TASKS: List[ReviewTask] = [
143
+ ReviewTask("problem", "T1_problem", "objectives", "Problem significance"),
144
+ ReviewTask("objectives", "T2_objectives", "objectives", "Objectives and scope"),
145
+ ReviewTask("methods", "T3_methods", "strategy", "Methods and approach"),
146
+ ReviewTask("evidence", "T4_evidence", "feasibility", "Evidence and resources"),
147
+ ReviewTask("feasibility", "T5_feasibility", "feasibility", "Execution feasibility"),
148
+ ReviewTask("innovation", "T6_innovation", "innovation", "Innovation"),
149
+ ReviewTask("risks", "T7_risks", "feasibility", "Risks and mitigation"),
150
+ ReviewTask("team", "T8_team", "team", "Team and governance"),
151
+ ReviewTask("outcomes", "T9_outcomes", "objectives", "Outcomes and evaluation"),
152
+ ]
153
+
154
+ QUESTION_SEARCH_HINTS = {
155
+ "team": ["team expertise", "roles and responsibilities", "governance"],
156
+ "objectives": ["problem significance", "scope", "outcomes", "evaluation"],
157
+ "strategy": ["methods", "technical approach", "workflow"],
158
+ "innovation": ["novelty", "differentiation", "prior work"],
159
+ "feasibility": ["resources", "risks", "execution plan", "dependencies"],
160
+ }
161
+
162
+ PROFILER_SYSTEM_PROMPT = """
163
+ You are a strict domain profiler for proposal review.
164
+
165
+ Your task is ONLY to derive a clean domain profile from the CURRENT document text.
166
+
167
+ Rules:
168
+ - Use ONLY the provided document text.
169
+ - Ignore any prior tasks, prior files, templates, examples, cached memory, or default domains.
170
+ - Do not rewrite the proposal.
171
+ - Do not generate review questions.
172
+ - Do not invent domain details not supported by the text.
173
+ - Prefer broad domain labels over overly specific labels.
174
+ - Keep all lists short, concrete, and normalized.
175
+ - If evidence is weak, return "unknown".
176
+ - Return valid JSON only.
177
+
178
+ Return JSON with this structure:
179
+ {
180
+ "domain": {
181
+ "primary": "string",
182
+ "secondary": ["string"]
183
+ },
184
+ "evaluation_focus": {
185
+ "problem": ["string"],
186
+ "objectives": ["string"],
187
+ "feasibility": ["string"],
188
+ "innovation": ["string"],
189
+ "team": ["string"],
190
+ "outcomes": ["string"]
191
+ },
192
+ "methods": ["string"],
193
+ "risks": ["string"],
194
+ "terminology": ["string"]
195
+ }
196
+ """.strip()
197
+
198
+
199
+ def normalize_list(items: Any, max_items: int = 6) -> List[str]:
200
+ if items is None:
201
+ return []
202
+
203
+ if isinstance(items, str):
204
+ items = [items]
205
+ elif not isinstance(items, (list, tuple, set)):
206
+ items = [items]
207
+
208
+ # Step 1: 基础清洗
209
+ cleaned = []
210
+ for item in items:
211
+ text = " ".join(str(item).split()).strip()
212
+ if not text:
213
+ continue
214
+
215
+ # 去掉纯标点(关键修复)
216
+ if text in {"、", ";", ";", ",", ".", "。"}:
217
+ continue
218
+
219
+ cleaned.append(text)
220
+
221
+ # Step 2: 合并“单字碎片”(核心修复)
222
+ merged = []
223
+ buffer = ""
224
+
225
+ for token in cleaned:
226
+ # 如果是单个中文字符 → 拼接
227
+ if len(token) == 1 and '\u4e00' <= token <= '\u9fff':
228
+ buffer += token
229
+ else:
230
+ if buffer:
231
+ merged.append(buffer)
232
+ buffer = ""
233
+ merged.append(token)
234
+
235
+ if buffer:
236
+ merged.append(buffer)
237
+
238
+ # Step 3: 去重 + 限制长度
239
+ seen, out = set(), []
240
+ for text in merged:
241
+ key = text.lower()
242
+ if key in seen:
243
+ continue
244
+ seen.add(key)
245
+ out.append(text)
246
+
247
+ if len(out) >= max_items:
248
+ break
249
+ return out
250
+
251
+
252
+ def _resolve(profile: Dict[str, Any], slot: str) -> str:
253
+ node: Any = profile
254
+ for part in slot.split("."):
255
+ if not isinstance(node, dict):
256
+ return "not specified in domain profile"
257
+ node = node.get(part)
258
+ if node is None:
259
+ return "not specified in domain profile"
260
+
261
+ if isinstance(node, list):
262
+ vals = normalize_list(node)
263
+ return "; ".join(vals) if vals else "not specified in domain profile"
264
+
265
+ if isinstance(node, str):
266
+ text = " ".join(node.split()).strip()
267
+ return text or "not specified in domain profile"
268
+
269
+ text = " ".join(str(node).split()).strip()
270
+ return text or "not specified in domain profile"
271
+
272
+
273
+ def sanitize_domain_profile(profile: Dict[str, Any]) -> Dict[str, Any]:
274
+ if not isinstance(profile, dict):
275
+ profile = {}
276
+
277
+ domain = profile.get("domain") if isinstance(profile.get("domain"), dict) else {}
278
+ eval_focus = profile.get("evaluation_focus") if isinstance(profile.get("evaluation_focus"), dict) else {}
279
+
280
+ primary = " ".join(str(domain.get("primary", "unknown")).split()).strip() or "unknown"
281
+ bad_domains = {
282
+ "general proposal",
283
+ "general",
284
+ "unknown",
285
+ "misc",
286
+ "other",
287
+ "n/a",
288
+ "na",
289
+ }
290
+ if primary.lower() in bad_domains:
291
+ primary = "unknown"
292
+
293
+ methods = normalize_list(profile.get("methods", []), max_items=6)
294
+ risks = normalize_list(profile.get("risks", []), max_items=6)
295
+ terminology = normalize_list(profile.get("terminology", []), max_items=12)
296
+
297
+ # 如果 methods / risks 太碎,就尽量用 terminology 补救
298
+ def _too_fragmented(values: List[str]) -> bool:
299
+ if not values:
300
+ return True
301
+ short_count = len([x for x in values if len(x) <= 1])
302
+ return short_count >= max(2, len(values) // 2 + 1)
303
+
304
+ if _too_fragmented(methods):
305
+ fallback_methods = [t for t in terminology if len(t) >= 2][:6]
306
+ if fallback_methods:
307
+ methods = fallback_methods
308
+
309
+ if _too_fragmented(risks):
310
+ fallback_risks = [t for t in terminology if len(t) >= 2][:6]
311
+ if fallback_risks:
312
+ risks = fallback_risks
313
+
314
+ return {
315
+ "domain": {
316
+ "primary": primary,
317
+ "secondary": normalize_list(domain.get("secondary", []), max_items=4),
318
+ },
319
+ "evaluation_focus": {
320
+ "problem": normalize_list(eval_focus.get("problem", []), max_items=5),
321
+ "objectives": normalize_list(eval_focus.get("objectives", []), max_items=5),
322
+ "feasibility": normalize_list(eval_focus.get("feasibility", []), max_items=5),
323
+ "innovation": normalize_list(eval_focus.get("innovation", []), max_items=5),
324
+ "team": normalize_list(eval_focus.get("team", []), max_items=5),
325
+ "outcomes": normalize_list(eval_focus.get("outcomes", []), max_items=5),
326
+ },
327
+ "methods": methods,
328
+ "risks": risks,
329
+ "terminology": terminology,
330
+ }
331
+
332
+
333
+ def inject_template(template_id: str, profile: Dict[str, Any]) -> str:
334
+ template = UNIVERSAL_REVIEW_TEMPLATES[template_id]["question_template"]
335
+ safe_profile = sanitize_domain_profile(profile)
336
+ rendered = template
337
+ for slot in TEMPLATE_SLOT_MAP[template_id]:
338
+ rendered = rendered.replace("{" + slot + "}", _resolve(safe_profile, slot))
339
+ return rendered
340
+
341
+
342
+ def build_specialized_question(task: ReviewTask, profile: Dict[str, Any]) -> str:
343
+ injected = inject_template(task.template_id, profile)
344
+ return f"[{task.title}] {injected}"
345
+
346
+
347
+ def load_domain_profile(path: Path) -> Dict[str, Any]:
348
+ if not path.exists():
349
+ return sanitize_domain_profile({})
350
+ try:
351
+ data = json.loads(path.read_text(encoding="utf-8"))
352
+ except Exception:
353
+ return sanitize_domain_profile({})
354
+ return sanitize_domain_profile(data)
src/tools/__init__.py ADDED
File without changes
src/tools/ai_expert_opinion.py ADDED
@@ -0,0 +1,875 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ AI Expert Opinion · v4.1
4
+ (dimension-first, QA-grounded, general_insights-aware, with local fallback)
5
+ --------------------------------------------------------------------
6
+ 设计目标:
7
+ - 严格基于 post_processing_v2 的 metrics.json + final_payload.json(已选中答案)
8
+ - 先对五个维度逐一做“带证据 + 领域通识层”的专家点评,再在本地代码中汇总成总体意见
9
+ - 不让 LLM 看到任何具体分数,仅提供“强/中/弱”的文字提示,避免分数泄漏
10
+ - 默认调用 OpenAI(.env 中的 OPENAI_*),无法调用时自动退化为“纯本地规则版专家评审”(不依赖 LLM)
11
+ - 总体意见采用「一段总括 + 分维度 bullet」形式,更适合给人看
12
+ """
13
+
14
+ import os
15
+ import re
16
+ import json
17
+ import time
18
+ import argparse
19
+ from pathlib import Path
20
+ from datetime import datetime
21
+ from typing import Dict, Any, List, Tuple
22
+
23
+ import requests
24
+ from dotenv import load_dotenv
25
+
26
+ # ----------------- 路径与常量 -----------------
27
+ BASE_DIR = Path(__file__).resolve().parents[2]
28
+ DATA_DIR = BASE_DIR / "src" / "data"
29
+ REFINED_ROOT = DATA_DIR / "refined_answers"
30
+ EXPERT_DIR = DATA_DIR / "expert_reports"
31
+
32
+ DIM_ORDER = ["team", "objectives", "strategy", "innovation", "feasibility"]
33
+ DIM_LABELS_ZH = {
34
+ "team": "团队与治理",
35
+ "objectives": "项目目标",
36
+ "strategy": "实施路径与战略",
37
+ "innovation": "技术与产品创新",
38
+ "feasibility": "资源与可行性"
39
+ }
40
+
41
+ # ----------------- 环境变量 -----------------
42
+ load_dotenv()
43
+ PROVIDER = os.getenv("PROVIDER", "openai").strip().lower()
44
+ OPENAI_API_KEY = os.getenv("OPENAI_API_KEY", "").strip()
45
+ OPENAI_API_BASE = os.getenv("OPENAI_API_BASE", "https://api.openai.com/v1").strip()
46
+ OPENAI_MODEL = os.getenv("OPENAI_MODEL", "gpt-4o-mini").strip()
47
+ TIMEOUT_CONNECT = int(os.getenv("HTTP_TIMEOUT_CONNECT", "12"))
48
+ TIMEOUT_READ = int(os.getenv("HTTP_TIMEOUT_READ", "60"))
49
+
50
+
51
+ # ----------------- 小工具 -----------------
52
+ def now_str() -> str:
53
+ return datetime.now().strftime("%Y-%m-%d %H:%M:%S")
54
+
55
+
56
+ def read_json(p: Path) -> Any:
57
+ return json.loads(p.read_text(encoding="utf-8"))
58
+
59
+
60
+ def write_json(p: Path, obj: Any) -> None:
61
+ p.parent.mkdir(parents=True, exist_ok=True)
62
+ p.write_text(json.dumps(obj, ensure_ascii=False, indent=2), encoding="utf-8")
63
+
64
+
65
+ def detect_latest_pid() -> str:
66
+ """从 refined_answers 下挑选最近更新且包含 postproc/metrics.json 的 pid"""
67
+ if not REFINED_ROOT.exists():
68
+ return ""
69
+ cands: List[Tuple[str, float]] = []
70
+ for d in REFINED_ROOT.iterdir():
71
+ if not d.is_dir():
72
+ continue
73
+ postproc_dir = d / "postproc"
74
+ if (postproc_dir / "metrics.json").exists() and (postproc_dir / "final_payload.json").exists():
75
+ cands.append((d.name, (postproc_dir / "metrics.json").stat().st_mtime))
76
+ cands.sort(key=lambda x: x[1], reverse=True)
77
+ return cands[0][0] if cands else ""
78
+
79
+
80
+ # ----------------- 维度级打分信号 -> 文字提示 -----------------
81
+ def _score_hint(v: float) -> str:
82
+ try:
83
+ v = float(v)
84
+ except Exception:
85
+ return "得分信号不明(信息可能不足)"
86
+ if v >= 0.75:
87
+ return "得分偏高,整体表现较强"
88
+ if v >= 0.62:
89
+ return "得分中上,有明显优势,但仍存在可优化空间"
90
+ if v >= 0.50:
91
+ return "得分中等偏弱,存在若干短板或信息缺口"
92
+ if v >= 0.35:
93
+ return "得分偏低,说明该维度存在明显不足或证据有限"
94
+ return "得分很低,属于明显短板,需要重点关注与补救"
95
+
96
+
97
+ def _align_hint(v: float) -> str:
98
+ try:
99
+ v = float(v)
100
+ except Exception:
101
+ return "跨模型一致性信号不明"
102
+ if v >= 0.8:
103
+ return "多模型之间观点高度一致,结论较稳健"
104
+ if v >= 0.6:
105
+ return "多模型之间观点大致一致,少量差异"
106
+ if v >= 0.4:
107
+ return "多模型之间存在较多分歧,需要谨慎解读"
108
+ return "多模型观点差异较大,该维度结论不稳定"
109
+
110
+
111
+ def _drift_hint(v: float) -> str:
112
+ try:
113
+ v = float(v)
114
+ except Exception:
115
+ return "内容漂移信号不明"
116
+ if v <= 0.18:
117
+ return "回答围绕同一核心,内容漂移较低"
118
+ if v <= 0.30:
119
+ return "回答存在一定漂移,但整体仍围绕同一主题"
120
+ if v <= 0.45:
121
+ return "回答存在明显漂移,需要甄别哪些点是稳定共识"
122
+ return "回答漂移程度较高,该维度存在语义不稳定风险"
123
+
124
+ def _split_keywords(text: str) -> List[str]:
125
+ """
126
+ 非严格分词:把 general_insights 里的句子切成若干关键片段,
127
+ 过滤掉太短的 token,用来做“是否在问答中被覆盖”的粗匹配。
128
+ """
129
+ if not text:
130
+ return []
131
+ tokens = re.split(r"[,。,.;;、/\\()()\s]+", text)
132
+ tokens = [t.strip().lower() for t in tokens if len(t.strip()) >= 3]
133
+ return tokens
134
+
135
+ def build_dim_inputs(metrics: Dict[str, Any],
136
+ final_payload: Dict[str, Any],
137
+ max_qas: int = 6,
138
+ max_answer_chars: int = 800) -> Dict[str, Any]:
139
+ """
140
+ 组装传给 LLM 的维度输入:
141
+ - 不包含任何具体分数,只给“强/中/弱”的文字提示
142
+ - 注入 post_processing_v2 新增的 top_evidence_phrases / general_insights
143
+ """
144
+ dim_inputs: Dict[str, Any] = {}
145
+ dim_metrics = metrics.get("dimensions", {}) or {}
146
+ fp_dims = final_payload.get("dimensions", {}) or {}
147
+
148
+ for dim in DIM_ORDER:
149
+ m = dim_metrics.get(dim, {}) or {}
150
+ f = fp_dims.get(dim, {}) or {}
151
+ qas = f.get("qas", []) or []
152
+
153
+ # 维度级通识经验层(general_insights)
154
+ dim_general_insights = f.get("general_insights") or []
155
+ # 证据短语(来自 post_processing_v2)
156
+ top_evid_phrases = m.get("top_evidence_phrases") or []
157
+ redlined_samples = m.get("redlined_samples") or []
158
+
159
+ # 汇总该维度所有问答内容,用于和 general_insights 做粗匹配
160
+ corpus_parts: List[str] = []
161
+ for qa in qas:
162
+ corpus_parts.append((qa.get("q") or ""))
163
+ corpus_parts.append((qa.get("answer") or ""))
164
+ for c in qa.get("claims") or []:
165
+ corpus_parts.append(c)
166
+ for h in qa.get("evidence_hints") or []:
167
+ corpus_parts.append(h)
168
+ corpus_text = " ".join(corpus_parts).lower()
169
+
170
+ # 将维度级 general_insights 分成:已部分覆盖 / 明显缺口
171
+ dim_general_insights_covered: List[str] = []
172
+ dim_general_insights_missing: List[str] = []
173
+ for gi in dim_general_insights:
174
+ if not gi:
175
+ continue
176
+ gi_tokens = _split_keywords(gi)
177
+ # 没有有效 token 的,直接当作“缺口提示”(避免误判为 covered)
178
+ if not gi_tokens:
179
+ dim_general_insights_missing.append(gi)
180
+ continue
181
+ hit = any(tok in corpus_text for tok in gi_tokens)
182
+ if hit:
183
+ dim_general_insights_covered.append(gi)
184
+ else:
185
+ dim_general_insights_missing.append(gi)
186
+
187
+ samples = []
188
+ for qa in qas[:max_qas]:
189
+ ans = (qa.get("answer") or "").strip()
190
+ if len(ans) > max_answer_chars:
191
+ ans = ans[:max_answer_chars] + "……"
192
+ samples.append({
193
+ "question": (qa.get("q") or "").strip(),
194
+ "answer": ans,
195
+ "key_claims": (qa.get("claims") or [])[:6],
196
+ "evidence_hints": (qa.get("evidence_hints") or [])[:6],
197
+ "provider": qa.get("provider", ""),
198
+ # 逐问通识经验层(仅作行业基准提示,不代表本项目已实现)
199
+ "general_insights": (qa.get("general_insights") or [])[:6],
200
+ })
201
+
202
+ dim_inputs[dim] = {
203
+ "dimension": dim,
204
+ "label_zh": DIM_LABELS_ZH.get(dim, dim),
205
+ "score_hint": _score_hint(m.get("avg")),
206
+ "alignment_hint": _align_hint(m.get("avg_alignment")),
207
+ "drift_hint": _drift_hint(m.get("avg_drift")),
208
+ "metric_strength_phrases": (m.get("strengths") or [])[:6],
209
+ "metric_risk_phrases": (m.get("risks") or [])[:6],
210
+ "metric_top_evidence_phrases": top_evid_phrases[:6],
211
+ "metric_redlined_samples": redlined_samples[:6],
212
+ # 维度级通识拆分:已部分在问答中体现 / 基本缺位
213
+ "dim_general_insights": dim_general_insights[:10],
214
+ "dim_general_insights_covered": dim_general_insights_covered[:10],
215
+ "dim_general_insights_missing": dim_general_insights_missing[:10],
216
+ "qa_samples": samples
217
+ }
218
+ return dim_inputs
219
+
220
+
221
+ # ----------------- OpenAI Chat 调用 -----------------
222
+ def call_openai_chat(model: str,
223
+ system_prompt: str,
224
+ user_payload: Dict[str, Any],
225
+ temperature: float = 0.25,
226
+ max_tokens: int = 2600,
227
+ seed: int = None,
228
+ max_retries: int = 3,
229
+ backoff: float = 1.8) -> Dict[str, Any]:
230
+ """
231
+ 只负责“按给定参数调用一次 Chat Completions”,不判断 PROVIDER。
232
+ 是否调用由上层根据 .env 配置 & 开关决定。
233
+ """
234
+ if not OPENAI_API_KEY:
235
+ raise RuntimeError("OPENAI_API_KEY 缺失,请在 .env 中配置。")
236
+
237
+ url = OPENAI_API_BASE.rstrip("/") + "/chat/completions"
238
+ headers = {
239
+ "Authorization": f"Bearer {OPENAI_API_KEY}",
240
+ "Content-Type": "application/json"
241
+ }
242
+ body: Dict[str, Any] = {
243
+ "model": model,
244
+ "temperature": temperature,
245
+ "max_tokens": max_tokens,
246
+ "messages": [
247
+ {"role": "system", "content": system_prompt},
248
+ {"role": "user", "content": json.dumps(user_payload, ensure_ascii=False)}
249
+ ],
250
+ "response_format": {"type": "json_object"}
251
+ }
252
+ if seed is not None:
253
+ body["seed"] = seed
254
+
255
+ last_err = None
256
+ for attempt in range(1, max_retries + 1):
257
+ try:
258
+ resp = requests.post(
259
+ url,
260
+ headers=headers,
261
+ json=body,
262
+ timeout=(TIMEOUT_CONNECT, TIMEOUT_READ)
263
+ )
264
+ if resp.status_code != 200:
265
+ last_err = f"{resp.status_code} - {resp.text[:400]}"
266
+ time.sleep(backoff ** attempt)
267
+ continue
268
+ data = resp.json()
269
+ content = data["choices"][0]["message"]["content"]
270
+ return json.loads(content)
271
+ except Exception as e:
272
+ last_err = str(e)
273
+ time.sleep(backoff ** attempt)
274
+ raise RuntimeError(f"OpenAI Chat 调用失败(已重试 {max_retries} 次):{last_err}")
275
+
276
+
277
+ # ----------------- Prompt 组装 -----------------
278
+ def build_dim_system_prompt() -> str:
279
+ return (
280
+ "你是一名长期参与复杂项目评审的专家顾问。"
281
+ "系统会给出五个维度(team/objectives/strategy/innovation/feasibility)的:"
282
+ "· 维度中文含义;· 分数强弱/一致性/漂移的‘文字信号’;"
283
+ "· metrics 抽取的 strengths/risks 短语;"
284
+ "· post_processing 聚合的 top_evidence_phrases(只代表证据方向,不是完整证据);"
285
+ "· 维度级与逐问级 general_insights(领域通识建议,仅作对比基准,并不代表本项目已达成);"
286
+ "· dim_general_insights_covered:领域通识中,已在当前问答里部分体现的点;"
287
+ "· dim_general_insights_missing:领域通识中,当前问答几乎未覆盖、但在实际评审中通常被视为重要的信息缺口;"
288
+ "· 该维度的部分问答样本(question/answer/claims/evidence_hints/general_insights)。"
289
+ "你的任务:\n"
290
+ "1)对于每个维度,基于问答内容 + strengths/risks 短语 + 证据方向 + general_insights,"
291
+ " 写出:summary / strengths / concerns / recommendations。\n"
292
+ " - 在 strengths 中,优先结合 dim_general_insights_covered 与具体问答内容,"
293
+ " 明确指出项目已经在哪些方面达到了行业普遍要求。\n"
294
+ " - 在 concerns 和 recommendations 中,必须至少点名 1–2 条 dim_general_insights_missing,"
295
+ " 解释这些点在行业内通常为什么重要,以及本项目目前材料中为何体现不足,并给出补齐建议。\n"
296
+ "2)每条 strengths/concerns 必须是‘有因有果’的一句话:"
297
+ " 可以使用“从……可以看出……从而说明……”“目前材料显示……这一点是亮点/存在不足……”"
298
+ " “与行业中成熟做法相比,……”等多种句式,明确指出‘为什么好/为什么有风险’。"
299
+ " 不要所有句子都以“因为……”“由于……”“从……来看”这类相同短语开头。\n"
300
+ "3)可以引用问答中的关键信息,但不要编造不存在的机构名称、注册号、具体数据。"
301
+ "4)严禁出现任何题号(如 Q1/Q2)、分数字符(0.71、71% 等)或内部指标名"
302
+ " (alignment/coverage/authority/drift/对齐/漂移/覆盖/权威/overall_score/置信度 等)。"
303
+ "5)如果某个维度信息明显不足,可以给出 1–2 条保守结论,例如“当前问答中几乎没有提到……,"
304
+ " 因此该维度信息覆盖不足,需要补充材料”。\n"
305
+ "输出必须是严格 JSON,对每个维度都给出:summary(2–4 句)、strengths(3–5 条)、"
306
+ "concerns(3–5 条)、recommendations(3–5 条)。"
307
+ )
308
+
309
+ def build_dim_user_payload(pid: str,
310
+ dim_inputs: Dict[str, Any]) -> Dict[str, Any]:
311
+ return {
312
+ "pid": pid,
313
+ "task": "dimension_level_expert_opinion",
314
+ "note": "仅输出 dimensions 字段,其余由系统补充。",
315
+ "dimensions": dim_inputs,
316
+ "output_schema_hint": {
317
+ "type": "object",
318
+ "required": ["dimensions"],
319
+ "properties": {
320
+ "dimensions": {
321
+ "type": "object",
322
+ "properties": {
323
+ dim: {
324
+ "type": "object",
325
+ "required": ["summary", "strengths", "concerns", "recommendations"],
326
+ "properties": {
327
+ "summary": {"type": "string"},
328
+ "strengths": {"type": "array", "items": {"type": "string"}},
329
+ "concerns": {"type": "array", "items": {"type": "string"}},
330
+ "recommendations": {"type": "array", "items": {"type": "string"}}
331
+ }
332
+ } for dim in DIM_ORDER
333
+ }
334
+ }
335
+ }
336
+ }
337
+ }
338
+
339
+
340
+ # ----------------- 文本清洗 & 聚合 -----------------
341
+ FORBID_PATTERNS = [
342
+ r"\bQ\d+\b",
343
+ r"\balign(?:ment)?\b",
344
+ r"\bcoverage\b",
345
+ r"\bauth(?:ority)?\b",
346
+ r"\bdrift\b",
347
+ r"对齐", r"漂移", r"覆盖", r"权威",
348
+ r"\boverall[_ ]?score\b",
349
+ r"\bconfidence\b",
350
+ r"\bjaccard\b",
351
+ r"冲突度",
352
+ r"\d+(\.\d+)?\s*%+",
353
+ ]
354
+
355
+
356
+ def clean_text(s: str) -> str:
357
+ s = (s or "").replace("\u0000", "").strip()
358
+ for pat in FORBID_PATTERNS:
359
+ s = re.sub(pat, "", s, flags=re.IGNORECASE)
360
+ s = re.sub(r"\s{2,}", " ", s)
361
+ s = s.replace("()", "").replace("()", "")
362
+ return s.strip()
363
+
364
+
365
+ def clean_list(items: List[str]) -> List[str]:
366
+ out: List[str] = []
367
+ for it in items or []:
368
+ t = clean_text(it)
369
+ if t:
370
+ out.append(t)
371
+ return out
372
+
373
+
374
+ def dedup_soft(items: List[str], thresh: float = 0.85) -> List[str]:
375
+ """非常简单的“字符级 Jaccard”去重,避免几乎一样的句子反复出现。"""
376
+ def to_set(x: str):
377
+ return set((x or "").lower())
378
+
379
+ uniq: List[str] = []
380
+ for s in items or []:
381
+ keep = True
382
+ a = to_set(s)
383
+ for t in uniq:
384
+ b = to_set(t)
385
+ if not a or not b:
386
+ continue
387
+ j = len(a & b) / len(a | b)
388
+ if j >= thresh:
389
+ keep = False
390
+ break
391
+ if keep:
392
+ uniq.append(s)
393
+ return uniq
394
+
395
+
396
+ def _shorten_sentence(text: str, max_len: int = 120) -> str:
397
+ """用于总体 summary 中的分维度 bullet:取第一句或截断到 max_len。"""
398
+ text = (text or "").strip()
399
+ if not text:
400
+ return ""
401
+ # 按中英文句号/问号/感叹号切分,取第一句
402
+ parts = re.split(r"[。!?!?.]", text)
403
+ for p in parts:
404
+ p = p.strip()
405
+ if p:
406
+ text = p
407
+ break
408
+ if len(text) > max_len:
409
+ return text[:max_len].rstrip() + "……"
410
+ return text
411
+
412
+
413
+ # ----------------- 仅本地的维度级专家点评(LLM 回退) -----------------
414
+ def build_local_dim_blocks(metrics: Dict[str, Any],
415
+ final_payload: Dict[str, Any]) -> Dict[str, Any]:
416
+ """
417
+ 当无法调用 LLM 时,基于 metrics + final_payload 直接构造维度点评。
418
+ 逻辑尽量简洁可解释,不引入额外“猜测”。
419
+ """
420
+ dims_metrics = metrics.get("dimensions", {}) or {}
421
+ fp_dims = final_payload.get("dimensions", {}) or {}
422
+
423
+ dim_blocks: Dict[str, Any] = {}
424
+
425
+ for dim in DIM_ORDER:
426
+ m = dims_metrics.get(dim, {}) or {}
427
+ f = fp_dims.get(dim, {}) or {}
428
+
429
+ score = float(m.get("avg", 0.0) or 0.0)
430
+ align = float(m.get("avg_alignment", 0.0) or 0.0)
431
+ drift = float(m.get("avg_drift", 0.0) or 0.0)
432
+ strengths_phr = (m.get("strengths") or [])[:5]
433
+ risks_phr = (m.get("risks") or [])[:5]
434
+ dim_gi = (f.get("general_insights") or [])[:8]
435
+
436
+ label = DIM_LABELS_ZH.get(dim, dim)
437
+
438
+ summary_parts: List[str] = []
439
+ if strengths_phr:
440
+ summary_parts.append(
441
+ f"{label} 维度中,自动评估识别出若干优势点,例如:" +
442
+ ";".join(strengths_phr[:2])
443
+ )
444
+ if risks_phr:
445
+ summary_parts.append(
446
+ "同时也暴露出一些潜在问题或风险,例如:" +
447
+ ";".join(risks_phr[:2])
448
+ )
449
+ if not summary_parts:
450
+ summary_parts.append(
451
+ f"当前问答与自动评估中,关于“{label}”维度的有效信息有限,结论仅供参考,建议补充更详细的事实和量化指标。"
452
+ )
453
+ summary = " ".join(summary_parts)
454
+
455
+ # 优势:优先使用 metrics.strengths;若为空,用少量 general_insights 补位
456
+ strengths_out = strengths_phr[:]
457
+ if not strengths_out and dim_gi:
458
+ strengths_out = [
459
+ f"从行业经验看,本维度若能达到以下实践将显著加分:{dim_gi[0]}"
460
+ ]
461
+
462
+ # 风险:直接用 metrics.risks
463
+ concerns_out = risks_phr[:]
464
+
465
+ # 建议:优先用 general_insights,若为空则给一条通用建议
466
+ recs_out: List[str] = []
467
+ for g in dim_gi:
468
+ recs_out.append(g)
469
+ if risks_phr and not dim_gi:
470
+ recs_out.append(
471
+ "针对上述风险,建议在后续版本中补充更具体的实施计划、里程碑与量化指标,以便评审。"
472
+ )
473
+ if not recs_out:
474
+ recs_out.append(
475
+ f"建议围绕“{label}”维度,系统梳理团队经验、资源保障和实施路径,并结合行业最佳实践补齐信息。"
476
+ )
477
+
478
+ dim_blocks[dim] = {
479
+ "score_echo": score,
480
+ "alignment_echo": align,
481
+ "drift_echo": drift,
482
+ "summary": summary,
483
+ "strengths": strengths_out,
484
+ "concerns": concerns_out,
485
+ "recommendations": recs_out
486
+ }
487
+
488
+ return dim_blocks
489
+
490
+
491
+ # ----------------- 总体意见(本地) -----------------
492
+ def build_overall_from_dims(dim_blocks: Dict[str, Any],
493
+ metrics_overall: Dict[str, Any],
494
+ metrics_dims: Dict[str, Any]) -> Dict[str, Any]:
495
+ """
496
+ 从各维度点评 + 分数,构造总体意见(不再调用 LLM)。
497
+
498
+ 优化点:
499
+ - verdict 阈值略放宽,让中间地带更多落在 HOLD,而不是一刀切 NO-GO;
500
+ - 总体 summary 采用“一段总括 + 分维度 bullet”的结构;
501
+ - 配合后续 markdown 渲染,读起来更像人写的评审意见。
502
+ """
503
+ overall_score = float(metrics_overall.get("overall_score", 0.0) or 0.0)
504
+ overall_conf = float(metrics_overall.get("overall_confidence", 0.0) or 0.0)
505
+
506
+ # 1) 判定 verdict
507
+ def verdict_rule(score: float, conf: float,
508
+ dims: Dict[str, Any]) -> (str, str):
509
+ inv = float(dims.get("innovation", {}).get("avg", 1.0) or 1.0)
510
+ fea = float(dims.get("feasibility", {}).get("avg", 1.0) or 1.0)
511
+
512
+ # ① 明确 GO:得分 + 信心都比较稳
513
+ if score >= 0.62 and conf >= 0.65:
514
+ return "GO", "综合得分与信心度均处于较高区间,关键维度表现扎实,整体风险可控,适合推进。"
515
+
516
+ # ② 明确 NO-GO:整体很低,或关键维度严重偏弱
517
+ if score < 0.40 or inv < 0.30 or fea < 0.30:
518
+ return (
519
+ "NO-GO",
520
+ "总体得分或关键维度(尤其是创新/可行性)处于明显偏低区间,"
521
+ "关键信息缺失或短板较多,目前不宜在本轮直接立项,建议补充材料后再行评估。"
522
+ )
523
+
524
+ # ③ 剩下全部归为 HOLD:有潜力,但证据/信息不够
525
+ return (
526
+ "HOLD",
527
+ "项目处于中间地带,一方面具备一定亮点和潜力,另一方面在若干关键维度上信息仍不充分,"
528
+ "建议在补充必要材料和澄清关键风险后,再做更明确的 go/no-go 决策。"
529
+ )
530
+
531
+ verdict, verdict_reason = verdict_rule(overall_score, overall_conf, metrics_dims)
532
+
533
+ # 2) 总体 summary:一段总括 + 分维度 bullet
534
+ # 2.1 总括句随 verdict 变化
535
+ if verdict == "GO":
536
+ head = (
537
+ "整体来看,该项目在当前轮次的综合表现较为扎实:关键假设相对清晰、实施路径具备一定可操作性,"
538
+ "在风险可控的前提下具备推进价值。"
539
+ )
540
+ elif verdict == "HOLD":
541
+ head = (
542
+ "整体来看,该项目在技术和应用场景上体现出一定潜力,但目前关键信息仍有缺口,"
543
+ "更适合作为“待补充材料后再议”的候选项目,而非直接进入大规模投入阶段。"
544
+ )
545
+ else: # NO-GO
546
+ head = (
547
+ "整体来看,该项目在技术构想或应用方向上虽有亮点,但现有材料无法支撑稳健的风险收益判断,"
548
+ "短板和不确定性占比较高,当前不宜在本轮直接立项。"
549
+ )
550
+
551
+ # 2.2 分维度 bullet:从各维度 summary 中提取第一句精简回顾
552
+ dim_snippets: List[str] = []
553
+ for dim in DIM_ORDER:
554
+ blk = dim_blocks.get(dim, {}) or {}
555
+ dim_sum = (blk.get("summary") or "").strip()
556
+ if not dim_sum:
557
+ continue
558
+ label = DIM_LABELS_ZH.get(dim, dim)
559
+ short = _shorten_sentence(dim_sum, max_len=140)
560
+ if not short:
561
+ continue
562
+ dim_snippets.append(f"- {label}:{short}")
563
+
564
+ lines: List[str] = [head]
565
+ if dim_snippets:
566
+ lines.append("分维度来看:")
567
+ lines.extend(dim_snippets)
568
+
569
+ summary_text = "\n".join(lines)
570
+
571
+ # 3) 从各维度 strengths/concerns 中抽取总体 key_strengths/key_risks
572
+ key_strengths: List[str] = []
573
+ key_risks: List[str] = []
574
+ for dim in DIM_ORDER:
575
+ blk = dim_blocks.get(dim, {}) or {}
576
+ label = DIM_LABELS_ZH.get(dim, dim)
577
+ for s in (blk.get("strengths") or [])[:2]:
578
+ key_strengths.append(f"【{label}】{s}")
579
+ for r in (blk.get("concerns") or [])[:2]:
580
+ key_risks.append(f"【{label}】{r}")
581
+
582
+ key_strengths = dedup_soft(clean_list(key_strengths))[:6]
583
+ key_risks = dedup_soft(clean_list(key_risks))[:6]
584
+
585
+ # 4) 总体 recommendations:从各维度 recommendations 抽样
586
+ recs: List[str] = []
587
+ for dim in DIM_ORDER:
588
+ blk = dim_blocks.get(dim, {}) or {}
589
+ label = DIM_LABELS_ZH.get(dim, dim)
590
+ for r in (blk.get("recommendations") or [])[:2]:
591
+ recs.append(f"【{label}】{r}")
592
+ recs = dedup_soft(clean_list(recs))[:8]
593
+
594
+ return {
595
+ "summary": summary_text,
596
+ "overall_score_echo": overall_score,
597
+ "confidence_echo": overall_conf,
598
+ "key_strengths": key_strengths,
599
+ "key_risks": key_risks,
600
+ "recommendations": recs,
601
+ "verdict": verdict,
602
+ "basis": [
603
+ f"结论依据:{verdict_reason}",
604
+ "结论完全基于已选中问答结果与自动评分信号,未引入外部资料。"
605
+ ]
606
+ }
607
+
608
+
609
+ # ----------------- Markdown 渲染 -----------------
610
+ def _bar(v: float, n: int = 20) -> str:
611
+ try:
612
+ v = float(v)
613
+ except Exception:
614
+ v = 0.0
615
+ v = max(0.0, min(1.0, v))
616
+ k = int(round(v * n))
617
+ return "█" * k + "░" * (n - k)
618
+
619
+
620
+ def render_markdown(opinion: Dict[str, Any]) -> str:
621
+ meta = opinion.get("meta", {}) or {}
622
+ overall = opinion.get("overall_opinion", {}) or {}
623
+ dims = opinion.get("dimensions", {}) or {}
624
+ scoring = opinion.get("scoring_explainer", {}) or {}
625
+ metrics_path = (meta.get("sources") or {}).get("metrics_path", "")
626
+
627
+ lines: List[str] = []
628
+ lines.append(f"# AI 专家评审 · {meta.get('pid', '')}")
629
+ lines.append("")
630
+ lines.append(f"- 生成时间:{meta.get('generated_at', '')}")
631
+ lines.append(f"- 模式:{meta.get('mode', '')}")
632
+ lines.append(f"- 模型/引擎:{meta.get('model', '')}(provider={meta.get('provider', '')})")
633
+ lines.append("")
634
+
635
+ # 总体
636
+ lines.append("## 总体意见")
637
+ lines.append(f"- 综合评分(回显):{overall.get('overall_score_echo', 0.0):.3f} {_bar(overall.get('overall_score_echo', 0.0))}")
638
+ lines.append(f"- 综合信心度(回显):{overall.get('confidence_echo', 0.0):.3f} {_bar(overall.get('confidence_echo', 0.0))}")
639
+ lines.append("")
640
+ if overall.get("summary"):
641
+ lines.append(overall["summary"])
642
+ lines.append("")
643
+ if overall.get("key_strengths"):
644
+ lines.append("**项目优势**")
645
+ for s in overall["key_strengths"]:
646
+ lines.append(f"- {s}")
647
+ lines.append("")
648
+ if overall.get("key_risks"):
649
+ lines.append("**项目不足/潜在风险**")
650
+ for r in overall["key_risks"]:
651
+ lines.append(f"- {r}")
652
+ lines.append("")
653
+ if overall.get("recommendations"):
654
+ lines.append("**总体建议**")
655
+ for r in overall["recommendations"]:
656
+ lines.append(f"- {r}")
657
+ lines.append("")
658
+ if overall.get("verdict"):
659
+ lines.append(f"**总体结论(verdict)**:{overall['verdict']}")
660
+ lines.append("")
661
+ if overall.get("basis"):
662
+ lines.append("**结论依据(系统自动生成)**")
663
+ for b in overall["basis"]:
664
+ lines.append(f"- {b}")
665
+ lines.append("")
666
+
667
+ # 维度表
668
+ lines.append("## 各维度评分一览")
669
+ lines.append("")
670
+ lines.append("| 维度 | 分数 |")
671
+ lines.append("|---|---:|")
672
+ for dim in DIM_ORDER:
673
+ blk = dims.get(dim, {}) or {}
674
+ lines.append(f"| {dim} | {blk.get('score_echo', 0.0):.3f} |")
675
+ lines.append("")
676
+
677
+ # 分维度详情
678
+ lines.append("## 分维度专家点评")
679
+ lines.append("")
680
+ for dim in DIM_ORDER:
681
+ label = DIM_LABELS_ZH.get(dim, dim)
682
+ blk = dims.get(dim, {}) or {}
683
+ lines.append(f"### {label}({dim})")
684
+ lines.append(f"- 评分回显:{blk.get('score_echo', 0.0):.3f} {_bar(blk.get('score_echo', 0.0))}")
685
+ lines.append("")
686
+ if blk.get("summary"):
687
+ lines.append(blk["summary"])
688
+ lines.append("")
689
+ if blk.get("strengths"):
690
+ lines.append("**优势**")
691
+ for s in blk["strengths"]:
692
+ lines.append(f"- {s}")
693
+ lines.append("")
694
+ if blk.get("concerns"):
695
+ lines.append("**问题/风险**")
696
+ for r in blk["concerns"]:
697
+ lines.append(f"- {r}")
698
+ lines.append("")
699
+ if blk.get("recommendations"):
700
+ lines.append("**改进建议**")
701
+ for r in blk["recommendations"]:
702
+ lines.append(f"- {r}")
703
+ lines.append("")
704
+
705
+ # 简单回显评分配置(方便审计)
706
+ if scoring:
707
+ lines.append("## 评分规则回显(来自 post_processing 配置)")
708
+ lines.append(f"- 一致性权重 consistency_weight:{scoring.get('consistency_weight', 0.0):.2f}")
709
+ if scoring.get("dimension_weight"):
710
+ dw = scoring["dimension_weight"]
711
+ order_str = ", ".join([f"{d}:{dw.get(d, 0.0):.2f}" for d in DIM_ORDER if d in dw])
712
+ lines.append(f"- 维度权重:{order_str}")
713
+ lines.append("")
714
+
715
+ if metrics_path:
716
+ lines.append("## 溯源")
717
+ lines.append(f"- metrics.json:{metrics_path}")
718
+ lines.append("")
719
+
720
+ return "\n".join(lines)
721
+
722
+
723
+ # ----------------- 主流程 -----------------
724
+ def main():
725
+ ap = argparse.ArgumentParser(description="Generate AI expert opinion (dimension-first, QA-grounded).")
726
+ ap.add_argument("--pid", type=str, default="", help="提案 ID;缺省则自动选择最新项目")
727
+ ap.add_argument("--model", type=str, default=OPENAI_MODEL, help="OpenAI 模型名")
728
+ ap.add_argument("--dry_run", action="store_true", help="仅生成 prompt,不调用 LLM")
729
+ ap.add_argument("--no_markdown", action="store_true", help="不输出 Markdown,仅 JSON")
730
+ ap.add_argument("--force_local", action="store_true", help="强制使用本地规则版,不调用 LLM(用于调试)")
731
+ args = ap.parse_args()
732
+
733
+ pid = args.pid.strip() or detect_latest_pid()
734
+ if not pid:
735
+ raise RuntimeError("未检测到可用项目(refined_answers 下无 postproc/metrics.json + final_payload.json)。")
736
+
737
+ postproc_dir = REFINED_ROOT / pid / "postproc"
738
+ metrics_path = postproc_dir / "metrics.json"
739
+ payload_path = postproc_dir / "final_payload.json"
740
+ if not metrics_path.exists():
741
+ raise FileNotFoundError(f"未找到 metrics.json:{metrics_path}")
742
+ if not payload_path.exists():
743
+ raise FileNotFoundError(f"未找到 final_payload.json:{payload_path}")
744
+
745
+ metrics = read_json(metrics_path)
746
+ final_payload = read_json(payload_path)
747
+
748
+ dim_inputs = build_dim_inputs(metrics, final_payload)
749
+
750
+ out_dir = EXPERT_DIR / pid
751
+ out_dir.mkdir(parents=True, exist_ok=True)
752
+ prompt_path = out_dir / "ai_expert_opinion.prompt.json"
753
+ json_path = out_dir / "ai_expert_opinion.json"
754
+ md_path = out_dir / "ai_expert_opinion.md"
755
+
756
+ # 先保存 prompt(即便是本地版,也方便调试看看输入长什么样)
757
+ system_prompt = build_dim_system_prompt()
758
+ user_payload = build_dim_user_payload(pid, dim_inputs)
759
+ write_json(prompt_path, {
760
+ "system": system_prompt,
761
+ "user": user_payload
762
+ })
763
+
764
+ # =========== 维度级点评:优先 LLM,失败则回退本地 ===========
765
+ dim_op_raw: Dict[str, Any] = {}
766
+ used_model = ""
767
+ used_mode = ""
768
+
769
+ use_llm = (not args.force_local) and (PROVIDER == "openai") and bool(OPENAI_API_KEY)
770
+
771
+ if args.dry_run:
772
+ print(f"📝 已导出 prompt -> {prompt_path}")
773
+ return
774
+
775
+ if use_llm:
776
+ try:
777
+ resp = call_openai_chat(
778
+ model=args.model,
779
+ system_prompt=system_prompt,
780
+ user_payload=user_payload,
781
+ temperature=0.25,
782
+ max_tokens=2600,
783
+ seed=None,
784
+ )
785
+ if "dimensions" not in resp:
786
+ raise RuntimeError("LLM 返回 JSON 中缺少 'dimensions' 字段")
787
+ dim_op_raw = resp["dimensions"] or {}
788
+ used_model = args.model
789
+ used_mode = "llm"
790
+ except Exception as e:
791
+ print(f"⚠️ LLM 生成维度点评失败,将启用纯本地规则回退:{e}")
792
+ dim_op_raw = {}
793
+ used_model = "local_rules"
794
+ used_mode = "local_fallback"
795
+ else:
796
+ used_model = "local_rules"
797
+ used_mode = "local_forced"
798
+
799
+ if not dim_op_raw:
800
+ # 纯本地规则版
801
+ dim_op_raw = build_local_dim_blocks(metrics, final_payload)
802
+
803
+ # 兜底:保证所有维度都有字段 & 文本清洗/去重 + 收紧条数
804
+ cleaned_dims: Dict[str, Any] = {}
805
+ metrics_dims = metrics.get("dimensions", {}) or {}
806
+ for dim in DIM_ORDER:
807
+ blk = dim_op_raw.get(dim, {}) or {}
808
+ m = metrics_dims.get(dim, {}) or {}
809
+
810
+ strengths = dedup_soft(clean_list(blk.get("strengths") or []))[:3]
811
+ concerns = dedup_soft(clean_list(blk.get("concerns") or []))[:3]
812
+ recs = dedup_soft(clean_list(blk.get("recommendations") or []))[:4]
813
+ summary = clean_text(blk.get("summary") or "")
814
+
815
+ # 如果维度几乎没有信息,补一条兜底 summary
816
+ if not summary and not strengths and not concerns:
817
+ label = DIM_LABELS_ZH.get(dim, dim)
818
+ summary = (
819
+ f"当前关于“{label}”维度的有效问答与证据信号非常有限,结论不稳定,"
820
+ f"建议项目方在后续版本中补充该维度的核心事实、量化指标与实施细节。"
821
+ )
822
+
823
+ cleaned_dims[dim] = {
824
+ "score_echo": float(m.get("avg", 0.0) or 0.0),
825
+ "alignment_echo": float(m.get("avg_alignment", 0.0) or 0.0),
826
+ "drift_echo": float(m.get("avg_drift", 0.0) or 0.0),
827
+ "summary": summary,
828
+ "strengths": strengths,
829
+ "concerns": concerns,
830
+ "recommendations": recs
831
+ }
832
+
833
+ overall_block = build_overall_from_dims(
834
+ dim_blocks=cleaned_dims,
835
+ metrics_overall=metrics.get("overall", {}) or {},
836
+ metrics_dims=metrics_dims
837
+ )
838
+
839
+ # 拼接最终 JSON
840
+ opinion: Dict[str, Any] = {
841
+ "meta": {
842
+ "pid": pid,
843
+ "generated_at": now_str(),
844
+ "model": used_model,
845
+ "mode": used_mode,
846
+ "provider": PROVIDER,
847
+ "sources": {
848
+ "metrics_path": str(metrics_path),
849
+ "final_payload_path": str(payload_path)
850
+ }
851
+ },
852
+ "overall_opinion": overall_block,
853
+ "dimensions": cleaned_dims,
854
+ "scoring_explainer": {
855
+ # 只回显最关键的几项,便于报告中解释“分是怎么算出来的”
856
+ "consistency_weight": float((metrics.get("config_used") or {}).get("consistency_weight", 0.20) or 0.20),
857
+ "dimension_weight": {
858
+ k: float(v) for k, v in ((metrics.get("config_used") or {}).get("dimension_weight") or {}).items()
859
+ }
860
+ }
861
+ }
862
+
863
+ write_json(json_path, opinion)
864
+ if not args.no_markdown:
865
+ md_text = render_markdown(opinion)
866
+ md_path.write_text(md_text, encoding="utf-8")
867
+
868
+ print(f"✅ ai_expert_opinion.json -> {json_path}")
869
+ if not args.no_markdown:
870
+ print(f"✅ ai_expert_opinion.md -> {md_path}")
871
+ print(f"🎯 专家评审生成完成({used_mode} 模式)。")
872
+
873
+
874
+ if __name__ == "__main__":
875
+ main()
src/tools/build_dimensions_from_facts.py ADDED
@@ -0,0 +1,473 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ Stage 2 · 维度构建器(build_dimensions_from_facts.py)
4
+ ----------------------------------------------------
5
+ 输入:
6
+ - src/data/extracted/<proposal_id>/raw_facts.jsonl (Stage 1 输出)
7
+
8
+ 输出:
9
+ - src/data/extracted/<proposal_id>/dimensions_v2.json (新的五维度文件)
10
+ - src/data/extracted/<proposal_id>/dimension_facts.json (按维度分组的 facts,方便后续调试与复用)
11
+ - src/data/parsed/parsed_dimensions.clean.llm.json (供 llm_answering 使用的全局 parsed 文件)
12
+
13
+ 核心职责:
14
+ - 按 dimensions 标签把 facts 分桶到 team/objectives/strategy/innovation/feasibility
15
+ - 对每个维度单独调用一次 LLM,只基于该维度的事实生成:
16
+ summary / key_points / risks / mitigations
17
+ - 严禁凭空造事实,所有内容必须能在事实列表中找到“影子”
18
+ - 尽量覆盖该维度下出现过的不同 type(team_member/pipeline/market/risk/...)
19
+ """
20
+
21
+ import os
22
+ import json
23
+ import argparse
24
+ from pathlib import Path
25
+ from typing import List, Dict, Any
26
+
27
+ from dotenv import load_dotenv
28
+ from openai import OpenAI
29
+
30
+ load_dotenv()
31
+
32
+ # 假设本文件路径:<project_root>/src/tools/build_dimensions_from_facts.py
33
+ BASE_DIR = Path(__file__).resolve().parents[2]
34
+ EXTRACTED_DIR = BASE_DIR / "src" / "data" / "extracted"
35
+ PARSED_DIR = BASE_DIR / "src" / "data" / "parsed" # 与 llm_answering 对齐
36
+
37
+ OPENAI_MODEL = os.getenv("OPENAI_MODEL", "gpt-4o-mini")
38
+
39
+ DIMENSION_NAMES = ["team", "objectives", "strategy", "innovation", "feasibility"]
40
+
41
+ # 全局 client 复用
42
+ client = OpenAI()
43
+
44
+ # 注意:这里用 {dimension_name} 标记占位,其它所有 { } 都是字面量 JSON 示例
45
+ # 后面用 .replace("{dimension_name}", xxx) 而不是 .format()
46
+
47
+ DIMENSION_PROMPT_TEMPLATE = """
48
+ 你是一个严谨的项目评审助手,现在要基于【已经抽取好的事实列表】为某一个维度生成结构化摘要。
49
+
50
+ 硬性约束:
51
+ 1)只能使用我给你的 facts,不得凭空添加新的机构、人名、对象、方法、数字、时间、结果或结论。
52
+ 2)允许压缩与合并,但每一个 key_point、risk、mitigation 都必须能在 facts 中找到依据。
53
+ 3)只能总结【当前维度】的内容;若需要提及其他维度的信息,也只能在与当前维度强相关时简短引用。
54
+ 4)不要写行业常识、通用背景或空泛判断,只写从 facts 能支持的内容。
55
+
56
+ 【当前维度】:{dimension_name}
57
+
58
+ 【维度含义提醒】
59
+ - team:团队、机构、角色分工、协作、治理
60
+ - objectives:问题背景、目标、范围、里程碑、交付物、评价指标
61
+ - strategy:方法、技术路线、实施流程、验证设计、外部协作路径
62
+ - innovation:新颖性、差异化、独特资源、已有证据、知识产权
63
+ - feasibility:资源、预算、时间安排、风险、限制、应对、执行条件
64
+
65
+ 输入 payload 结构:
66
+ payload = {
67
+ "all_facts": [{"text": "...", "dimensions": ["team"], "type": "team_member", "meta": {...}}, ...],
68
+ "risk_facts": [...],
69
+ "mitigation_facts": [...]
70
+ }
71
+
72
+ 请输出 JSON:
73
+ {
74
+ "summary": "2-4 句,概括该维度当前情况",
75
+ "key_points": ["3-10 条,不重复、尽量覆盖不同主题"],
76
+ "risks": ["基于 risk_facts 或相关 facts 总结;若信息不足,可明确写‘该维度风险信息较少/未详细说明’"],
77
+ "mitigations": ["基于 mitigation_facts 或相关 facts 总结;若信息不足,可明确写‘该维度风险应对措施未详细说明’"]
78
+ }
79
+
80
+ 补充要求:
81
+ - facts 足够多时,优先覆盖不同 type,而不是重复同一类信息。
82
+ - 不要输出任何 JSON 以外的文字。
83
+ """
84
+
85
+
86
+ def load_raw_facts(proposal_id: str) -> List[Dict[str, Any]]:
87
+ path = EXTRACTED_DIR / proposal_id / "raw_facts.jsonl"
88
+ if not path.exists():
89
+ raise FileNotFoundError(f"raw_facts.jsonl 不存在,请先运行 extract_facts_by_chunk.py: {path}")
90
+
91
+ facts: List[Dict[str, Any]] = []
92
+ with path.open("r", encoding="utf-8") as f:
93
+ for line in f:
94
+ line = line.strip()
95
+ if not line:
96
+ continue
97
+ try:
98
+ obj = json.loads(line)
99
+ except json.JSONDecodeError:
100
+ continue
101
+ if isinstance(obj, dict):
102
+ facts.append(obj)
103
+ print(f"[INFO] 读取 facts 数量: {len(facts)} 来自 {path}")
104
+ return facts
105
+
106
+
107
+ def group_facts_by_dimension(facts: List[Dict[str, Any]]) -> Dict[str, List[Dict[str, Any]]]:
108
+ """
109
+ 按维度标签把 facts 分桶到五个维度。
110
+ 一条 fact 可能属于多个维度,会出现在多个桶里。
111
+ """
112
+ grouped: Dict[str, List[Dict[str, Any]]] = {dim: [] for dim in DIMENSION_NAMES}
113
+
114
+ for fact in facts:
115
+ dims = fact.get("dimensions", [])
116
+ if not isinstance(dims, list):
117
+ continue
118
+ for dim in dims:
119
+ if dim in grouped:
120
+ grouped[dim].append(fact)
121
+
122
+ for dim in DIMENSION_NAMES:
123
+ print(f"[INFO] 维度 {dim} 相关事实数: {len(grouped[dim])}")
124
+ return grouped
125
+
126
+
127
+ def sort_facts_for_dimension(dimension_name: str, facts: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
128
+ """
129
+ 按维度定义 type 优先级排序,保证更关键信息排在前面。
130
+ 这是纯通用规则,不依赖具体提案内容。
131
+ """
132
+ priority_map = {
133
+ "team": [
134
+ # 先看具体成员履历,再看组织结构和协作模式
135
+ "team_member", "org_structure", "collaboration", "resource", "other"
136
+ ],
137
+ "objectives": [
138
+ # 目标、里程碑、验证设计与交付信息优先
139
+ "milestone", "pipeline", "clinical_design", "product", "market", "other"
140
+ ],
141
+ "strategy": [
142
+ # 方法路线、外部协作、应用环境与制度流程都属于 strategy 视角
143
+ "tech_route", "product", "collaboration", "market",
144
+ "funding_source", "regulatory", "other"
145
+ ],
146
+ "innovation": [
147
+ "ip_asset", "evidence", "ai_model", "tech_route", "product", "other"
148
+ ],
149
+ "feasibility": [
150
+ "resource", "budget_item", "funding_source",
151
+ "risk", "mitigation", "regulatory", "other"
152
+ ],
153
+ }
154
+ order = priority_map.get(dimension_name, ["other"])
155
+
156
+ def type_rank(t: str) -> int:
157
+ return order.index(t) if t in order else len(order)
158
+
159
+ return sorted(
160
+ facts,
161
+ key=lambda f: type_rank(f.get("type", "other"))
162
+ )
163
+
164
+
165
+ def truncate_facts_for_prompt(facts: List[Dict[str, Any]], max_chars: int = 10000) -> List[Dict[str, Any]]:
166
+ """
167
+ 为了防止单次 prompt 爆 context,对 facts 做一个简单的字符长度截断。
168
+ 按顺序累加 text,超过 max_chars 就停(meta 仍然保留)。
169
+ """
170
+ kept: List[Dict[str, Any]] = []
171
+ total = 0
172
+ for fact in facts:
173
+ t = fact.get("text", "") or ""
174
+ t_len = len(t)
175
+ if total + t_len > max_chars and kept:
176
+ break
177
+ kept.append(fact)
178
+ total += t_len
179
+ return kept
180
+
181
+
182
+ # ===== 辅助:基于文本再兜底识别 risk / mitigation =====
183
+
184
+ _RISK_CN = ["风险", "挑战", "瓶颈", "不确定性", "不足", "局限", "缺陷", "障碍", "难点"]
185
+ _RISK_EN = ["risk", "risks", "challenge", "challenges", "bottleneck", "bottlenecks",
186
+ "uncertainty", "limitation", "limitations", "weakness", "weaknesses",
187
+ "barrier", "barriers", "issue", "issues", "difficulty", "difficulties"]
188
+
189
+ _MITIG_CN = ["应对", "缓解", "降低", "减少", "解决", "克服", "应对措施", "改进", "优化", "管控"]
190
+ _MITIG_EN = ["mitigation", "mitigate", "mitigating", "address", "addresses", "addressing",
191
+ "solve", "solves", "solving", "overcome", "overcoming",
192
+ "reduce", "reduces", "reducing", "decrease", "decreases", "decreasing",
193
+ "improve", "improves", "improving", "optimize", "optimizing", "optimization"]
194
+
195
+
196
+ def _looks_like_risk(text: str) -> bool:
197
+ if not text:
198
+ return False
199
+ t = text.lower()
200
+ if any(k in text for k in _RISK_CN):
201
+ return True
202
+ if any(k in t for k in _RISK_EN):
203
+ return True
204
+ return False
205
+
206
+ def reclassify_risk_mitigation_global(facts: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
207
+ """
208
+ 在进入维度构建之前,对所有 facts 做一次全局的 risk/mitigation 纠偏:
209
+ - 如果 type 不是 risk/mitigation,但文本明显是风险/挑战,就改成 risk
210
+ - 如果 type 不是 risk/mitigation,但文本明显是应对/缓解方案,就改成 mitigation
211
+ - 如果本身标成 risk/mitigation 但文本看起来不像风险/应对,则降级为 other
212
+ (必要时在 risk 和 mitigation 之间互换)
213
+ """
214
+ new_facts: List[Dict[str, Any]] = []
215
+
216
+ for f in facts:
217
+ t = f.get("type", "other") or "other"
218
+ txt = f.get("text", "") or ""
219
+
220
+ # 优先纠错:如果已经标成 risk/mitigation,但文本不符合,就降级/互换
221
+ if t == "risk":
222
+ if not _looks_like_risk(txt):
223
+ # 文本更像应对措施,就改成 mitigation;否则降级成 other
224
+ if _looks_like_mitigation(txt):
225
+ t = "mitigation"
226
+ else:
227
+ t = "other"
228
+ elif t == "mitigation":
229
+ if not _looks_like_mitigation(txt):
230
+ # 文本更像风险描述,就改成 risk;否则降级成 other
231
+ if _looks_like_risk(txt):
232
+ t = "risk"
233
+ else:
234
+ t = "other"
235
+ else:
236
+ # 如果原始类型既不是 risk 也不是 mitigation,就尝试“升格”
237
+ if _looks_like_risk(txt):
238
+ t = "risk"
239
+ elif _looks_like_mitigation(txt):
240
+ t = "mitigation"
241
+
242
+ f["type"] = t
243
+ new_facts.append(f)
244
+
245
+ return new_facts
246
+
247
+ def _looks_like_mitigation(text: str) -> bool:
248
+ if not text:
249
+ return False
250
+ t = text.lower()
251
+ if any(k in text for k in _MITIG_CN):
252
+ return True
253
+ if any(k in t for k in _MITIG_EN):
254
+ return True
255
+ return False
256
+
257
+
258
+ def call_llm_for_dimension(dimension_name: str, facts: List[Dict[str, Any]]) -> Dict[str, Any]:
259
+ """
260
+ 对单一维度调用一次 LLM,生成 summary/key_points/risks/mitigations。
261
+ 强约束:只能基于 facts,不得脑补。
262
+ """
263
+
264
+ # 1) 先按 type 排序,再截断,确保更重要的信息优先被看到
265
+ sorted_facts = sort_facts_for_dimension(dimension_name, facts)
266
+ all_facts_for_prompt = truncate_facts_for_prompt(sorted_facts, max_chars=10000)
267
+
268
+ # 2) 识别风险 / 对策 facts:不仅看 type,还看文本关键词
269
+ risk_facts: List[Dict[str, Any]] = []
270
+ mitigation_facts: List[Dict[str, Any]] = []
271
+
272
+ for f in all_facts_for_prompt:
273
+ t = f.get("type", "")
274
+ txt = f.get("text", "") or ""
275
+
276
+ # 明确标成 risk 的,或者文本看起来是在描述风险/挑战,都归入
277
+ if t == "risk" or _looks_like_risk(txt):
278
+ risk_facts.append(f)
279
+
280
+ # 明确标成 mitigation 的,或者文本看起来是在描述应对/解决方案,也归入
281
+ if t == "mitigation" or _looks_like_mitigation(txt):
282
+ mitigation_facts.append(f)
283
+
284
+ payload = {
285
+ "all_facts": all_facts_for_prompt,
286
+ "risk_facts": risk_facts,
287
+ "mitigation_facts": mitigation_facts,
288
+ }
289
+ facts_json_str = json.dumps(payload, ensure_ascii=False, indent=2)
290
+
291
+ # 用 replace,而不是 format,避免 JSON 里的 { } 被当成占位符
292
+ prompt = DIMENSION_PROMPT_TEMPLATE.replace("{dimension_name}", dimension_name)
293
+
294
+ messages = [
295
+ {
296
+ "role": "system",
297
+ "content": "你是一个严谨的项目评审助手,只能基于给定的事实列表进行总结,不得编造。",
298
+ },
299
+ {
300
+ "role": "user",
301
+ "content": prompt + "\n\n=== payload 开始 ===\n" + facts_json_str,
302
+ },
303
+ ]
304
+
305
+ resp = client.chat.completions.create(
306
+ model=OPENAI_MODEL,
307
+ messages=messages,
308
+ response_format={"type": "json_object"},
309
+ temperature=0.0,
310
+ max_tokens=2600,
311
+ )
312
+ raw = resp.choices[0].message.content
313
+
314
+ try:
315
+ data = json.loads(raw)
316
+ except json.JSONDecodeError as e:
317
+ print(f"[WARN] 维度 {dimension_name} JSON 解析失败,原始内容如下:")
318
+ print(raw)
319
+ raise e
320
+
321
+ # 兜底:保证四个字段存在
322
+ summary = data.get("summary", "")
323
+ if not isinstance(summary, str):
324
+ summary = ""
325
+ key_points = data.get("key_points", [])
326
+ if not isinstance(key_points, list):
327
+ key_points = []
328
+ risks = data.get("risks", [])
329
+ if not isinstance(risks, list):
330
+ risks = []
331
+ mitigations = data.get("mitigations", [])
332
+ if not isinstance(mitigations, list):
333
+ mitigations = []
334
+
335
+ # 轻量兜底:facts 足够多但 key_points 太少,打日志提醒(先不强制重试)
336
+ fact_count = len(all_facts_for_prompt)
337
+ if fact_count >= 20 and len(key_points) < 6:
338
+ print(
339
+ f"[WARN] 维度 {dimension_name}: all_facts={fact_count} 但 key_points 只有 {len(key_points)} 条,"
340
+ f"如有需要可以在此处加重试/补点逻辑。"
341
+ )
342
+
343
+ return {
344
+ "summary": summary.strip(),
345
+ "key_points": [str(x).strip() for x in key_points if str(x).strip()],
346
+ "risks": [str(x).strip() for x in risks if str(x).strip()],
347
+ "mitigations": [str(x).strip() for x in mitigations if str(x).strip()],
348
+ }
349
+
350
+
351
+ def run_build(proposal_id: str):
352
+ facts = load_raw_facts(proposal_id)
353
+ # 全局先做一次 risk/mitigation 纠偏
354
+ facts = reclassify_risk_mitigation_global(facts)
355
+ grouped = group_facts_by_dimension(facts)
356
+
357
+ # 额外输出一份 dimension_facts.json,方便后续人工检查和调试
358
+ out_dir = EXTRACTED_DIR / proposal_id
359
+ out_dir.mkdir(parents=True, exist_ok=True)
360
+ dim_facts_path = out_dir / "dimension_facts.json"
361
+ dim_facts_path.write_text(
362
+ json.dumps(grouped, ensure_ascii=False, indent=2),
363
+ encoding="utf-8",
364
+ )
365
+ print(f"[OK] 已写出按维度分组的 facts: {dim_facts_path}")
366
+
367
+ dimensions_result: Dict[str, Dict[str, Any]] = {}
368
+
369
+ for dim in DIMENSION_NAMES:
370
+ dim_facts = grouped.get(dim, [])
371
+ print(f"\n[INFO] 开始构建维度 {dim} ...")
372
+
373
+ if not dim_facts:
374
+ # 没有任何事实,给一个空壳(显式指出信息缺失)
375
+ dimensions_result[dim] = {
376
+ "summary": f"提案文本中关于 {dim} 维度的明确信息较少,无法做出详细总结。",
377
+ "key_points": [],
378
+ "risks": [f"提案中关于 {dim} 维度的细节信息较少,可能影响评估。"],
379
+ "mitigations": ["提案未具体说明如何补充或缓解该维度信息不足的问题。"],
380
+ }
381
+ print(f"[INFO] 维度 {dim} 无事实,写入占位结果。")
382
+ continue
383
+
384
+ data = call_llm_for_dimension(dim, dim_facts)
385
+
386
+ # ==== 风险覆盖度标记 ====
387
+ risk_count = len(data.get("risks", []) or [])
388
+ if risk_count == 0:
389
+ level = "low"
390
+ reason = "提案文本中几乎没有显式描述该维度相关的风险,系统无法进行充分的风险细化。"
391
+ elif risk_count <= 2:
392
+ level = "medium"
393
+ reason = "该维度仅有少量风险相关描述,风险分析的粒度有限。"
394
+ else:
395
+ level = "high"
396
+ reason = "该维度在提案中有较为丰富的风险相关描述,可以进行较细致的风险分析。"
397
+
398
+ data["risk_coverage"] = {
399
+ "level": level,
400
+ "reason": reason,
401
+ "risk_count": risk_count,
402
+ }
403
+
404
+ dimensions_result[dim] = data
405
+
406
+ print(
407
+ f"[INFO] 维度 {dim} 完成:summary_len={len(data['summary'])}, "
408
+ f"key_points={len(data['key_points'])}, risks={len(data['risks'])}, "
409
+ f"mitigations={len(data['mitigations'])}, "
410
+ f"risk_coverage={level}"
411
+ )
412
+
413
+ # 1) 写入 per-proposal 维度文件
414
+ out_path = out_dir / "dimensions_v2.json"
415
+ out_path.write_text(
416
+ json.dumps(dimensions_result, ensure_ascii=False, indent=2),
417
+ encoding="utf-8",
418
+ )
419
+ print(f"\n[OK] 新版五维度文件已生成: {out_path}")
420
+
421
+ # 2) 同时写一份全局 parsed 文件给 llm_answering 用
422
+ PARSED_DIR.mkdir(parents=True, exist_ok=True)
423
+ parsed_path = PARSED_DIR / "parsed_dimensions.clean.llm.json"
424
+
425
+ parsed_obj = {
426
+ dim: {
427
+ "summary": data.get("summary", ""),
428
+ "key_points": data.get("key_points", []),
429
+ "risks": data.get("risks", []),
430
+ "mitigations": data.get("mitigations", []),
431
+ }
432
+ for dim, data in dimensions_result.items()
433
+ }
434
+
435
+ parsed_path.write_text(
436
+ json.dumps(parsed_obj, ensure_ascii=False, indent=2),
437
+ encoding="utf-8",
438
+ )
439
+ print(f"[OK] 已写出 parsed 维度文件供 llm_answering 使用: {parsed_path}")
440
+
441
+
442
+ def main():
443
+ parser = argparse.ArgumentParser(
444
+ description="Stage 2: 基于 raw_facts.jsonl 构建五个维度的 dimensions_v2.json"
445
+ )
446
+ parser.add_argument(
447
+ "--proposal_id",
448
+ required=False,
449
+ help="提案 ID(对应 src/data/extracted/<proposal_id>)",
450
+ )
451
+ args = parser.parse_args()
452
+
453
+ if args.proposal_id:
454
+ pid = args.proposal_id
455
+ else:
456
+ # 默认用 extracted 里最新的一个子目录
457
+ if not EXTRACTED_DIR.exists():
458
+ raise FileNotFoundError(f"未找到 extracted 目录: {EXTRACTED_DIR}")
459
+ candidates = [
460
+ (d.stat().st_mtime, d.name)
461
+ for d in EXTRACTED_DIR.iterdir()
462
+ if d.is_dir()
463
+ ]
464
+ if not candidates:
465
+ raise FileNotFoundError(f"extracted 目录下没有任何子目录: {EXTRACTED_DIR}")
466
+ pid = max(candidates, key=lambda x: x[0])[1]
467
+ print(f"[INFO] [auto] 选中最新提案 ID: {pid}")
468
+
469
+ run_build(pid)
470
+
471
+
472
+ if __name__ == "__main__":
473
+ main()
src/tools/build_vector_db.py ADDED
@@ -0,0 +1,218 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ 阶段 5:构建向量化知识库(Knowledge Base Builder, v2025.12R · Fusion-Sync Edition)
4
+ 增强内容:
5
+ ✅ 同步兼容 fusion_search v2025.12R 输出结构(含 confidence_distribution / embedding_model)
6
+ ✅ 动态过滤(短文本 + 低置信文本)
7
+ ✅ 分层统计:高/中/低置信度文档数量与均值
8
+ ✅ 自动 GPU 检测 + 自适应 batch_size
9
+ ✅ 健康检查:计算入库率与记录分布
10
+ ✅ 输出增强索引(含 confidence_tiers 与全局平均置信度)
11
+ """
12
+
13
+ import os
14
+ import sys
15
+ import json
16
+ import numpy as np
17
+ from datetime import datetime
18
+ from pathlib import Path
19
+ from typing import List, Dict, Any
20
+ from sentence_transformers import SentenceTransformer
21
+ import chromadb
22
+
23
+ # ========= 路径与环境 =========
24
+ CURRENT_DIR = Path(__file__).resolve().parent
25
+ SRC_ROOT = CURRENT_DIR.parent
26
+ if str(SRC_ROOT) not in sys.path:
27
+ sys.path.insert(0, str(SRC_ROOT))
28
+
29
+ DATA_DIR = SRC_ROOT / "data"
30
+ FUSION_ROOT = DATA_DIR / "fused_evidence"
31
+ VECTOR_DB_DIR = DATA_DIR / "vector_db"
32
+ VECTOR_DB_DIR.mkdir(parents=True, exist_ok=True)
33
+
34
+ # ========= 自动检测最新提案 =========
35
+ subdirs = [d for d in FUSION_ROOT.iterdir() if d.is_dir()]
36
+ if not subdirs:
37
+ raise FileNotFoundError("❌ 未找到 fused_evidence,请先运行 fusion_search.py")
38
+ latest_dir = max(subdirs, key=lambda d: d.stat().st_mtime)
39
+ proposal_id = latest_dir.name
40
+ print(f"📂 当前导入提案:{proposal_id}")
41
+
42
+ # ========= 模型加载 =========
43
+ EMBED_MODEL = "BAAI/bge-large-zh-v1.5"
44
+ print(f"🧠 正在加载嵌入模型:{EMBED_MODEL} ...")
45
+ model = SentenceTransformer(EMBED_MODEL)
46
+
47
+ device = "cuda" if getattr(model, "device", None) and model.device.type == "cuda" else "cpu"
48
+ BATCH_SIZE = int(os.getenv("EMBED_BATCH", "8"))
49
+ print(f"💻 推理设备:{device.upper()} | batch_size={BATCH_SIZE}")
50
+
51
+ # ========= 初始化数据库 =========
52
+ collection_name = f"fusion_{proposal_id}"
53
+ client = chromadb.PersistentClient(path=str(VECTOR_DB_DIR))
54
+ collection = client.get_or_create_collection(
55
+ name=collection_name,
56
+ metadata={
57
+ "desc": "RAG Vector Knowledge Base (Fusion 2025.12R)",
58
+ "proposal_id": proposal_id,
59
+ "created_at": str(datetime.now())
60
+ }
61
+ )
62
+ print(f"✅ 已加载向量集合:{collection_name}")
63
+
64
+ # ========= 工具函数 =========
65
+ def _as_list(x):
66
+ if isinstance(x, list): return x
67
+ if not x: return []
68
+ return [str(x)]
69
+
70
+ def _join_urls(urls: List[str], limit=8) -> str:
71
+ urls = urls[:limit]
72
+ return "; ".join(urls)
73
+
74
+ def _load_fused_jsons(fusion_dir: Path) -> List[Dict[str, Any]]:
75
+ """
76
+ 加载 fusion_search 输出文件:
77
+ - data["fused_texts"] : [{text, urls, avg_conf}]
78
+ - 跳过短文本与低置信文本
79
+ """
80
+ docs = []
81
+ for f in fusion_dir.glob("*_fused.json"):
82
+ try:
83
+ dim = f.stem.replace("_fused", "")
84
+ data = json.loads(f.read_text(encoding="utf-8"))
85
+ fused_texts = data.get("fused_texts", [])
86
+ threshold_lowconf = 0.45
87
+ valid = 0
88
+ for i, item in enumerate(fused_texts):
89
+ if not isinstance(item, dict):
90
+ continue
91
+ text = (item.get("text") or "").strip()
92
+ urls = _as_list(item.get("urls") or [])
93
+ avg_conf = float(item.get("avg_conf") or data.get("avg_confidence_weighted", 0.6))
94
+ if len(text) < 100 or avg_conf < threshold_lowconf:
95
+ continue
96
+ doc_id = f"{proposal_id}_{dim}_{i}"
97
+ meta = {
98
+ "dimension": dim,
99
+ "proposal_id": proposal_id,
100
+ "source_file": f.name,
101
+ "urls": _join_urls(urls),
102
+ "avg_confidence": avg_conf,
103
+ "fusion_threshold": data.get("threshold", 0.65),
104
+ "embedding_model": data.get("embedding_model", EMBED_MODEL)
105
+ }
106
+ docs.append({"id": doc_id, "text": text, "meta": meta})
107
+ valid += 1
108
+ print(f"✅ 加载 {dim:<12} → {valid} 条有效文本")
109
+ except Exception as e:
110
+ print(f"⚠️ 解析 {f.name} 出错: {e}")
111
+ return docs
112
+
113
+ def _dedup_ids(ids: List[str]) -> List[bool]:
114
+ try:
115
+ existing = set(collection.get(ids=ids).get("ids", []))
116
+ except Exception:
117
+ existing = set()
118
+ return [(_id not in existing) for _id in ids]
119
+
120
+ # ========= 主流程 =========
121
+ def build_vector_db():
122
+ docs = _load_fused_jsons(latest_dir)
123
+ if not docs:
124
+ print("❌ 无融合数据可导入。")
125
+ return
126
+
127
+ texts = [d["text"] for d in docs]
128
+ ids = [d["id"] for d in docs]
129
+ metas = [d["meta"] for d in docs]
130
+ avg_len = sum(len(t) for t in texts) / max(len(texts), 1)
131
+
132
+ print(f"\n📊 文本总数:{len(texts)} | 平均长度:{avg_len:.1f} 字符")
133
+
134
+ # ===== 嵌入生成 =====
135
+ print("🧩 正在生成嵌入向量 ...")
136
+ embeddings = model.encode(
137
+ texts,
138
+ normalize_embeddings=True,
139
+ show_progress_bar=True,
140
+ batch_size=BATCH_SIZE,
141
+ device=device
142
+ )
143
+
144
+ # ===== 去重写入 =====
145
+ print("💾 正在写入 Chroma 数据库 ...")
146
+ new_mask = _dedup_ids(ids)
147
+ new_ids = [i for i, keep in zip(ids, new_mask) if keep]
148
+ if not new_ids:
149
+ print("⚠️ 无新增记录(可能已导入过该提案)。")
150
+ else:
151
+ collection.add(
152
+ ids=new_ids,
153
+ documents=[t for t, keep in zip(texts, new_mask) if keep],
154
+ embeddings=[e.tolist() for e, keep in zip(embeddings, new_mask) if keep],
155
+ metadatas=[m for m, keep in zip(metas, new_mask) if keep]
156
+ )
157
+ print(f"✅ 已写入 {len(new_ids)} 条新记录。")
158
+
159
+ # ===== 统计分布 =====
160
+ dim_stats = {}
161
+ conf_all = []
162
+ for m in metas:
163
+ dim = m["dimension"]
164
+ conf = float(m.get("avg_confidence", 0.6))
165
+ conf_all.append(conf)
166
+ dim_stats.setdefault(dim, {"count": 0, "conf_sum": 0.0, "high": 0, "mid": 0, "low": 0})
167
+ dim_stats[dim]["count"] += 1
168
+ dim_stats[dim]["conf_sum"] += conf
169
+ if conf >= 0.8:
170
+ dim_stats[dim]["high"] += 1
171
+ elif conf >= 0.6:
172
+ dim_stats[dim]["mid"] += 1
173
+ else:
174
+ dim_stats[dim]["low"] += 1
175
+
176
+ dim_distribution = {
177
+ k: {
178
+ "count": v["count"],
179
+ "avg_conf": round(v["conf_sum"] / max(v["count"], 1), 2),
180
+ "confidence_tiers": {
181
+ "high": v["high"],
182
+ "mid": v["mid"],
183
+ "low": v["low"]
184
+ }
185
+ } for k, v in dim_stats.items()
186
+ }
187
+
188
+ global_avg_conf = round(np.mean(conf_all), 3)
189
+ total_records = sum(v["count"] for v in dim_stats.values())
190
+ stored_count = collection.count()
191
+ insert_rate = round((stored_count / max(total_records, 1)) * 100, 1)
192
+
193
+ # ===== 保存索引 =====
194
+ index_info = {
195
+ "proposal_id": proposal_id,
196
+ "collection_name": collection_name,
197
+ "record_count": total_records,
198
+ "stored_count": stored_count,
199
+ "insert_rate_percent": insert_rate,
200
+ "avg_text_length": round(avg_len, 1),
201
+ "dimension_distribution": dim_distribution,
202
+ "global_avg_confidence": global_avg_conf,
203
+ "embedding_model": EMBED_MODEL,
204
+ "timestamp": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
205
+ "db_path": str(VECTOR_DB_DIR.resolve())
206
+ }
207
+
208
+ index_path = VECTOR_DB_DIR / f"{proposal_id}_vector_index.json"
209
+ index_path.write_text(json.dumps(index_info, ensure_ascii=False, indent=2), encoding="utf-8")
210
+
211
+ print(f"\n🧾 向量库索引已保存:{index_path}")
212
+ print(f"📦 当前集合条目:{stored_count} | 全局平均置信度:{global_avg_conf}")
213
+ print(f"📊 入库率:{insert_rate:.1f}%")
214
+ print("🎯 向量知识库构建完成。")
215
+
216
+ # ========= 入口 =========
217
+ if __name__ == "__main__":
218
+ build_vector_db()
src/tools/domain_profiler.py ADDED
@@ -0,0 +1,213 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ from __future__ import annotations
3
+
4
+ import json
5
+ import os
6
+ import re
7
+ from collections import Counter
8
+ from typing import Any, Dict, List, Sequence, Tuple
9
+
10
+ from src.prompting.domain_adaptive import sanitize_domain_profile
11
+
12
+ try:
13
+ from openai import OpenAI # type: ignore
14
+ except Exception:
15
+ OpenAI = None
16
+
17
+
18
+ DEFAULT_MODEL = os.getenv("OPENAI_MODEL", "gpt-4o-mini")
19
+
20
+
21
+ def _clean_text(text: str) -> str:
22
+ return re.sub(r"\s+", " ", text or "").strip()
23
+
24
+
25
+ def extract_keywords(text: str, max_terms: int = 18) -> List[str]:
26
+ tokens = re.findall(r"[A-Za-z][A-Za-z\-\+\.]{2,}|[\u4e00-\u9fff]{2,}", text)
27
+ stop = {
28
+ "proposal", "project", "system", "technology", "market", "company", "team", "development",
29
+ "研究", "项目", "公司", "系统", "技术", "团队", "市场", "发展", "方案", "计划", "应用",
30
+ "产品", "核心", "页码", "介绍", "融资", "背景", "价值",
31
+ }
32
+ counts = Counter(tok.lower() for tok in tokens if tok.lower() not in stop)
33
+ return [t for t, _ in counts.most_common(max_terms)]
34
+
35
+
36
+ def heuristic_domain_profile(full_text: str, pages: Sequence[Dict[str, Any]]) -> Dict[str, Any]:
37
+ """
38
+ Strictly derive a lightweight domain profile from the CURRENT document only.
39
+ No hard-coded domain catalog, no historical/default domain injection.
40
+ """
41
+ text = _clean_text(full_text)
42
+ keywords = extract_keywords(text)
43
+
44
+ generic_terms = {
45
+ "model", "models", "system", "systems", "method", "methods",
46
+ "analysis", "data", "study", "research", "approach",
47
+ "based", "using", "results", "validation", "evaluation"
48
+ }
49
+
50
+ top_keywords = [kw for kw in keywords if kw and kw.lower() not in generic_terms][:8]
51
+
52
+ if not top_keywords:
53
+ return sanitize_domain_profile({
54
+ "domain": {"primary": "unknown", "secondary": []},
55
+ "evaluation_focus": {
56
+ "problem": [],
57
+ "objectives": [],
58
+ "feasibility": [],
59
+ "innovation": [],
60
+ "team": [],
61
+ "outcomes": [],
62
+ },
63
+ "methods": [],
64
+ "risks": [],
65
+ "terminology": [],
66
+ })
67
+
68
+ primary_domain = top_keywords[0]
69
+ secondary_domains = top_keywords[1:4]
70
+
71
+ profile = {
72
+ "domain": {
73
+ "primary": primary_domain,
74
+ "secondary": secondary_domains,
75
+ },
76
+ "evaluation_focus": {
77
+ "problem": [
78
+ "problem importance",
79
+ "context clarity",
80
+ "need definition",
81
+ ],
82
+ "objectives": [
83
+ "objective clarity",
84
+ "scope definition",
85
+ "deliverable specificity",
86
+ ],
87
+ "feasibility": [
88
+ "resource readiness",
89
+ "timeline realism",
90
+ "dependency management",
91
+ ],
92
+ "innovation": [
93
+ "novelty",
94
+ "differentiation",
95
+ "comparative advantage",
96
+ ],
97
+ "team": [
98
+ "relevant expertise",
99
+ "role coverage",
100
+ "execution capability",
101
+ ],
102
+ "outcomes": [
103
+ "measurable outputs",
104
+ "impact logic",
105
+ "evaluation plan",
106
+ ],
107
+ },
108
+ "methods": top_keywords[:4],
109
+ "risks": [],
110
+ "terminology": top_keywords[:8],
111
+ }
112
+
113
+ return sanitize_domain_profile(profile)
114
+
115
+
116
+ LLM_PROFILER_SYSTEM = """
117
+ You are a strict domain profiler for proposal review.
118
+
119
+ Your task is ONLY to derive a clean domain profile from the CURRENT document text.
120
+
121
+ Hard rules:
122
+ - Use ONLY the provided document text.
123
+ - Ignore any prior tasks, prior files, templates, examples, cached memory, or default domains.
124
+ - Do NOT guess a domain from common proposal patterns.
125
+ - Do NOT mix domains unless the document clearly and repeatedly supports both.
126
+ - If the evidence is weak, return "unknown".
127
+ - Prefer short, concrete domain labels grounded in repeated terms from the text.
128
+ - Do NOT output aerospace, biomedical, or any specific field unless clearly supported by repeated explicit terms.
129
+ - Keep all lists short and text-grounded.
130
+ - Return valid JSON only.
131
+
132
+ Return JSON with exactly this structure:
133
+ {
134
+ "domain": {
135
+ "primary": "string",
136
+ "secondary": ["string"]
137
+ },
138
+ "evaluation_focus": {
139
+ "problem": ["string"],
140
+ "objectives": ["string"],
141
+ "feasibility": ["string"],
142
+ "innovation": ["string"],
143
+ "team": ["string"],
144
+ "outcomes": ["string"]
145
+ },
146
+ "methods": ["string"],
147
+ "risks": ["string"],
148
+ "terminology": ["string"]
149
+ }
150
+ """.strip()
151
+
152
+
153
+ def profile_with_optional_llm(full_text: str, pages: Sequence[Dict[str, Any]]) -> Tuple[Dict[str, Any], str]:
154
+ heuristic = heuristic_domain_profile(full_text, pages)
155
+
156
+ api_key = os.getenv("OPENAI_API_KEY", "").strip()
157
+ if not api_key or OpenAI is None:
158
+ return heuristic, "heuristic"
159
+
160
+ try:
161
+ client = OpenAI(api_key=api_key)
162
+ response = client.chat.completions.create(
163
+ model=DEFAULT_MODEL,
164
+ messages=[
165
+ {"role": "system", "content": LLM_PROFILER_SYSTEM},
166
+ {"role": "user", "content": full_text[:16000]},
167
+ ],
168
+ response_format={"type": "json_object"},
169
+ temperature=0.0,
170
+ max_tokens=900,
171
+ )
172
+
173
+ raw = response.choices[0].message.content or "{}"
174
+ data = sanitize_domain_profile(json.loads(raw))
175
+
176
+ primary = _clean_text(data.get("domain", {}).get("primary", ""))
177
+ terminology = data.get("terminology", []) or []
178
+ methods = data.get("methods", []) or []
179
+ risks = data.get("risks", []) or []
180
+
181
+ bad_primary = {
182
+ "",
183
+ "general",
184
+ "general proposal",
185
+ "proposal",
186
+ "research proposal",
187
+ "project proposal",
188
+ "unknown",
189
+ }
190
+
191
+ def is_fragment_list(lst):
192
+ if not lst:
193
+ return True
194
+ short = len([x for x in lst if len(x.strip()) <= 1])
195
+ return short >= max(1, len(lst) * 0.6)
196
+
197
+ def is_punctuation_noise(lst):
198
+ return any(re.fullmatch(r"[、,;;\.\-]+", x.strip()) for x in lst)
199
+
200
+ # 🚨 强力质量拦截
201
+ if (
202
+ primary.lower() in bad_primary
203
+ or is_fragment_list(methods)
204
+ or is_fragment_list(risks)
205
+ or is_punctuation_noise(methods)
206
+ or is_punctuation_noise(risks)
207
+ ):
208
+ return heuristic, "heuristic_fallback_bad_quality"
209
+
210
+ return data, "llm"
211
+
212
+ except Exception:
213
+ return heuristic, "heuristic_fallback"
src/tools/extract_facts_by_chunk.py ADDED
@@ -0,0 +1,786 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ Stage 1 · 块级事实抽取器(extract_facts_by_chunk.py)
4
+ ----------------------------------------------------
5
+ 输入:
6
+ - src/data/prepared/<proposal_id>/full_text.txt (由 prepare_proposal_text.py 生成)
7
+
8
+ 输出:
9
+ - src/data/extracted/<proposal_id>/raw_facts.jsonl 每行一个 JSON fact
10
+
11
+ 职责:
12
+ - 将长文按字符切块(带 overlap)
13
+ - 对每个块,用 LLM 抽取“原子事实列表”
14
+ - 每条事实附带:dimensions[], type, meta(chunk_index, char_range 由代码填充)
15
+ - 后续 Stage 2 再用这些 facts 去构建五个维度的最终 dimensions_v2.json
16
+
17
+ 本版优化要点:
18
+ 1)减小 chunk 大小、加大 overlap,提高单块信息覆盖率,避免一个块塞太多信息导致抽不全。
19
+ 2)加强 Prompt 中对“五维覆盖”的要求,显式提醒模型不要只抽单一维度的信息。
20
+ 3)更激进的 type→dimensions 映射,让跨维度事实被多个维度同时看到,避免某维度信息过于稀薄。
21
+ 4)扩充关键词推断逻辑 _infer_dims_from_text,补充 objectives / innovation / feasibility 等隐性表述。
22
+ 5)在运行结束时输出五个维度的事实数量分布,并对明显偏少的维度给出警告,便于调参与排错。
23
+ 6)新增:对“文本很长但抽取事实过少”的 chunk 自动再跑一轮 dense 模式,强化召回。
24
+ """
25
+
26
+ import os
27
+ import json
28
+ import argparse
29
+ import re
30
+ from pathlib import Path
31
+ from typing import List, Dict, Any
32
+
33
+ from dotenv import load_dotenv
34
+ try:
35
+ from openai import OpenAI
36
+ except Exception: # pragma: no cover
37
+ OpenAI = None
38
+
39
+ from .layout_reconstruction import semantic_pages_for_stage1
40
+
41
+ load_dotenv()
42
+
43
+ # ========= 路径配置 =========
44
+
45
+ BASE_DIR = Path(__file__).resolve().parents[2]
46
+ PREPARED_DIR = BASE_DIR / "src" / "data" / "prepared"
47
+ EXTRACTED_DIR = BASE_DIR / "src" / "data" / "extracted"
48
+
49
+ OPENAI_MODEL = os.getenv("OPENAI_MODEL", "gpt-4o-mini")
50
+
51
+ # 全局复用一个 OpenAI client,避免每个 chunk 反复初始化
52
+ client = OpenAI() if OpenAI is not None else None
53
+
54
+ VALID_DIMENSIONS = ["team", "objectives", "strategy", "innovation", "feasibility"]
55
+ VALID_TYPES = [
56
+ "team_member",
57
+ "org_structure",
58
+ "collaboration",
59
+ "resource",
60
+ "pipeline",
61
+ "milestone",
62
+ "market",
63
+ "tech_route",
64
+ "product",
65
+ "ip_asset",
66
+ "evidence",
67
+ "budget_item",
68
+ "funding_source",
69
+ "risk",
70
+ "mitigation",
71
+ "ai_model",
72
+ "clinical_design",
73
+ "regulatory",
74
+ "other",
75
+ ]
76
+
77
+ # ⚠️ Prompt:既要防幻觉,又要尽量“捞干净”五维相关信息,并保证多维覆盖
78
+
79
+ FACT_PROMPT = """
80
+ 你是一个“事实抽取器”,负责从一小段项目文本中逐条抽取可核对的原子事实。
81
+
82
+ 硬性约束:
83
+ 1)只处理当前文本块,不猜测其他页面或上下文。
84
+ 2)不得编造任何文本中没有出现的机构、人物、对象、方法、数据、数字、时间、地点或结果。
85
+ 3)每条 fact 必须能在原文中找到对应依据,可以轻微改写,但必须保留关键名词。
86
+ 4)每条 fact 尽量只表达一个独立事实;若一句话包含多个主题,请拆开。
87
+ 5)只输出 JSON,不输出解释。
88
+
89
+ 适用范围:
90
+ - 任意科研、技术、教育、产业、治理或跨学科项目
91
+ - 不得预设具体领域
92
+
93
+ 优先抽取的信息:
94
+ - 团队与组织:成员、角色、机构、职责、协作关系、治理安排
95
+ - 目标与范围:总体目标、阶段目标、任务边界、里程碑、交付物、评价指标
96
+ - 方法与路线:技术路线、研究方法、实施步骤、验证设计、工作流、平台或工具
97
+ - 创新与差异化:新颖点、独特资源、已有证据、比较优势、知识产权
98
+ - 可行性与风险:资源基础、数据/设备/资金/时间条件、风险、限制、应对措施
99
+ - 外部环境:用户/对象/场景/应用环境/政策约束/市场与需求信息(若文本明确提及)
100
+
101
+ 抽取粒度要求:
102
+ - 信息丰富时尽量抽取 12–25 条 facts;信息较少时可少于 10 条;总数不要超过 25 条。
103
+ - 尽量覆盖不同主题,不要只集中在单一维度。
104
+
105
+ 维度标签(dimensions):
106
+ - "team": 团队、角色、机构、分工、协同、治理
107
+ - "objectives": 问题背景、目标、范围、里程碑、交付物、评价指标
108
+ - "strategy": 方法、技术路线、实施流程、实验/验证/落地路径、外部协作策略
109
+ - "innovation": 新颖性、差异化、独特资源、证据优势、知识产权
110
+ - "feasibility": 资源、预算、时间、风险、限制、依赖关系、应对措施
111
+
112
+ 要求:
113
+ - 每条 fact 至少有 1 个维度标签;允许多标签。
114
+ - 若无法判断,使用 ["feasibility"] 兜底。
115
+
116
+ type 可选值:
117
+ - team_member, org_structure, collaboration, resource, pipeline, milestone,
118
+ market, tech_route, product, ip_asset, evidence, budget_item, funding_source,
119
+ risk, mitigation, ai_model, clinical_design, regulatory, other
120
+
121
+ type 选���规则:
122
+ - 选择最接近文本语义的类型即可;若不确定,用 other。
123
+ - market / regulatory / clinical_design 等类型可以用于任何“外部约束、验证设计、制度流程”场景,
124
+ 不代表你必须假设项目属于某个特定行业。
125
+
126
+ 输出格式:
127
+ {
128
+ "facts": [
129
+ {
130
+ "text": "一条具体、可核对的事实",
131
+ "dimensions": ["team", "strategy"],
132
+ "type": "team_member"
133
+ }
134
+ ]
135
+ }
136
+
137
+ 如果当前文本块没有任何有用事实,返回 {"facts": []}。
138
+ 现在开始处理我给你的文本块。
139
+ """
140
+
141
+
142
+ def find_latest_prepared_proposal() -> str:
143
+ if not PREPARED_DIR.exists():
144
+ raise FileNotFoundError(f"未找到 prepared 目录: {PREPARED_DIR}")
145
+
146
+ candidates = []
147
+ for d in PREPARED_DIR.iterdir():
148
+ if d.is_dir():
149
+ candidates.append((d.stat().st_mtime, d.name))
150
+
151
+ if not candidates:
152
+ raise FileNotFoundError(f"prepared 目录下没有任何提案子目录: {PREPARED_DIR}")
153
+
154
+ proposal_id = max(candidates, key=lambda x: x[0])[1]
155
+ print(f"[INFO] [auto] 选中最新提案 ID: {proposal_id}")
156
+ return proposal_id
157
+
158
+
159
+
160
+
161
+ def load_page_semantics(proposal_id: str) -> Dict[str, Any]:
162
+ path = PREPARED_DIR / proposal_id / "page_semantics.json"
163
+ if not path.exists():
164
+ return {}
165
+ try:
166
+ data = json.loads(path.read_text(encoding="utf-8"))
167
+ print(f"[INFO] 读取 page_semantics: {path}")
168
+ return data
169
+ except Exception as e:
170
+ print(f"[WARN] page_semantics.json 读取失败,退回 full_text 模式: {e}")
171
+ return {}
172
+
173
+
174
+ def load_semantic_units(proposal_id: str) -> List[Dict[str, Any]]:
175
+ page_sem = load_page_semantics(proposal_id)
176
+ if not page_sem:
177
+ return []
178
+ units = semantic_pages_for_stage1(page_sem)
179
+ return [u for u in units if (u.get("text") or "").strip()]
180
+
181
+ def load_full_text(proposal_id: str) -> str:
182
+ path = PREPARED_DIR / proposal_id / "full_text.txt"
183
+ if not path.exists():
184
+ raise FileNotFoundError(f"full_text.txt 不存在: {path}")
185
+ text = path.read_text(encoding="utf-8")
186
+ print(f"[INFO] 读取 full_text: {path} (长度 {len(text)} 字符)")
187
+ return text
188
+
189
+
190
+ def make_chunks(text: str, max_chars: int = 1800, overlap: int = 400) -> List[Dict[str, Any]]:
191
+ """
192
+ 简单按字符切块,带 overlap;不做句子级别切分。
193
+ 默认 max_chars=1800, overlap=400,比之前更细、更密,有利于提高抽取覆盖率。
194
+ 返回列表,每个元素包含: chunk_text, start, end, index
195
+ """
196
+ chunks = []
197
+ n = len(text)
198
+ if n == 0:
199
+ return chunks
200
+
201
+ idx = 0
202
+ chunk_idx = 0
203
+ while idx < n:
204
+ end = min(n, idx + max_chars)
205
+ chunk_text = text[idx:end]
206
+ chunks.append(
207
+ {
208
+ "index": chunk_idx,
209
+ "start": idx,
210
+ "end": end,
211
+ "text": chunk_text,
212
+ }
213
+ )
214
+ chunk_idx += 1
215
+ if end == n:
216
+ break
217
+ idx = max(0, end - overlap)
218
+
219
+ print(f"[INFO] 已切分为 {len(chunks)} 个 chunk (max_chars={max_chars}, overlap={overlap})")
220
+ return chunks
221
+
222
+
223
+ def make_semantic_chunks(units: List[Dict[str, Any]], max_chars: int = 2200) -> List[Dict[str, Any]]:
224
+ chunks: List[Dict[str, Any]] = []
225
+ current_parts: List[str] = []
226
+ current_pages: List[int] = []
227
+ current_len = 0
228
+ chunk_idx = 0
229
+
230
+ def flush():
231
+ nonlocal current_parts, current_pages, current_len, chunk_idx
232
+ if not current_parts:
233
+ return
234
+ text = "\n\n".join(current_parts).strip()
235
+ chunks.append({
236
+ "index": chunk_idx,
237
+ "start": 0,
238
+ "end": len(text),
239
+ "text": text,
240
+ "page_indices": sorted(set(current_pages)),
241
+ "source": "page_semantics",
242
+ })
243
+ chunk_idx += 1
244
+ current_parts = []
245
+ current_pages = []
246
+ current_len = 0
247
+
248
+ for unit in units:
249
+ text = (unit.get("text") or "").strip()
250
+ if not text:
251
+ continue
252
+ page_idx = unit.get("page_index")
253
+ if current_len and current_len + len(text) + 2 > max_chars:
254
+ flush()
255
+ current_parts.append(text)
256
+ if page_idx is not None:
257
+ current_pages.append(page_idx)
258
+ current_len += len(text) + 2
259
+ flush()
260
+ print(f"[INFO] 基于 page_semantics 构建 {len(chunks)} 个语义 chunk")
261
+ return chunks
262
+
263
+
264
+ def call_llm_for_chunk(chunk_text: str, attempt: int = 1, dense: bool = False) -> Dict[str, Any]:
265
+ """
266
+ 调用 OpenAI,对单个 chunk 抽取 facts。
267
+ - attempt > 1:用于 JSON 解析失败后的重试(提示“上一次 JSON 不合法”)。
268
+ - dense = True:用于“当前 chunk 文本很长但事实过少”的第二轮密集抽取,会额外要求多抽一些 facts。
269
+ """
270
+ extra_hint_parts = []
271
+
272
+ if attempt > 1:
273
+ extra_hint_parts.append(
274
+ "⚠️ 注意:上一次你返回的 JSON 因为太��或不合法导致解析失败。"
275
+ "这一次请严格控制 facts 数量不超过 18 条,并且务必保证 JSON 语法完全正确。"
276
+ )
277
+
278
+ if dense:
279
+ extra_hint_parts.append(
280
+ "⚠️ 当前文本块信息非常丰富,你在本次抽取时应尽量覆盖文本中出现的所有与团队、目标、"
281
+ "策略、创新、可行性相关的关键事实。请优先抽取 15–22 条 facts,"
282
+ "并尽量覆盖不同维度和不同主题,不要只聚焦在单一方面。"
283
+ )
284
+
285
+ extra_hint = ""
286
+ if extra_hint_parts:
287
+ extra_hint = "\n\n" + "\n".join(extra_hint_parts)
288
+
289
+ messages = [
290
+ {
291
+ "role": "system",
292
+ "content": "你是一个严谨的事实抽取器,只能基于给定文本块抽取原子事实,不得编造。",
293
+ },
294
+ {
295
+ "role": "user",
296
+ "content": FACT_PROMPT
297
+ + extra_hint
298
+ + "\n\n=== 文本块开始 ===\n"
299
+ + chunk_text.strip(),
300
+ },
301
+ ]
302
+
303
+ if client is None:
304
+ raise RuntimeError("OpenAI SDK 未安装或不可用,无法执行 Stage 1 facts 抽取。")
305
+
306
+ resp = client.chat.completions.create(
307
+ model=OPENAI_MODEL,
308
+ messages=messages,
309
+ response_format={"type": "json_object"},
310
+ temperature=0.0,
311
+ max_tokens=1800,
312
+ )
313
+
314
+ raw = resp.choices[0].message.content
315
+ try:
316
+ data = json.loads(raw)
317
+ except json.JSONDecodeError as e:
318
+ print("[WARN] JSON 解析失败,返回原始内容,便于你排查:")
319
+ print(raw)
320
+ if attempt == 1:
321
+ print("[INFO] 尝试使用收缩版 prompt 重试该 chunk ...")
322
+ return call_llm_for_chunk(chunk_text, attempt=2, dense=dense)
323
+ # 第二次还失败就直接抛出
324
+ raise e
325
+
326
+ if not isinstance(data, dict):
327
+ data = {"facts": []}
328
+ if "facts" not in data or not isinstance(data["facts"], list):
329
+ data["facts"] = []
330
+
331
+ return data
332
+
333
+ # ===== 市场类关键词 & 识别函数 =====
334
+
335
+ _MARKET_CN = [
336
+ "市场", "市场规模", "市场容量", "市场需求", "市场前景", "市场潜力",
337
+ "目标市场", "细分市场", "市场份额", "渗透率",
338
+ "客户", "客户群体", "目标客户", "目标人群",
339
+ "患者群体", "目标患者",
340
+ "销售", "销售额", "销量", "营收", "收入", "收益",
341
+ "定价", "价格", "报销", "支付方", "医保", "保险",
342
+ "商业化", "商业模式", "商业机会",
343
+ "竞争", "竞品", "竞争对手", "竞争格局",
344
+ "CAGR", "增长率"
345
+ ]
346
+
347
+ _MARKET_EN = [
348
+ "market", "market size", "market volume", "market demand", "market potential",
349
+ "target market", "segment", "niche",
350
+ "market share", "penetration",
351
+ "customer", "customers", "client", "clients",
352
+ "patient population", "target patients",
353
+ "sales", "revenue", "turnover", "income",
354
+ "pricing", "price", "reimbursement", "payer", "insurance",
355
+ "commercialization", "commercialisation", "business model",
356
+ "competition", "competitive", "competitor", "competitors",
357
+ "cagr", "growth rate"
358
+ ]
359
+
360
+
361
+ def _looks_like_market_fact(text: str) -> bool:
362
+ """
363
+ 判断一条 fact 的文本是否“明显是市场/商业相关”的内容。
364
+ 注意:这里只用来纠偏 type,对 risk/mitigation 不覆盖。
365
+ """
366
+ if not text:
367
+ return False
368
+ t = text.lower()
369
+
370
+ # 中文关键字
371
+ if any(k in text for k in _MARKET_CN):
372
+ return True
373
+
374
+ # 英文关键字
375
+ if any(k in t for k in _MARKET_EN):
376
+ return True
377
+
378
+ return False
379
+
380
+ def _infer_dims_from_text(text: str) -> List[str]:
381
+ """
382
+ 当 LLM 没有给出 dimensions 且 type 也无法可靠映射时,
383
+ 用中英文关键词做一轮粗略的维度推断,尽量不要丢掉有用信息。
384
+ 这里只做“补充”,不会覆盖已有维度。
385
+ """
386
+ t_lower = (text or "").lower()
387
+ t = text or ""
388
+
389
+ candidates = set()
390
+
391
+ # ---- team ----
392
+ team_keywords = [
393
+ "团队", "小组", "联合体", "合作单位", "合作方", "协作单位",
394
+ "研究者", "研究团队", "专业团队", "项目组",
395
+ "负责人", "项目负责人", "带头人",
396
+ "教授", "副教授", "主任", "专家", "研究员", "博士", "博士后",
397
+ "机构", "大学", "研究所", "中心", "实验室",
398
+ "ceo", "coo", "cto", "cso", "vp", "vice president",
399
+ "founder", "co-founder", "chief executive officer",
400
+ ]
401
+ if any(k in t for k in team_keywords) or any(k in t_lower for k in team_keywords):
402
+ candidates.add("team")
403
+
404
+ # ---- objectives ----
405
+ obj_keywords_cn = [
406
+ "目标", "总体目标", "阶段性目标", "里程碑", "阶段性里程碑",
407
+ "计划", "任务", "工作包", "kpi", "终点", "主要终点", "次要终点",
408
+ "本项目旨在", "本项目将", "本项目计划", "预期达到", "希望实现",
409
+ ]
410
+ obj_keywords_en = [
411
+ "aim", "aims to", "aimed to",
412
+ "objective", "objectives", "goal", "goals",
413
+ "milestone", "milestones", "endpoint", "endpoints",
414
+ "is designed to", "seeks to", "intends to", "in order to",
415
+ ]
416
+ if any(k in t for k in obj_keywords_cn) or any(k in t_lower for k in obj_keywords_en):
417
+ candidates.add("objectives")
418
+
419
+ # ---- strategy ----
420
+ strat_keywords_cn = [
421
+ "策略", "路径", "路线", "方案", "技术路线", "实施方案",
422
+ "商业模式", "市场进入", "商业化", "推广策略",
423
+ "合作模式", "运营模式", "联合开发", "授权引进",
424
+ "市场", "市场规模", "市场需求", "市场前景", "市场潜力",
425
+ "市场份额", "竞争格局", "竞品", "竞争对手"
426
+ ]
427
+ strat_keywords_en = [
428
+ "strategy", "strategies", "pathway", "roadmap",
429
+ "commercial", "commercialization", "business model",
430
+ "market entry", "go-to-market", "go to market",
431
+ "market", "market size", "market demand", "market potential",
432
+ "market share", "competitive landscape", "competition", "competitor", "competitors",
433
+ "partnership", "licensing", "co-development",
434
+ "development plan", "regulatory strategy",
435
+ ]
436
+
437
+ if any(k in t for k in strat_keywords_cn) or any(k in t_lower for k in strat_keywords_en):
438
+ candidates.add("strategy")
439
+
440
+ # ---- innovation ----
441
+ inno_keywords_cn = [
442
+ "创新", "创新性", "差异化", "独特", "首创", "领先", "颠覆",
443
+ "新一代", "新型", "原创", "填补空白", "突破性", "首个", "第一例",
444
+ ]
445
+ inno_keywords_en = [
446
+ "novel", "novelty", "innovative", "innovation",
447
+ "differentiated", "differentiation", "unique",
448
+ "first-in-class", "best-in-class", "state-of-the-art",
449
+ "cutting-edge", "breakthrough", "original", "disruptive",
450
+ "fills the gap", "fill the gap",
451
+ ]
452
+ if any(k in t for k in inno_keywords_cn) or any(k in t_lower for k in inno_keywords_en):
453
+ candidates.add("innovation")
454
+
455
+ # ---- feasibility ----
456
+ feas_keywords_cn = [
457
+ "可行性", "可行", "可实施", "资源", "平台", "基础设施",
458
+ "预算", "经费", "资金", "成本", "成本负担",
459
+ "风险", "挑战", "瓶颈", "不确定性",
460
+ "时间表", "进度", "周期", "排期",
461
+ "入组难度", "依从性", "工作量", "实施复杂度",
462
+ ]
463
+ feas_keywords_en = [
464
+ "feasibility", "feasible", "resource", "resources", "infrastructure",
465
+ "budget", "funding", "cost", "costs", "cost-effectiveness",
466
+ "risk", "risks", "challenge", "challenges", "bottleneck", "uncertainty",
467
+ "timeline", "schedule", "timeframe",
468
+ "enrollment", "recruitment", "compliance", "adherence", "burden",
469
+ ]
470
+ if any(k in t for k in feas_keywords_cn) or any(k in t_lower for k in feas_keywords_en):
471
+ candidates.add("feasibility")
472
+
473
+ # ---- 为“需求/应用/外部环境”类段落兜底:补充 strategy / objectives ----
474
+ market_keywords_cn = [
475
+ "市场", "市场规模", "市场分析", "市场份额", "市场占有率",
476
+ "cagr", "复合年增长率", "销售额", "营收", "收入", "销售收入",
477
+ "增长率", "增长幅度", "客户", "用户", "消费群体",
478
+ ]
479
+ market_keywords_en = [
480
+ "market size", "market", "cagr", "market share", "share of",
481
+ "sales", "revenue", "revenues", "turnover",
482
+ "growth rate", "compound annual growth", "customer", "customers",
483
+ "payer", "payers",
484
+ ]
485
+ if any(k in t for k in market_keywords_cn) or any(k in t_lower for k in market_keywords_en):
486
+ candidates.add("strategy")
487
+ candidates.add("objectives")
488
+
489
+ return [d for d in candidates if d in VALID_DIMENSIONS]
490
+
491
+ def mark_numeric_suspect(fact: Dict[str, Any], chunk_text: str) -> Dict[str, Any]:
492
+ """
493
+ 对包含数字的 fact 做简单校验:
494
+ - 把 fact.text 里的数字片段(连续数字,不管是年份/金额)提取出来
495
+ - 如果某个数字完全不出现在 chunk_text 中,则认为这条 fact 存在数字幻觉风险
496
+ - 加 meta.suspect_numeric = True/False
497
+ """
498
+ text = fact.get("text", "") or ""
499
+ nums = re.findall(r"\d+", text)
500
+ if not nums:
501
+ return fact # 没数字,不管
502
+
503
+ chunk_flat = (chunk_text or "").replace(" ", "")
504
+ suspect = False
505
+ for n in nums:
506
+ if n not in chunk_flat:
507
+ suspect = True
508
+ break
509
+
510
+ meta = fact.get("meta", {})
511
+ if not isinstance(meta, dict):
512
+ meta = {}
513
+ meta["suspect_numeric"] = suspect
514
+ fact["meta"] = meta
515
+ return fact
516
+
517
+ def normalize_fact(
518
+ fact: Dict[str, Any],
519
+ proposal_id: str,
520
+ chunk_index: int,
521
+ start: int,
522
+ end: int,
523
+ ) -> Dict[str, Any]:
524
+ """
525
+ 给每条 fact 填上 meta 信息;清洗 dimensions / type。
526
+ 同时对维度标签做 type→dimensions 的通用“补标签”映射(支持多维度),
527
+ 尽量保证五个维度的信息都不会被漏掉。
528
+ """
529
+ text = fact.get("text", "")
530
+ if not isinstance(text, str):
531
+ text = str(text)
532
+
533
+ # 原始维度标签清洗
534
+ dims = fact.get("dimensions", [])
535
+ if not isinstance(dims, list):
536
+ dims = []
537
+ dims_clean = [d for d in dims if isinstance(d, str) and d in VALID_DIMENSIONS]
538
+
539
+ type_val = fact.get("type", "other")
540
+ if type_val not in VALID_TYPES:
541
+ type_val = "other"
542
+
543
+ # ===== 外部环境/市场类事实的自动纠偏(在 type→dimensions 映射之前)=====
544
+ # 如果 LLM 没有标成 market,但文本里明显是需求/应用/市场相关内容,则强制改为 "market"
545
+ # (避免所有市场信息都被丢在 "other" 或 "product" 里)
546
+ if type_val not in ["market", "risk", "mitigation"]:
547
+ if _looks_like_market_fact(text):
548
+ type_val = "market"
549
+
550
+ # 先把已有维度放进一个 set,后面按 type / 文本内容补充
551
+ dim_set = set(dims_clean)
552
+
553
+ # ===== 1. 通用 type→dimensions 映射(更激进版本,优先保证信息被多个维度看到) =====
554
+ if type_val in ["team_member", "org_structure"]:
555
+ # 团队成员 / 组织结构 → 明确归入 team
556
+ dim_set.add("team")
557
+
558
+ elif type_val == "collaboration":
559
+ # 协作既是团队协同,也是合作策略
560
+ dim_set.update(["team", "strategy"])
561
+
562
+ elif type_val == "pipeline":
563
+ # 工作包/子任务:目标 + 路线,既体现 objectives,也体现 strategy
564
+ dim_set.update(["objectives", "strategy"])
565
+
566
+ elif type_val == "milestone":
567
+ # 里程碑:目标的阶段性拆分 + 执行可行性
568
+ dim_set.update(["objectives", "feasibility"])
569
+
570
+ elif type_val == "clinical_design":
571
+ # 验证/评估设计:目标路径 + 方案策略 + 执行可行性
572
+ dim_set.update(["objectives", "strategy", "feasibility"])
573
+
574
+ elif type_val in ["market", "product"]:
575
+ # 产品/应用环境:实施策略 + 落地可行性
576
+ dim_set.update(["objectives", "strategy", "feasibility"])
577
+
578
+ elif type_val in ["tech_route", "regulatory"]:
579
+ # 技术路线 / 制度流程:典型 strategy,但对可行性也有影响
580
+ dim_set.update(["strategy", "feasibility"])
581
+
582
+ elif type_val == "funding_source":
583
+ # 资金来源:既体现策略布局,也影响可行性
584
+ dim_set.update(["strategy", "feasibility"])
585
+
586
+ elif type_val == "ip_asset":
587
+ # IP 资产:创新优势 + 中长期可行性
588
+ dim_set.update(["innovation", "feasibility"])
589
+
590
+ elif type_val == "evidence":
591
+ # 证据:创新价值 + 可行性(证据越强,可行性越高)
592
+ dim_set.update(["innovation", "feasibility"])
593
+
594
+ elif type_val == "ai_model":
595
+ # AI 模型既是创新亮点,也是策略的一部分
596
+ dim_set.update(["innovation", "strategy"])
597
+
598
+ elif type_val in ["resource", "budget_item", "risk", "mitigation"]:
599
+ # 资源、预算、风险与应对 → 可行性
600
+ dim_set.add("feasibility")
601
+
602
+ # ===== 2. 如果还是没维度,用文本关键词再判断一轮 =====
603
+ if not dim_set:
604
+ inferred = _infer_dims_from_text(text)
605
+ for d in inferred:
606
+ dim_set.add(d)
607
+
608
+ # ===== 3. 最终兜底:还没有,就放到 feasibility,避免彻底丢失 =====
609
+ if not dim_set:
610
+ dim_set.add("feasibility")
611
+
612
+ dims_final = [d for d in dim_set if d in VALID_DIMENSIONS]
613
+
614
+ meta = fact.get("meta", {})
615
+ if not isinstance(meta, dict):
616
+ meta = {}
617
+ meta.update(
618
+ {
619
+ "proposal_id": proposal_id,
620
+ "chunk_index": chunk_index,
621
+ "char_start": start,
622
+ "char_end": end,
623
+ }
624
+ )
625
+
626
+ # ==== 计算 primary_dimension ====
627
+ primary_dim = None
628
+ if dims_final:
629
+ # 简单的优先级规则:尽量按内容来,不行就取第一个
630
+ # 你也可以自己按业务微调优先级
631
+ priority = ["team", "objectives", "strategy", "innovation", "feasibility"]
632
+ # 从 dims_final 里选出优先级最高的那一个
633
+ for d in priority:
634
+ if d in dims_final:
635
+ primary_dim = d
636
+ break
637
+ if primary_dim is None:
638
+ primary_dim = dims_final[0]
639
+ else:
640
+ primary_dim = "feasibility" # 理论上不会走到这里,因为上面兜底了
641
+
642
+ return {
643
+ "text": text.strip(),
644
+ "dimensions": dims_final,
645
+ "type": type_val,
646
+ "primary_dimension": primary_dim,
647
+ "meta": meta,
648
+ }
649
+
650
+ def run_extract(proposal_id: str, max_chars: int = 1800, overlap: int = 400):
651
+ semantic_units = load_semantic_units(proposal_id)
652
+ if semantic_units:
653
+ chunks = make_semantic_chunks(semantic_units, max_chars=max(max_chars, 2200))
654
+ print(f"[INFO] Stage 1 将优先使用 page_semantics 作为 facts 抽取输入")
655
+ else:
656
+ full_text = load_full_text(proposal_id)
657
+ chunks = make_chunks(full_text, max_chars=max_chars, overlap=overlap)
658
+
659
+ out_dir = EXTRACTED_DIR / proposal_id
660
+ out_dir.mkdir(parents=True, exist_ok=True)
661
+ out_path = out_dir / "raw_facts.jsonl"
662
+
663
+ total_facts = 0
664
+ # 统计每个维度的 fact 数量,便于 sanity check
665
+ dim_counts = {dim: 0 for dim in VALID_DIMENSIONS}
666
+
667
+ with out_path.open("w", encoding="utf-8") as f_out:
668
+ for ch in chunks:
669
+ idx = ch["index"]
670
+ chunk_len = ch["end"] - ch["start"]
671
+ print(f"\n[INFO] 处理 chunk {idx+1}/{len(chunks)} (chars={chunk_len})...")
672
+
673
+ # 第一次正常抽取
674
+ data = call_llm_for_chunk(ch["text"])
675
+ facts = data.get("facts", [])
676
+ if not isinstance(facts, list):
677
+ facts = []
678
+
679
+ # 如果 chunk 很长,但抽取的 facts 过少,则尝试 dense 模式重跑一次
680
+ if chunk_len >= 1200 and len(facts) < 5:
681
+ print(
682
+ f"[INFO] 当前 chunk 文本较长(chars={chunk_len}),但只抽取到 {len(facts)} 条事实,"
683
+ f"尝试使用 dense 模式重试以提高召回..."
684
+ )
685
+ dense_data = call_llm_for_chunk(ch["text"], dense=True)
686
+ dense_facts = dense_data.get("facts", [])
687
+ if isinstance(dense_facts, list) and len(dense_facts) > len(facts):
688
+ print(
689
+ f"[INFO] dense 模式抽取到 {len(dense_facts)} 条事实(优于原先的 {len(facts)} 条),"
690
+ f"采用 dense 结果。"
691
+ )
692
+ facts = dense_facts
693
+ else:
694
+ print(
695
+ f"[INFO] dense 模式未显著提升(原 {len(facts)} 条,dense={len(dense_facts)} 条),"
696
+ f"保留原始抽取结果。"
697
+ )
698
+
699
+ normalized_list = []
700
+ for fact in facts:
701
+ if not isinstance(fact, dict):
702
+ continue
703
+
704
+ # 先基于 chunk_text 标记 suspect_numeric
705
+ fact = mark_numeric_suspect(fact, ch["text"])
706
+
707
+ norm = normalize_fact(
708
+ fact,
709
+ proposal_id=proposal_id,
710
+ chunk_index=idx,
711
+ start=ch.get("start", 0),
712
+ end=ch.get("end", len(ch.get("text", ""))),
713
+ )
714
+ norm.setdefault("meta", {})["source"] = ch.get("source", "full_text")
715
+ if ch.get("page_indices"):
716
+ norm.setdefault("meta", {})["page_indices"] = ch.get("page_indices")
717
+
718
+ # 过滤掉空文本
719
+ if norm["text"]:
720
+ normalized_list.append(norm)
721
+
722
+ for fact in normalized_list:
723
+ f_out.write(json.dumps(fact, ensure_ascii=False) + "\n")
724
+ total_facts += 1
725
+ # 更新维度计数
726
+ for d in fact.get("dimensions", []):
727
+ if d in dim_counts:
728
+ dim_counts[d] += 1
729
+
730
+ print(
731
+ f"[INFO] 该 chunk 最终写入 {len(normalized_list)} 条事实,"
732
+ f"目前累计 {total_facts} 条。"
733
+ )
734
+
735
+ print(f"\n[OK] 已写出事实文件: {out_path} (总事实数={total_facts})")
736
+
737
+ # ===== 全局维度分布检查 =====
738
+ print("\n[SUMMARY] 维度分布统计(基于 raw_facts.jsonl):")
739
+ for dim in VALID_DIMENSIONS:
740
+ print(f" - {dim}: {dim_counts[dim]} facts")
741
+
742
+ # 简单 sanity check:如果某个维度明显偏少,打个警告(这里只做提示,不终止)
743
+ if total_facts > 0:
744
+ avg = total_facts / len(VALID_DIMENSIONS)
745
+ for dim in VALID_DIMENSIONS:
746
+ if dim_counts[dim] < max(8, 0.25 * avg):
747
+ print(
748
+ f"[WARN] 维度 {dim} 的事实数仅 {dim_counts[dim]},"
749
+ f"显著低于平均值 {avg:.1f},可能存在抽取不足或映射偏差,"
750
+ f"建议检查 raw_facts.jsonl 或适当调整 Prompt/映射。"
751
+ )
752
+
753
+
754
+ def main():
755
+ parser = argparse.ArgumentParser(
756
+ description="Stage 1: 按 chunk 抽取原子事实(raw_facts.jsonl)"
757
+ )
758
+ parser.add_argument(
759
+ "--proposal_id",
760
+ required=False,
761
+ help="提案 ID(对应 src/data/prepared/<proposal_id>)",
762
+ )
763
+ parser.add_argument(
764
+ "--max_chars",
765
+ type=int,
766
+ default=1800,
767
+ help="每个 chunk 最大字符数(默认 1800)",
768
+ )
769
+ parser.add_argument(
770
+ "--overlap",
771
+ type=int,
772
+ default=400,
773
+ help="chunk 之间的字符重叠数(默认 400)",
774
+ )
775
+ args = parser.parse_args()
776
+
777
+ if args.proposal_id:
778
+ pid = args.proposal_id
779
+ else:
780
+ pid = find_latest_prepared_proposal()
781
+
782
+ run_extract(pid, max_chars=args.max_chars, overlap=args.overlap)
783
+
784
+
785
+ if __name__ == "__main__":
786
+ main()
src/tools/fusion_search.py ADDED
@@ -0,0 +1,326 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ 阶段 3:Fusion Search(v2025.12 ProClean · Safe & Adaptive + Domain Hygiene)
4
+ 保持输出:src/data/fused_evidence/{proposal_id}/{dimension}_fused.json 等
5
+ """
6
+ import os
7
+ # 并行/线程稳定化
8
+ os.environ.setdefault("TOKENIZERS_PARALLELISM","false")
9
+ os.environ.setdefault("OMP_NUM_THREADS","1")
10
+ os.environ.setdefault("OPENBLAS_NUM_THREADS","1")
11
+ os.environ.setdefault("MKL_NUM_THREADS","1")
12
+ os.environ.setdefault("VECLIB_MAXIMUM_THREADS","1")
13
+ os.environ.setdefault("NUMEXPR_NUM_THREADS","1")
14
+ os.environ.setdefault("LOKY_MAX_CPU_COUNT","1")
15
+
16
+ import sys, re, json, numpy as np
17
+ from pathlib import Path
18
+ from datetime import datetime
19
+ from collections import defaultdict, Counter
20
+ from concurrent.futures import ThreadPoolExecutor, as_completed
21
+ from dotenv import load_dotenv
22
+ load_dotenv()
23
+
24
+ CURRENT_DIR = Path(__file__).resolve().parent
25
+ SRC_ROOT = CURRENT_DIR.parent
26
+ if str(SRC_ROOT) not in sys.path: sys.path.insert(0, str(SRC_ROOT))
27
+
28
+ from sentence_transformers import SentenceTransformer, util
29
+ from backend.utils.model_selector import get_llm_client
30
+
31
+ DATA_DIR = SRC_ROOT / "data"
32
+ EVIDENCE_DIR = DATA_DIR / "evidence"
33
+ OUTPUT_DIR = DATA_DIR / "fused_evidence"
34
+ for d in [EVIDENCE_DIR, OUTPUT_DIR]: d.mkdir(parents=True, exist_ok=True)
35
+
36
+ # ==== 同步上游:优先使用环境变量指定的 proposal_id(与 search_by_dimension / web_search 一致) ====
37
+ _env_pid = os.getenv("CURRENT_PROPOSAL_ID", "").strip()
38
+ if _env_pid and (EVIDENCE_DIR / _env_pid).exists():
39
+ proposal_dir = EVIDENCE_DIR / _env_pid
40
+ else:
41
+ subdirs = [d for d in EVIDENCE_DIR.iterdir() if d.is_dir()]
42
+ if not subdirs:
43
+ raise FileNotFoundError("❌ 未找到 evidence 子目录,请先运行 search_by_dimension.py")
44
+ proposal_dir = max(subdirs, key=lambda d: d.stat().st_mtime)
45
+
46
+ proposal_id = proposal_dir.name
47
+ print(f"📂 当前融合目标提案: {proposal_id}")
48
+
49
+ FUSION_DIR = OUTPUT_DIR / proposal_id
50
+ FUSION_DIR.mkdir(parents=True, exist_ok=True)
51
+
52
+ # ==== 同步上游:学术/监管白名单加入 NMPA(与 web_search 对齐) ====
53
+ ACADEMIC_DOMAINS = [
54
+ "pubmed.ncbi.nlm.nih.gov","pmc.ncbi.nlm.nih.gov","nature.com","sciencedirect.com",
55
+ "nih.gov","who.int","ema.europa.eu","fda.gov","clinicaltrials.gov",
56
+ "thelancet.com","bmj.com","cell.com","springer.com","biorxiv.org","medrxiv.org","arxiv.org","nejm.org",
57
+ "nmpa.gov.cn"
58
+ ]
59
+ INSTITUTIONAL_HINTS = (".edu",".ac.","university","hospital")
60
+
61
+ DOMAIN_BLACKLIST = (
62
+ "sol-war.ru","moomoo.com","money.finance.","islandenergy.je","xmind.com","scribd.com",
63
+ "pinterest.","medium.com","reddit.","bilibili.","zhihu.","weibo.","press","news","careers","recruit"
64
+ )
65
+
66
+ def _is_whitelisted_domain(host: str) -> bool:
67
+ h = (host or "").lower()
68
+ return any(h.endswith(d) for d in ACADEMIC_DOMAINS) or any(x in h for x in INSTITUTIONAL_HINTS)
69
+
70
+ EMBED_MODEL = "BAAI/bge-m3"
71
+ print(f"🧠 加载嵌入模型(CPU, safe):{EMBED_MODEL}")
72
+ embedder = SentenceTransformer(EMBED_MODEL, device="cpu")
73
+ _EMB_KW = dict(normalize_embeddings=True, batch_size=16, show_progress_bar=False, convert_to_numpy=True)
74
+
75
+ llm = get_llm_client()
76
+ llm_client = llm["client"]; llm_model = llm["model_name"]; provider = llm["provider"]
77
+ print(f"💬 使用 {provider.upper()} 模型:{llm_model}")
78
+
79
+ def clean_text(text: str) -> str:
80
+ text = re.sub(r"\s+", " ", text)
81
+ text = re.sub(r"http\S+", "", text)
82
+ return text.strip()
83
+
84
+ def extract_domain(url: str) -> str:
85
+ try:
86
+ return re.sub(r"^www\.", "", re.search(r"https?://([^/]+)/?", url).group(1))
87
+ except Exception:
88
+ return "unknown"
89
+
90
+ def load_evidence_files(proposal_path: Path):
91
+ dim_map = defaultdict(list); total_loaded = 0
92
+ for f in proposal_path.glob("*_combined.json"):
93
+ dimension = f.stem.replace("_combined","")
94
+ try:
95
+ data = json.loads(f.read_text(encoding="utf-8")) or []
96
+ for item in data:
97
+ txt = clean_text(item.get("text",""))
98
+ url = item.get("url",""); conf = float(item.get("confidence",0.5))
99
+ if len(txt) > 100 and url:
100
+ dom = extract_domain(url)
101
+ # 兜底黑名单
102
+ if any(bad in dom for bad in DOMAIN_BLACKLIST): continue
103
+ # 主页剔除(除白名单域/文本极长)
104
+ if re.match(r"^https?://[^/]+/?$", url) and (not _is_whitelisted_domain(dom)) and len(txt) < 800:
105
+ continue
106
+ dim_map[dimension].append({"url": url,"domain": dom,"text": txt,"confidence": conf})
107
+ print(f"✅ {dimension} 加载 {len(dim_map[dimension])} 条 combined。")
108
+ total_loaded += len(dim_map[dimension])
109
+ except Exception as e:
110
+ print(f"⚠️ 读取 {f.name} 出错: {e}")
111
+
112
+ if not dim_map:
113
+ print("⚠️ 未检测到 combined,尝试 *_cache.json。")
114
+ for f in proposal_path.glob("*_cache.json"):
115
+ dimension = f.stem.replace("_cache","")
116
+ try:
117
+ cache_data = json.loads(f.read_text(encoding="utf-8"))
118
+ all_items = []
119
+ if isinstance(cache_data, dict):
120
+ for v in cache_data.values(): all_items.extend(v)
121
+ else:
122
+ all_items = cache_data
123
+ for item in all_items:
124
+ txt = clean_text(item.get("text",""))
125
+ url = item.get("url",""); conf = float(item.get("confidence",0.5))
126
+ if len(txt) > 100 and url:
127
+ dom = extract_domain(url)
128
+ if any(bad in dom for bad in DOMAIN_BLACKLIST): continue
129
+ if re.match(r"^https?://[^/]+/?$", url) and (not _is_whitelisted_domain(dom)) and len(txt) < 800:
130
+ continue
131
+ dim_map[dimension].append({"url": url,"domain": dom,"text": txt,"confidence": conf})
132
+ print(f"🟡 {dimension} 使用 cache 加载 {len(dim_map[dimension])} 条。")
133
+ total_loaded += len(dim_map[dimension])
134
+ except Exception as e:
135
+ print(f"⚠️ 读取 {f.name} 出错: {e}")
136
+
137
+ print(f"\n📊 共加载 {len(dim_map)} 个维度,总 evidence:{total_loaded}")
138
+ return dim_map
139
+
140
+ def llm_chat(prompt: str, temperature: float = 0.35) -> str:
141
+ try:
142
+ resp = llm_client.chat.completions.create(
143
+ model=llm_model, messages=[{"role":"user","content":prompt}], temperature=temperature
144
+ )
145
+ return resp.choices[0].message.content.strip()
146
+ except Exception as e:
147
+ print(f"⚠️ LLM 摘要失败: {e}")
148
+ return ""
149
+
150
+ def summarize_with_llm(dimension: str, fused_blocks: list) -> str:
151
+ if not fused_blocks: return "摘要生成失败。"
152
+ ranked = sorted(fused_blocks, key=lambda b: (b.get("avg_conf",0.0), len(b.get("text",""))), reverse=True)[:6]
153
+ joined = "\n\n".join([b["text"] for b in ranked])[:9000]
154
+ # 引用集合(≥70% 学术/监管域)
155
+ refs = []
156
+ for i,b in enumerate(ranked,1):
157
+ for u in b.get("urls",[])[:2]:
158
+ refs.append((i,u))
159
+ # 保证学术占比
160
+ def _is_academic(u: str) -> bool:
161
+ host = extract_domain(u)
162
+ return any(host.endswith(d) for d in ACADEMIC_DOMAINS)
163
+ if refs:
164
+ ac_ratio = sum(1 for _,u in refs if _is_academic(u)) / len(refs)
165
+ if ac_ratio < 0.7:
166
+ # 学术不足:只取学术/监管域引用
167
+ refs = [(i,u) for i,u in refs if _is_academic(u)]
168
+ refs_text = "\n".join([f"[{i}] {u}" for i,u in refs[:6]])
169
+
170
+ prompt = f"""
171
+ 你是通用项目评审专家,请基于以下高置信整合内容为“{dimension}”写客观摘要。
172
+ 要求:事实化;引用以 [1][2] 标注;涉及年份、标准、政策或关键事实时必须来自可靠来源;180~250字。
173
+ 【内容】{joined}
174
+ 【来源】{refs_text}
175
+ Summary Quality: 请在末尾给出 高/中/低。
176
+ """
177
+ return llm_chat(prompt) or "摘要生成失败。"
178
+
179
+ def representative_score(doc_len: int, conf: float, mean_len: float, domain: str = "") -> float:
180
+ len_norm = min(doc_len / max(mean_len, 1.0), 2.0)
181
+ bonus = 0.05 if _is_whitelisted_domain(domain) else 0.0
182
+ return 0.7 * float(conf + bonus) + 0.3 * float(len_norm)
183
+
184
+ def _domain_penalty(domain: str) -> float:
185
+ d = (domain or "").lower()
186
+ if any(x in d for x in ["news","blog","medium.com"]): return 0.05
187
+ return 0.0
188
+
189
+ def _is_homepage(url: str) -> bool:
190
+ return bool(re.match(r"^https?://[^/]+/?$", url or ""))
191
+
192
+ def greedy_grouping(embs: np.ndarray, texts: list, threshold: float):
193
+ if len(embs) == 0: return []
194
+ clusters, centers = [], []
195
+ for idx, v in enumerate(embs):
196
+ placed = False
197
+ for ci, c in enumerate(centers):
198
+ if float(np.dot(v, c)) >= threshold:
199
+ clusters[ci].append(idx)
200
+ new_center = np.mean([embs[i] for i in clusters[ci]], axis=0)
201
+ new_center = new_center / (np.linalg.norm(new_center) + 1e-12)
202
+ centers[ci] = new_center; placed = True; break
203
+ if not placed:
204
+ clusters.append([idx]); centers.append(v)
205
+ return clusters
206
+
207
+ def fuse_dimension(dimension: str, docs: list):
208
+ if not docs:
209
+ print(f"⚠️ {dimension} 无 evidence"); return None, None
210
+
211
+ print(f"\n🔹 融合维度: {dimension}({len(docs)} 条)")
212
+ corpus = [d["text"] for d in docs]
213
+ urls_all = [d["url"] for d in docs]
214
+ domains_all = [d["domain"] for d in docs]
215
+ conf_all = [float(d["confidence"]) for d in docs]
216
+ lens_all = [len(t) for t in corpus]
217
+ mean_len = float(np.mean(lens_all)) if lens_all else 400.0
218
+ doc_n = len(docs)
219
+
220
+ if mean_len > 800 and doc_n >= 12: threshold, min_k = 0.72, 3
221
+ elif mean_len < 300 or doc_n < 6: threshold, min_k = 0.58, 2
222
+ else: threshold, min_k = 0.65, 3
223
+
224
+ embeddings = embedder.encode(corpus, **_EMB_KW)
225
+ try:
226
+ clusters = util.community_detection(embeddings, threshold=threshold, min_community_size=min_k)
227
+ except Exception as e:
228
+ print(f"⚠️ community_detection 异常,改用贪心聚类:{e}")
229
+ clusters = [c for c in greedy_grouping(embeddings, corpus, threshold) if len(c) >= min_k]
230
+
231
+ print(f"🧩 聚类 {len(clusters)} 个(阈 {threshold},min_k={min_k},均长 {int(mean_len)})")
232
+
233
+ fused_blocks, used = [], set()
234
+ for cluster in clusters:
235
+ cluster_docs = [docs[i] for i in cluster]
236
+ filtered = [d for d in cluster_docs if not _is_homepage(d["url"])]
237
+ ranked = sorted((filtered or cluster_docs),
238
+ key=lambda x: representative_score(len(x["text"]), x["confidence"] - _domain_penalty(x["domain"]), mean_len, x["domain"]),
239
+ reverse=True)
240
+ texts = [cd["text"] for cd in ranked]
241
+ urls = [cd["url"] for cd in ranked]
242
+ confs = np.array([max(float(cd["confidence"]) - _domain_penalty(cd["domain"]), 0.0) for cd in ranked])
243
+ w = confs / (confs.sum() + 1e-6)
244
+ avg_conf = float((w * confs).sum())
245
+ combined = " ".join(sorted(texts, key=len, reverse=True)[:3])[:2000]
246
+ fused_blocks.append({"text": combined, "urls": urls, "avg_conf": round(avg_conf, 2)})
247
+ for i in cluster: used.add(i)
248
+
249
+ isolated_idx = [i for i in range(len(corpus)) if i not in used]
250
+ isolated_entries = [{"text": corpus[i], "url": urls_all[i], "confidence": conf_all[i], "domain": domains_all[i]} for i in isolated_idx]
251
+
252
+ pre_summary = summarize_with_llm(dimension, fused_blocks)
253
+ if isolated_entries and pre_summary:
254
+ embs_iso = embedder.encode([e["text"] for e in isolated_entries], **_EMB_KW)
255
+ seed = embedder.encode([pre_summary[:1200]], **_EMB_KW)[0]
256
+ sims = np.dot(embs_iso, seed)
257
+ order = np.argsort(sims)[::-1][: min(5, len(sims))]
258
+ for j in order:
259
+ e = isolated_entries[int(j)]
260
+ fused_blocks.append({"text": e["text"][:1500], "urls": [e["url"]], "avg_conf": round(float(e.get("confidence",0.6)),2)})
261
+
262
+ confs_global = np.array(conf_all, dtype=float)
263
+ w_global = confs_global / (confs_global.sum() + 1e-6)
264
+ avg_conf_weighted = float((w_global * confs_global).sum())
265
+ final_summary = summarize_with_llm(dimension, fused_blocks)
266
+
267
+ high_blocks = [b for b in fused_blocks if b.get("avg_conf",0) >= 0.80]
268
+ mid_blocks = [b for b in fused_blocks if 0.60 <= b.get("avg_conf",0) < 0.80]
269
+ low_blocks = [b for b in fused_blocks if b.get("avg_conf",0) < 0.60]
270
+
271
+ out_path = FUSION_DIR / f"{dimension}_fused.json"
272
+ result = {
273
+ "dimension": dimension, "fused_texts": fused_blocks, "summary": final_summary,
274
+ "cluster_count": len(clusters), "doc_count": len(docs), "isolated_count": len(isolated_idx),
275
+ "avg_confidence_weighted": round(avg_conf_weighted, 2),
276
+ "confidence_distribution": {"high": len(high_blocks), "mid": len(mid_blocks), "low": len(low_blocks)},
277
+ "avg_text_len": int(np.mean([len(b['text']) for b in fused_blocks])) if fused_blocks else 0,
278
+ "threshold": threshold, "min_community_size": min_k,
279
+ "embedding_model": EMBED_MODEL, "timestamp": datetime.now().strftime("%Y-%m-%d %H:%M:%S")
280
+ }
281
+ out_path.write_text(json.dumps(result, ensure_ascii=False, indent=2), encoding="utf-8")
282
+
283
+ top_domains = Counter(domains_all).most_common(10)
284
+ print(f"✅ {dimension} 融合完成 → {out_path}")
285
+ print(f"📊 聚类: {len(clusters)} | 加权均置信: {result['avg_confidence_weighted']} | 分层 H/M/L: {len(high_blocks)}/{len(mid_blocks)}/{len(low_blocks)} | Top域: {top_domains[:3]}")
286
+ return result, top_domains
287
+
288
+ if __name__ == "__main__":
289
+ print("🚀 启动 Fusion Search(Safe) ...")
290
+ dim_docs = load_evidence_files(proposal_dir)
291
+ if not dim_docs:
292
+ print("❌ 无 evidence"); sys.exit(0)
293
+
294
+ fusion_index = {}; all_domains_global = Counter()
295
+ with ThreadPoolExecutor(max_workers=3) as ex:
296
+ futures = {ex.submit(fuse_dimension, dim, docs): dim for dim, docs in dim_docs.items()}
297
+ for f in as_completed(futures):
298
+ dim = futures[f]
299
+ try:
300
+ result, top_domains = f.result()
301
+ if result:
302
+ for d, c in (top_domains or []): all_domains_global[d] += c
303
+ fusion_index[dim] = {
304
+ "fused_file": str((FUSION_DIR / f"{dim}_fused.json").resolve()),
305
+ "cluster_count": result["cluster_count"], "isolated_count": result["isolated_count"],
306
+ "avg_confidence_weighted": result["avg_confidence_weighted"],
307
+ "confidence_distribution": result["confidence_distribution"],
308
+ "avg_text_len": result["avg_text_len"], "threshold": result["threshold"],
309
+ "min_community_size": result["min_community_size"], "embedding_model": result["embedding_model"],
310
+ "evidence_count": result["doc_count"], "top_domains": top_domains,
311
+ "summary_preview": (result["summary"] or "")[:150]
312
+ }
313
+ except Exception as e:
314
+ print(f"❌ {dim} 融合失败: {e}")
315
+
316
+ (FUSION_DIR / "fusion_report.json").write_text(json.dumps({
317
+ "proposal_id": proposal_id, "fusion_time": datetime.now().strftime("%Y-%m-%d %H:%M:%S"),
318
+ "dimension_stats": fusion_index, "top_domains_global": all_domains_global.most_common(15)
319
+ }, ensure_ascii=False, indent=2), encoding="utf-8")
320
+ (FUSION_DIR / "dimension_fusion_index.json").write_text(json.dumps(fusion_index, ensure_ascii=False, indent=2), encoding="utf-8")
321
+
322
+ total_dims = len(fusion_index); total_docs = sum(v["evidence_count"] for v in fusion_index.values())
323
+ print("\n🎯 融合完成(Safe)")
324
+ print(f"📁 输出:{FUSION_DIR}")
325
+ print(f"📊 维度:{total_dims} | 融合 evidence:{total_docs}")
326
+ print(f"🌐 全局 Top 域:{all_domains_global.most_common(10)}")
src/tools/generate_final_report.py ADDED
@@ -0,0 +1,397 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ 阶段 9:Final Report Generator(v3 · 一页摘要 + 专家评审 + 精简问答 · 基于 final_payload)
4
+
5
+ 功能:
6
+ 1) 自动读取:
7
+ - refined_answers/<pid>/postproc/final_payload.json
8
+ - expert_reports/<pid>/ai_expert_opinion.json
9
+ - expert_reports/<pid>/ai_expert_opinion.md
10
+ 2) 生成综合 Markdown 报告:
11
+ - 顶部:项目综合评审报告(含生成时间与说明)
12
+ - 0. 一页摘要(Executive Summary):
13
+ · verdict + 一句话结论依据
14
+ · 综合评分、信心度 + 区间解释
15
+ · Top 3 优势 & Top 3 主要风险
16
+ · 关键补充材料建议
17
+ - 1. AI 专家总体评审(来自 ai_expert_opinion.md,自动降一级标题)
18
+ - 2. 维度问答与打分依据(基于 final_payload):
19
+ · 按五个维度展示
20
+ · 每个维度:综合得分 + 打分依据摘要(rationales)+ 行业通识要点(general_insights)
21
+ · 每道题:问题、选中回答、来源模型、置信度、对齐度、漂移等元信息
22
+ · 每个维度仅展示前 MAX_QA_PER_DIM 条高价值问答(按置信度 + 对齐度排序)
23
+
24
+ 用法:
25
+ cd 到项目根目录(包含 src/)
26
+ python -m src.tools.generate_final_report
27
+ python -m src.tools.generate_final_report --pid Ebovir_LNP
28
+ """
29
+
30
+ import json
31
+ import argparse
32
+ from pathlib import Path
33
+ from datetime import datetime
34
+
35
+ # ========= 路径配置 =========
36
+ BASE_DIR = Path(__file__).resolve().parents[2] # 项目根(包含 src/)
37
+ DATA_DIR = BASE_DIR / "src" / "data"
38
+ REFINED_ROOT = DATA_DIR / "refined_answers"
39
+ EXPERT_ROOT = DATA_DIR / "expert_reports"
40
+ REPORT_ROOT = DATA_DIR / "reports"
41
+
42
+ DIM_ORDER = ["team", "objectives", "strategy", "innovation", "feasibility"]
43
+ DIM_LABELS_ZH = {
44
+ "team": "团队与治理",
45
+ "objectives": "项目目标",
46
+ "strategy": "实施路径与战略",
47
+ "innovation": "技术与产品创新",
48
+ "feasibility": "资源与可行性",
49
+ }
50
+
51
+ # 每个维度最多展示多少条问答(从高置信度 / 高对齐度往下选)
52
+ MAX_QA_PER_DIM = 6
53
+
54
+
55
+ # ========= 工具函数 =========
56
+ def now_str() -> str:
57
+ return datetime.now().strftime("%Y-%m-%d %H:%M:%S")
58
+
59
+
60
+ def load_json(path: Path):
61
+ return json.loads(path.read_text(encoding="utf-8"))
62
+
63
+
64
+ def detect_latest_pid() -> str:
65
+ """
66
+ 从 refined_answers 下自动选一个“最新且有 postproc/final_payload.json + 对应专家评审”的 pid
67
+ """
68
+ if not REFINED_ROOT.exists():
69
+ return ""
70
+ cands = []
71
+ for d in REFINED_ROOT.iterdir():
72
+ if not d.is_dir():
73
+ continue
74
+ postproc_dir = d / "postproc"
75
+ fp_path = postproc_dir / "final_payload.json"
76
+ expert_md_path = EXPERT_ROOT / d.name / "ai_expert_opinion.md"
77
+ if fp_path.exists() and expert_md_path.exists():
78
+ cands.append((d.name, fp_path.stat().st_mtime))
79
+ if not cands:
80
+ return ""
81
+ cands.sort(key=lambda x: x[1], reverse=True)
82
+ return cands[0][0]
83
+
84
+
85
+ def adjust_expert_markdown(md_text: str) -> str:
86
+ """
87
+ 将 ai_expert_opinion.md 的标题全部降一级:
88
+ # → ##
89
+ ## → ###
90
+ ###→ ####
91
+ 避免和顶层报告标题冲突。
92
+ """
93
+ lines = md_text.splitlines()
94
+ adjusted = []
95
+ for line in lines:
96
+ if line.startswith("# "):
97
+ adjusted.append("#" + line) # '# ' → '## '
98
+ elif line.startswith("## "):
99
+ adjusted.append("#" + line) # '## ' → '### '
100
+ elif line.startswith("### "):
101
+ adjusted.append("#" + line) # '### '→ '#### '
102
+ else:
103
+ adjusted.append(line)
104
+ return "\n".join(adjusted)
105
+
106
+
107
+ def _fmt_float(v, ndigits=1, default="0.0"):
108
+ try:
109
+ return f"{float(v):.{ndigits}f}"
110
+ except Exception:
111
+ return default
112
+
113
+
114
+ # ========= 一页摘要生成 =========
115
+ def build_executive_summary(expert_json: dict) -> str:
116
+ """
117
+ 基于 ai_expert_opinion.json 生成“0. 一页摘要(Executive Summary)”Markdown 文本。
118
+ 如果 expert_json 为空,则返回空字符串,调用方自己判断是否插入。
119
+ """
120
+ if not expert_json:
121
+ return ""
122
+
123
+ overall = expert_json.get("overall_opinion", {}) or {}
124
+ score = overall.get("overall_score_echo", 0.0) # 0–1 区间
125
+ conf = overall.get("confidence_echo", 0.0) # 0–1 区间
126
+ verdict = (overall.get("verdict") or "").strip()
127
+ summary = (overall.get("summary") or "").strip()
128
+ key_strengths = overall.get("key_strengths") or []
129
+ key_risks = overall.get("key_risks") or []
130
+ recs = overall.get("recommendations") or []
131
+ basis = overall.get("basis") or []
132
+
133
+ # verdict 补充中文解释
134
+ verdict_zh = ""
135
+ if verdict == "GO":
136
+ verdict_zh = "(建议在可控风险前提下推进)"
137
+ elif verdict == "HOLD":
138
+ verdict_zh = "(建议补充材料后再决策)"
139
+ elif verdict == "NO-GO":
140
+ verdict_zh = "(当前条件下不建议立项)"
141
+
142
+ # 一句话结论依据:优先用 basis[0],其次用 summary
143
+ brief_reason = ""
144
+ if basis:
145
+ brief_reason = basis[0]
146
+ elif summary:
147
+ brief_reason = summary
148
+
149
+ lines = []
150
+ lines.append("## 0. 一页摘要(Executive Summary)")
151
+ lines.append("")
152
+ if verdict:
153
+ lines.append(f"- **总体结论(verdict)**:{verdict} {verdict_zh}".strip())
154
+ else:
155
+ lines.append("- **总体结论(verdict)**:暂无明确结论")
156
+ if brief_reason:
157
+ lines.append(f"- **结论依据简述**:{brief_reason}")
158
+ lines.append(f"- **综合评分**:{_fmt_float(score, 3)}(0–1 区间)")
159
+ lines.append(f"- **信心度**:{_fmt_float(conf, 3)}")
160
+ lines.append("")
161
+ lines.append("> **评分区间说明(供非技术评审参考)**:")
162
+ lines.append("> - ≥ 0.62:整体条件较好,可在控制风险前提下推进;")
163
+ lines.append("> - 0.45–0.62:信息不充分或优劣并存,建议补充材料后再决策;")
164
+ lines.append("> - < 0.45:关键维度存在明显短板或高不确定性,一般不建议立项。")
165
+ lines.append("")
166
+
167
+ # Top 优势 / 风险
168
+ if key_strengths:
169
+ lines.append("**Top 项目优势(按重要性排序,最多列出 3 条)**")
170
+ for s in key_strengths[:3]:
171
+ lines.append(f"- {s}")
172
+ lines.append("")
173
+ if key_risks:
174
+ lines.append("**Top 主要风险 / 不足(最多列出 3 条)**")
175
+ for r in key_risks[:3]:
176
+ lines.append(f"- {r}")
177
+ lines.append("")
178
+
179
+ # 关键补充材料 / 建议:直接从 recommendations 抽几条
180
+ if recs:
181
+ lines.append("**关键后续建议 / 需补充材料要点(节选 3–5 条)**")
182
+ for r in recs[:5]:
183
+ lines.append(f"- {r}")
184
+ lines.append("")
185
+
186
+ return "\n".join(lines)
187
+
188
+
189
+ # ========= 维度问答部分 =========
190
+ def build_qa_section_from_final_payload(final_payload: dict) -> str:
191
+ """
192
+ 输入:postproc/final_payload.json(dict)
193
+ 输出:Markdown 文本(每个维度下:
194
+ 综合得分 + 打分依据摘要 + 行业通识要点 + 精选问题/回答明细)
195
+ - 每个维度只展示前 MAX_QA_PER_DIM 条问答
196
+ (按 confidence + alignment 排序,从高到低截断)
197
+ - 逐题不再重复展示 general_insights(避免和维度级/专家意见重复啰嗦)
198
+ """
199
+ dims = final_payload.get("dimensions", {}) or {}
200
+
201
+ lines = []
202
+ lines.append("## 2. 维度问答与打分依据")
203
+ lines.append("")
204
+ lines.append("> 本部分基于 post_processing 选中的问答结果生成,用于支撑专家评审结论的溯源。")
205
+ lines.append("> 如阅读时间有限,可主要关注「0. 一页摘要」与「1. AI 专家总体评审」,本部分主要面向技术评审与审计。")
206
+ lines.append("")
207
+
208
+ for dim in DIM_ORDER:
209
+ block = dims.get(dim) or {}
210
+ qas = block.get("qas") or []
211
+ if not qas:
212
+ continue
213
+
214
+ # 按置信度 + 对齐度排序,截断为前 MAX_QA_PER_DIM 条
215
+ def _qa_key(qa_item):
216
+ try:
217
+ conf = float(qa_item.get("confidence") or 0.0)
218
+ except Exception:
219
+ conf = 0.0
220
+ try:
221
+ align = float(qa_item.get("alignment") or 0.0)
222
+ except Exception:
223
+ align = 0.0
224
+ return (conf + align)
225
+
226
+ qas_sorted = sorted(qas, key=_qa_key, reverse=True)
227
+ qas_sorted = qas_sorted[:MAX_QA_PER_DIM]
228
+
229
+ zh_label = DIM_LABELS_ZH.get(dim, dim)
230
+ score = block.get("score", 0.0) # 已是 0–100 之间的小数
231
+ rationales = block.get("rationales") or []
232
+ gi_dim = block.get("general_insights") or []
233
+
234
+ # 维度标题 + 分数
235
+ lines.append(f"### 维度:{zh_label}({dim}) · 综合得分:{_fmt_float(score, 1)} / 100")
236
+ lines.append("")
237
+
238
+ # 打分依据(来源于 post_processing 的 strengths 聚合)
239
+ if rationales:
240
+ lines.append("**打分依据(摘要)**")
241
+ for r in rationales:
242
+ r = str(r).strip()
243
+ if r:
244
+ lines.append(f"- {r}")
245
+ lines.append("")
246
+
247
+ # 维度级 industry general insights(来自 general_insights 聚合层)
248
+ if gi_dim:
249
+ lines.append("**行业通识要点(不代表本项目已达成,仅作参照)**")
250
+ for g in gi_dim[:8]:
251
+ g = str(g).strip()
252
+ if g:
253
+ lines.append(f"- {g}")
254
+ lines.append("")
255
+
256
+ # 逐题问答(已按重要性排序 + 截断)
257
+ for idx, qa in enumerate(qas_sorted, start=1):
258
+ q_text = (qa.get("q") or "").strip()
259
+ ans = (qa.get("answer") or "").strip()
260
+ provider = (qa.get("provider") or "").strip()
261
+ model = (qa.get("model") or "").strip()
262
+ conf = qa.get("confidence", 0.0)
263
+ align = qa.get("alignment", 0.0)
264
+ drift = qa.get("dimension_drift", 0.0)
265
+ claims = qa.get("claims") or []
266
+ evids = qa.get("evidence_hints") or []
267
+
268
+ lines.append("")
269
+ lines.append(f"#### Q{idx}. {q_text}")
270
+ lines.append("")
271
+ meta_parts = []
272
+ if provider:
273
+ meta_parts.append(f"来源模型:{provider}")
274
+ if model:
275
+ meta_parts.append(model)
276
+ meta_parts.append(f"置信度:{_fmt_float(conf, 2)}")
277
+ meta_parts.append(f"对齐度:{_fmt_float(align, 2)}")
278
+ meta_parts.append(f"维度漂移:{_fmt_float(drift, 2)}")
279
+
280
+ lines.append("_" + " | ".join(meta_parts) + "_")
281
+ lines.append("")
282
+
283
+ # 核心要点(claims)
284
+ if claims:
285
+ lines.append("**核心要点(claims)**")
286
+ for c in claims:
287
+ c = str(c).strip()
288
+ if c:
289
+ lines.append(f"- {c}")
290
+ lines.append("")
291
+
292
+ # 选中回答
293
+ lines.append("**选中回答**")
294
+ lines.append("")
295
+ if ans:
296
+ lines.append(ans)
297
+ else:
298
+ lines.append("_(无有效回答)_")
299
+ lines.append("")
300
+
301
+ # 证据线索(evidence_hints)
302
+ if evids:
303
+ lines.append("**证据线索(evidence_hints)**")
304
+ for e in evids:
305
+ e = str(e).strip()
306
+ if e:
307
+ lines.append(f"- {e}")
308
+ lines.append("")
309
+
310
+ # ⚠️ 不再展示逐题 general_insights,避免与维度级 / 专家意见重复啰嗦
311
+ # gi_q = qa.get("general_insights") or []
312
+ # if gi_q:
313
+ # lines.append("**行业通识经验(general_insights)**")
314
+ # for g in gi_q[:6]:
315
+ # g = str(g).strip()
316
+ # if g:
317
+ # lines.append(f"- {g}")
318
+ # lines.append("")
319
+
320
+ return "\n".join(lines)
321
+
322
+
323
+ # ========= 主流程 =========
324
+ def main():
325
+ ap = argparse.ArgumentParser(description="Generate final integrated markdown report (executive summary + expert opinion + Q&A, v3).")
326
+ ap.add_argument("--pid", type=str, default="", help="提案 ID;若缺省则自动检测最新的一个。")
327
+ args = ap.parse_args()
328
+
329
+ pid = args.pid.strip() or detect_latest_pid()
330
+ if not pid:
331
+ raise RuntimeError(
332
+ "未检测到可用项目:需要至少存在 "
333
+ "`refined_answers/<pid>/postproc/final_payload.json` 与 "
334
+ "`expert_reports/<pid>/ai_expert_opinion.md`。"
335
+ )
336
+
337
+ # ---------- 路径 ----------
338
+ refined_dir = REFINED_ROOT / pid
339
+ postproc_dir = refined_dir / "postproc"
340
+ fp_path = postproc_dir / "final_payload.json"
341
+
342
+ expert_dir = EXPERT_ROOT / pid
343
+ expert_md_path = expert_dir / "ai_expert_opinion.md"
344
+ expert_json_path = expert_dir / "ai_expert_opinion.json"
345
+
346
+ if not fp_path.exists():
347
+ raise FileNotFoundError(f"未找到 final_payload.json:{fp_path}")
348
+ if not expert_md_path.exists():
349
+ raise FileNotFoundError(f"未找到专家评审 Markdown:{expert_md_path}")
350
+ # ai_expert_opinion.json 缺失时不会报错,只是无法生成一页摘要
351
+ expert_json = {}
352
+ if expert_json_path.exists():
353
+ expert_json = load_json(expert_json_path)
354
+
355
+ REPORT_ROOT.mkdir(parents=True, exist_ok=True)
356
+ out_path = REPORT_ROOT / f"{pid}_final_report.md"
357
+
358
+ # ---------- 读取数据 ----------
359
+ final_payload = load_json(fp_path)
360
+ expert_md_raw = expert_md_path.read_text(encoding="utf-8")
361
+
362
+ # ---------- 组装报告 ----------
363
+ report_lines = []
364
+
365
+ # 顶层标题
366
+ report_lines.append(f"# 项目综合评审报告 · {pid}")
367
+ report_lines.append("")
368
+ report_lines.append(f"_生成时间:{now_str()}_")
369
+ report_lines.append("")
370
+ report_lines.append("_本报告由 AI 辅助评审系统自动生成,供内部专家和决策委员会参考使用。_")
371
+ report_lines.append("")
372
+
373
+ # 0. 一页摘要(如果有 ai_expert_opinion.json)
374
+ exec_summary_md = build_executive_summary(expert_json)
375
+ if exec_summary_md:
376
+ report_lines.append(exec_summary_md)
377
+ report_lines.append("")
378
+
379
+ # 1. AI 专家总体评审(来自 ai_expert_opinion.md)
380
+ report_lines.append("## 1. AI 专家总体评审(详细版)")
381
+ report_lines.append("")
382
+ report_lines.append(adjust_expert_markdown(expert_md_raw))
383
+ report_lines.append("")
384
+
385
+ # 2. 维度问答与打分依据(来自 final_payload.json 的问题 + 选中的一个回答)
386
+ qa_section = build_qa_section_from_final_payload(final_payload)
387
+ report_lines.append(qa_section)
388
+ report_lines.append("")
389
+
390
+ final_md = "\n".join(report_lines)
391
+ out_path.write_text(final_md, encoding="utf-8")
392
+
393
+ print(f"✅ Final report generated -> {out_path}")
394
+
395
+
396
+ if __name__ == "__main__":
397
+ main()
src/tools/generate_questions.py ADDED
@@ -0,0 +1,173 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ Stage 3 · Domain-adaptive question generation
4
+
5
+ Design goals:
6
+ - no domain-specific prompt text inside this stage
7
+ - deterministic question generation from:
8
+ universal template + domain_profile.json + fixed task registry
9
+ - keep downstream compatibility with llm_answering.py
10
+ """
11
+
12
+ import json
13
+ import argparse
14
+ from pathlib import Path
15
+ from datetime import datetime
16
+ from typing import Dict, Any, List
17
+
18
+ from src.prompting.domain_adaptive import (
19
+ REVIEW_TASKS,
20
+ QUESTION_SEARCH_HINTS,
21
+ build_specialized_question,
22
+ load_domain_profile,
23
+ )
24
+
25
+ BASE_DIR = Path(__file__).resolve().parents[2]
26
+ DATA_DIR = BASE_DIR / "src" / "data"
27
+ EXTRACTED_DIR = DATA_DIR / "extracted"
28
+ QUESTIONS_DIR = DATA_DIR / "questions"
29
+ CONFIG_QS_DIR = DATA_DIR / "config" / "question_sets"
30
+
31
+ DIMENSION_NAMES = ["team", "objectives", "strategy", "innovation", "feasibility"]
32
+ QUESTIONS_DIR.mkdir(parents=True, exist_ok=True)
33
+ CONFIG_QS_DIR.mkdir(parents=True, exist_ok=True)
34
+
35
+
36
+ def now_str() -> str:
37
+ return datetime.now().strftime("%Y-%m-%d %H:%M:%S")
38
+
39
+
40
+ def find_latest_extracted_proposal_id() -> str:
41
+ cands = [d for d in EXTRACTED_DIR.iterdir() if d.is_dir()] if EXTRACTED_DIR.exists() else []
42
+ if not cands:
43
+ raise FileNotFoundError("No proposal found under src/data/extracted")
44
+ cands.sort(key=lambda p: p.stat().st_mtime, reverse=True)
45
+ pid = cands[0].name
46
+ print(f"[INFO] [auto] selected latest proposal_id: {pid}")
47
+ return pid
48
+
49
+
50
+ def load_dimensions(proposal_id: str) -> Dict[str, Any]:
51
+ path = EXTRACTED_DIR / proposal_id / "dimensions_v2.json"
52
+ if not path.exists():
53
+ raise FileNotFoundError(f"dimensions_v2.json not found: {path}")
54
+ return json.loads(path.read_text(encoding="utf-8"))
55
+
56
+
57
+ def _dimension_summary(dimensions: Dict[str, Any], dim_name: str) -> Dict[str, Any]:
58
+ block = dimensions.get(dim_name, {}) if isinstance(dimensions, dict) else {}
59
+ if not isinstance(block, dict):
60
+ block = {}
61
+ return {
62
+ "summary": block.get("summary", ""),
63
+ "key_points": block.get("key_points", []) or [],
64
+ "risks": block.get("risks", []) or [],
65
+ "mitigations": block.get("mitigations", []) or [],
66
+ "meta": block.get("meta", {}) or {},
67
+ }
68
+
69
+
70
+ def _write_json(path: Path, payload: Dict[str, Any]) -> None:
71
+ path.write_text(json.dumps(payload, ensure_ascii=False, indent=2), encoding="utf-8")
72
+
73
+
74
+ def _write_json_with_log(path: Path, payload: Dict[str, Any], label: str) -> None:
75
+ _write_json(path, payload)
76
+ print(f"✅ {label} -> {path}")
77
+
78
+
79
+ def build_question_record(task, profile: Dict[str, Any], dim_summary: Dict[str, Any], local_index: int) -> Dict[str, Any]:
80
+ question_zh = build_specialized_question(task, profile)
81
+ question_en = str(question_zh)
82
+
83
+ links_to = {
84
+ "key_points": list(range(min(3, len(dim_summary.get("key_points", []))))),
85
+ "risks": list(range(min(2, len(dim_summary.get("risks", []))))),
86
+ "mitigations": list(range(min(2, len(dim_summary.get("mitigations", []))))),
87
+ }
88
+
89
+ rating_tasks = {"team", "feasibility", "innovation", "outcomes", "objectives"}
90
+ analysis_tasks = {"methods", "risks", "evidence", "problem"}
91
+ answer_type = "rating" if task.task_id in rating_tasks else "analysis"
92
+
93
+ return {
94
+ "qid": f"{task.dimension}_Q{local_index}",
95
+ "task_id": task.task_id,
96
+ "template_id": task.template_id,
97
+ "dimension": task.dimension,
98
+ "aspect": task.task_id,
99
+ "title": task.title,
100
+ "question_zh": question_zh,
101
+ "question_en": question_en,
102
+ "answer_type": answer_type,
103
+ "priority": 1,
104
+ "links_to": links_to,
105
+ }
106
+
107
+
108
+ def run_generate_questions(proposal_id: str):
109
+ dimensions = load_dimensions(proposal_id)
110
+ profile = load_domain_profile(EXTRACTED_DIR / proposal_id / "domain_profile.json")
111
+
112
+ per_dim_counter = {dim: 0 for dim in DIMENSION_NAMES}
113
+
114
+ all_dim_questions: Dict[str, Any] = {
115
+ dim: {
116
+ "dimension": dim,
117
+ "questions": [],
118
+ "search_hints": QUESTION_SEARCH_HINTS.get(dim, []),
119
+ "source_proposal_id": proposal_id,
120
+ "domain_profile": profile,
121
+ "dimension_summary": _dimension_summary(dimensions, dim),
122
+ }
123
+ for dim in DIMENSION_NAMES
124
+ }
125
+
126
+ for task in REVIEW_TASKS:
127
+ per_dim_counter[task.dimension] += 1
128
+ dim_summary = _dimension_summary(dimensions, task.dimension)
129
+ q = build_question_record(task, profile, dim_summary, per_dim_counter[task.dimension])
130
+ all_dim_questions[task.dimension]["questions"].append(q)
131
+
132
+ detail_out = {
133
+ "proposal_id": proposal_id,
134
+ "generated_at": now_str(),
135
+ "mode": "domain_adaptive_deterministic",
136
+ "template_version": "universal_v1",
137
+ "profile_version": "domain_profile_v1",
138
+ "dimensions": all_dim_questions,
139
+ }
140
+
141
+ out_dir = QUESTIONS_DIR / proposal_id
142
+ out_dir.mkdir(parents=True, exist_ok=True)
143
+ detail_out_path = out_dir / "generated_questions.json"
144
+ _write_json_with_log(detail_out_path, detail_out, "detailed questions")
145
+
146
+ config_out: Dict[str, Any] = {
147
+ "proposal_id": proposal_id,
148
+ "generated_at": now_str(),
149
+ "mode": "domain_adaptive_deterministic",
150
+ "template_version": "universal_v1",
151
+ "profile_version": "domain_profile_v1",
152
+ "domain_profile": profile,
153
+ }
154
+ for dim in DIMENSION_NAMES:
155
+ q_objs = all_dim_questions[dim]["questions"]
156
+ config_out[dim] = {
157
+ "dimension": dim,
158
+ "questions": [q["question_zh"] for q in q_objs],
159
+ "question_objects": q_objs,
160
+ "search_hints": QUESTION_SEARCH_HINTS.get(dim, []),
161
+ "source_proposal_id": proposal_id,
162
+ }
163
+
164
+ config_out_path = CONFIG_QS_DIR / "generated_questions.json"
165
+ _write_json_with_log(config_out_path, config_out, "runtime question set")
166
+
167
+
168
+ if __name__ == "__main__":
169
+ ap = argparse.ArgumentParser(description="Generate domain-adaptive questions from universal templates")
170
+ ap.add_argument("--proposal_id", type=str, default="", help="proposal id under src/data/extracted/<proposal_id>")
171
+ args = ap.parse_args()
172
+ pid = args.proposal_id.strip() or find_latest_extracted_proposal_id()
173
+ run_generate_questions(pid)
src/tools/layout_reconstruction.py ADDED
@@ -0,0 +1,668 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ Stage 0 升级模块:layout_reconstruction.py
4
+
5
+ 目标:
6
+ - 将 PDF 逐页渲染成图片,保留页面视觉结构
7
+ - 使用 OCR / layout / table / vision-LLM 组合构建 page_semantics.json
8
+ - 生成更适合 Stage 1 的重建文本(reconstructed_text)
9
+
10
+ 设计原则:
11
+ 1) 默认轻量可运行:即使没有 LayoutLM / Table Transformer 依赖,也能使用 OCR + 几何规则回退
12
+ 2) 可渐进增强:如果安装了 paddleocr / transformers / torch,则自动启用更强能力
13
+ 3) 输出可验证:每个页面都保留 blocks/tables/reading_order/source,便于追溯
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import base64
18
+ import io
19
+ import json
20
+ import logging
21
+ import os
22
+ import re
23
+ from collections import defaultdict
24
+ from dataclasses import dataclass, asdict
25
+ from pathlib import Path
26
+ from typing import Any, Dict, Iterable, List, Optional, Sequence, Tuple
27
+
28
+ from pdf2image import convert_from_path
29
+ from PIL import Image
30
+ import pdfplumber
31
+ import pytesseract
32
+ from pytesseract import Output
33
+
34
+ try:
35
+ from openai import OpenAI # type: ignore
36
+ except Exception: # pragma: no cover
37
+ OpenAI = None
38
+
39
+ logger = logging.getLogger(__name__)
40
+
41
+ VISION_MODEL = os.getenv("OPENAI_VISION_MODEL", os.getenv("OPENAI_MODEL", "gpt-4o-mini"))
42
+ TESSERACT_LANG = os.getenv("TESS_LANG", "chi_sim+eng")
43
+ ENABLE_VISION_LLM = os.getenv("ENABLE_VISION_LLM", "1").strip().lower() not in {"0", "false", "no"}
44
+ ENABLE_TABLE_MODEL = os.getenv("ENABLE_TABLE_MODEL", "1").strip().lower() not in {"0", "false", "no"}
45
+ ENABLE_LAYOUT_MODEL = os.getenv("ENABLE_LAYOUT_MODEL", "1").strip().lower() not in {"0", "false", "no"}
46
+ OCR_MIN_CONF = int(os.getenv("OCR_MIN_CONF", "35"))
47
+
48
+
49
+ @dataclass
50
+ class OCRWord:
51
+ text: str
52
+ x0: int
53
+ y0: int
54
+ x1: int
55
+ y1: int
56
+ conf: float
57
+
58
+ @property
59
+ def cx(self) -> float:
60
+ return (self.x0 + self.x1) / 2
61
+
62
+ @property
63
+ def cy(self) -> float:
64
+ return (self.y0 + self.y1) / 2
65
+
66
+ @property
67
+ def height(self) -> float:
68
+ return max(1, self.y1 - self.y0)
69
+
70
+
71
+ @dataclass
72
+ class Block:
73
+ block_id: str
74
+ block_type: str
75
+ text: str
76
+ bbox: List[int]
77
+ source: str
78
+ reading_order: int = -1
79
+ confidence: Optional[float] = None
80
+ meta: Optional[Dict[str, Any]] = None
81
+
82
+
83
+ @dataclass
84
+ class TableCell:
85
+ row: int
86
+ col: int
87
+ text: str
88
+ bbox: List[int]
89
+
90
+
91
+ @dataclass
92
+ class PageSemantic:
93
+ page_index: int
94
+ image_path: str
95
+ width: int
96
+ height: int
97
+ pdf_text: str
98
+ ocr_text: str
99
+ page_type: str
100
+ title: str
101
+ blocks: List[Dict[str, Any]]
102
+ tables: List[Dict[str, Any]]
103
+ reconstructed_text: str
104
+ vision_summary: Optional[Dict[str, Any]] = None
105
+ source_notes: Optional[List[str]] = None
106
+
107
+
108
+ def render_pdf_pages(pdf_path: Path, image_dir: Path, dpi: int = 220) -> List[Path]:
109
+ image_dir.mkdir(parents=True, exist_ok=True)
110
+ pages = convert_from_path(str(pdf_path), dpi=dpi)
111
+ paths: List[Path] = []
112
+ for idx, page in enumerate(pages, start=1):
113
+ out = image_dir / f"page_{idx:03d}.png"
114
+ page.save(out, "PNG")
115
+ paths.append(out)
116
+ return paths
117
+
118
+
119
+ def _load_image_size(path: Path) -> Tuple[int, int]:
120
+ with Image.open(path) as img:
121
+ return img.size
122
+
123
+
124
+
125
+
126
+ def normalize_ocr_text(text: str) -> str:
127
+ text = text or ""
128
+
129
+ # 统一换行和空白
130
+ text = text.replace("\r\n", "\n").replace("\r", "\n")
131
+ text = re.sub(r"[ \t]+", " ", text)
132
+
133
+ # 合并“逐字拆开”的中文,例如:
134
+ # 自 场 性 主 景 技 术 -> 自场性主景技术(后续再靠块合并减少这类情况)
135
+ text = re.sub(r"(?<=[\u4e00-\u9fff])\s+(?=[\u4e00-\u9fff])", "", text)
136
+
137
+ # 去掉中文与常见中文标点之间多余空格
138
+ text = re.sub(r"(?<=[\u4e00-\u9fff])\s+(?=[,。;:!?、“”‘’()《》])", "", text)
139
+ text = re.sub(r"(?<=[,。;:!?、“”‘’()《》])\s+(?=[\u4e00-\u9fff])", "", text)
140
+
141
+ # 修复中文与项目符号之间的空格
142
+ text = re.sub(r"(?<=[\u4e00-\u9fff])\s+(?=[•▪◦·])", "", text)
143
+ text = re.sub(r"(?<=[•▪◦·])\s+(?=[\u4e00-\u9fff])", "", text)
144
+
145
+ # 只保留英文/数字之间单空格
146
+ text = re.sub(r"(?<=[A-Za-z0-9]) {2,}(?=[A-Za-z0-9])", " ", text)
147
+ # 清理多余空行
148
+ text = re.sub(r"\n{3,}", "\n\n", text)
149
+
150
+ return text.strip()
151
+
152
+ def _ocr_words(image_path: Path) -> List[OCRWord]:
153
+ data = pytesseract.image_to_data(Image.open(image_path), lang=TESSERACT_LANG, output_type=Output.DICT)
154
+ words: List[OCRWord] = []
155
+ n = len(data.get("text", []))
156
+ for i in range(n):
157
+ text = (data["text"][i] or "").strip()
158
+ if not text:
159
+ continue
160
+ try:
161
+ conf = float(data["conf"][i])
162
+ except Exception:
163
+ conf = -1.0
164
+ if conf < OCR_MIN_CONF:
165
+ continue
166
+ x0 = int(data["left"][i])
167
+ y0 = int(data["top"][i])
168
+ x1 = x0 + int(data["width"][i])
169
+ y1 = y0 + int(data["height"][i])
170
+ words.append(OCRWord(text=text, x0=x0, y0=y0, x1=x1, y1=y1, conf=conf))
171
+ return words
172
+
173
+
174
+ def _group_words_into_lines(words: Sequence[OCRWord]) -> List[List[OCRWord]]:
175
+ if not words:
176
+ return []
177
+ sorted_words = sorted(words, key=lambda w: (w.cy, w.x0))
178
+ lines: List[List[OCRWord]] = []
179
+ for word in sorted_words:
180
+ placed = False
181
+ for line in lines:
182
+ ref = line[0]
183
+ tol = max(ref.height * 0.65, 12)
184
+ if abs(word.cy - ref.cy) <= tol:
185
+ line.append(word)
186
+ placed = True
187
+ break
188
+ if not placed:
189
+ lines.append([word])
190
+ for line in lines:
191
+ line.sort(key=lambda w: w.x0)
192
+ lines.sort(key=lambda l: min(w.y0 for w in l))
193
+ return lines
194
+
195
+
196
+ def _merge_lines_to_blocks(lines: Sequence[Sequence[OCRWord]], page_w: int, page_h: int) -> List[Block]:
197
+ blocks: List[Block] = []
198
+ if not lines:
199
+ return blocks
200
+
201
+ current: List[Sequence[OCRWord]] = []
202
+ block_counter = 0
203
+
204
+ def _line_text(line: Sequence[OCRWord]) -> str:
205
+ return "".join(w.text for w in line).strip()
206
+
207
+ def _is_vertical_fragment(line: Sequence[OCRWord]) -> bool:
208
+ text = _line_text(line)
209
+ if not text:
210
+ return False
211
+ compact = re.sub(r"\s+", "", text)
212
+ return len(compact) <= 2 and all(("\u4e00" <= ch <= "\u9fff") or ch in "、,。;:!?,.·•" for ch in compact)
213
+
214
+ def flush() -> None:
215
+ nonlocal current, block_counter
216
+ if not current:
217
+ return
218
+
219
+ flat = [w for line in current for w in line]
220
+ raw_lines = []
221
+ for line in current:
222
+ raw = "".join(w.text for w in line).strip()
223
+ raw_lines.append(raw)
224
+
225
+ # 关键:如果是一串“逐字拆行”的中文,直接拼成一行
226
+ if raw_lines and sum(1 for x in raw_lines if len(re.sub(r"\s+", "", x)) <= 2) >= max(3, len(raw_lines) // 2):
227
+ text = "".join(raw_lines).strip()
228
+ else:
229
+ text = "\n".join(raw_lines).strip()
230
+
231
+ text = normalize_ocr_text(text)
232
+
233
+ x0 = min(w.x0 for w in flat)
234
+ y0 = min(w.y0 for w in flat)
235
+ x1 = max(w.x1 for w in flat)
236
+ y1 = max(w.y1 for w in flat)
237
+ avg_conf = sum(w.conf for w in flat) / max(1, len(flat))
238
+
239
+ block_type = classify_block(text, [x0, y0, x1, y1], page_w, page_h)
240
+ blocks.append(Block(
241
+ block_id=f"b{block_counter:03d}",
242
+ block_type=block_type,
243
+ text=text,
244
+ bbox=[x0, y0, x1, y1],
245
+ source="ocr_layout",
246
+ confidence=round(avg_conf, 2),
247
+ meta={"line_count": len(current)}
248
+ ))
249
+ block_counter += 1
250
+ current = []
251
+
252
+ prev_bottom = None
253
+ prev_x0 = None
254
+ prev_is_vertical = False
255
+
256
+ for line in lines:
257
+ flat = list(line)
258
+ y0 = min(w.y0 for w in flat)
259
+ x0 = min(w.x0 for w in flat)
260
+ this_is_vertical = _is_vertical_fragment(line)
261
+
262
+ if not current:
263
+ current = [line]
264
+ prev_bottom = max(w.y1 for w in flat)
265
+ prev_x0 = x0
266
+ prev_is_vertical = this_is_vertical
267
+ continue
268
+
269
+ gap = y0 - (prev_bottom or y0)
270
+
271
+ # 普通段落合并
272
+ same_paragraph = gap <= 18 and prev_x0 is not None and abs(x0 - prev_x0) <= 60
273
+
274
+ # 关键:纵向碎字列,允许更强合并
275
+ same_vertical_stream = (
276
+ prev_x0 is not None
277
+ and this_is_vertical
278
+ and prev_is_vertical
279
+ and abs(x0 - prev_x0) <= 40
280
+ and gap <= 35
281
+ )
282
+
283
+ if same_paragraph or same_vertical_stream:
284
+ current.append(line)
285
+ prev_bottom = max(w.y1 for w in flat)
286
+ prev_x0 = (prev_x0 + x0) / 2 if prev_x0 is not None else x0
287
+ prev_is_vertical = this_is_vertical
288
+ else:
289
+ flush()
290
+ current = [line]
291
+ prev_bottom = max(w.y1 for w in flat)
292
+ prev_x0 = x0
293
+ prev_is_vertical = this_is_vertical
294
+
295
+ flush()
296
+ return blocks
297
+
298
+
299
+ def classify_block(text: str, bbox: List[int], page_w: int, page_h: int) -> str:
300
+ stripped = re.sub(r"\s+", " ", text).strip()
301
+ x0, y0, x1, y1 = bbox
302
+ width = x1 - x0
303
+ height = y1 - y0
304
+ line_count = stripped.count("\n") + 1
305
+ compact = stripped.replace(" ", "")
306
+ if len(compact) <= 2 and not re.search(r"[\u4e00-\u9fffA-Za-z0-9]", compact):
307
+ return "noise"
308
+ if y0 > page_h * 0.92:
309
+ return "footer"
310
+ if y0 < page_h * 0.18 and len(compact) <= 40:
311
+ return "title"
312
+ if re.fullmatch(r"[0-9]+([.,][0-9]+)?[A-Za-z%°·sN\.]*", compact):
313
+ return "numeric_label"
314
+ if line_count >= 2 and re.search(r"[||]", stripped):
315
+ return "table_like"
316
+ if width > page_w * 0.6 and line_count >= 2:
317
+ return "paragraph"
318
+ if height < 24 and len(compact) <= 3:
319
+ return "noise"
320
+ if len(stripped) <= 30:
321
+ return "label"
322
+ return "text"
323
+
324
+
325
+ def recover_reading_order(blocks: Sequence[Block]) -> List[Block]:
326
+ if not blocks:
327
+ return []
328
+ sorted_blocks = sorted(blocks, key=lambda b: (b.bbox[1], b.bbox[0]))
329
+ rows: List[List[Block]] = []
330
+ for block in sorted_blocks:
331
+ placed = False
332
+ cy = (block.bbox[1] + block.bbox[3]) / 2
333
+ for row in rows:
334
+ ref = row[0]
335
+ ref_cy = (ref.bbox[1] + ref.bbox[3]) / 2
336
+ tol = max((ref.bbox[3] - ref.bbox[1]) * 0.7, 18)
337
+ if abs(cy - ref_cy) <= tol:
338
+ row.append(block)
339
+ placed = True
340
+ break
341
+ if not placed:
342
+ rows.append([block])
343
+ rows.sort(key=lambda r: min(b.bbox[1] for b in r))
344
+ ordered: List[Block] = []
345
+ idx = 0
346
+ for row in rows:
347
+ row.sort(key=lambda b: b.bbox[0])
348
+ for block in row:
349
+ block.reading_order = idx
350
+ ordered.append(block)
351
+ idx += 1
352
+ return ordered
353
+
354
+
355
+ def _extract_pdf_text_page(pdf_path: Path, page_index: int) -> str:
356
+ try:
357
+ with pdfplumber.open(pdf_path) as pdf:
358
+ if page_index < len(pdf.pages):
359
+ return (pdf.pages[page_index].extract_text() or "").strip()
360
+ except Exception as exc:
361
+ logger.warning("pdfplumber page text failed: %s", exc)
362
+ return ""
363
+
364
+
365
+ def detect_tables_with_pdfplumber(pdf_path: Path, page_index: int) -> List[Dict[str, Any]]:
366
+ tables: List[Dict[str, Any]] = []
367
+ try:
368
+ with pdfplumber.open(pdf_path) as pdf:
369
+ if page_index >= len(pdf.pages):
370
+ return tables
371
+ page = pdf.pages[page_index]
372
+ for t_idx, table in enumerate(page.extract_tables() or []):
373
+ if not table:
374
+ continue
375
+ rows = []
376
+ for r_idx, row in enumerate(table):
377
+ if row is None:
378
+ continue
379
+ cells = []
380
+ for c_idx, cell in enumerate(row):
381
+ txt = (cell or "").strip()
382
+ cells.append({"row": r_idx, "col": c_idx, "text": txt})
383
+ rows.append(cells)
384
+ tables.append({
385
+ "table_id": f"pdf_table_{page_index+1}_{t_idx+1}",
386
+ "source": "pdfplumber",
387
+ "cells": rows,
388
+ "table_text": "\n".join(" | ".join(c["text"] for c in row) for row in rows if row),
389
+ })
390
+ except Exception as exc:
391
+ logger.warning("pdfplumber table extraction failed: %s", exc)
392
+ return tables
393
+
394
+
395
+ def detect_tables_with_transformer(image_path: Path) -> List[Dict[str, Any]]:
396
+ if not ENABLE_TABLE_MODEL:
397
+ return []
398
+ try:
399
+ import torch # type: ignore
400
+ from transformers import AutoImageProcessor, TableTransformerForObjectDetection # type: ignore
401
+ except Exception:
402
+ return []
403
+
404
+ model_name = os.getenv("TABLE_TRANSFORMER_MODEL", "microsoft/table-transformer-detection")
405
+ try:
406
+ processor = AutoImageProcessor.from_pretrained(model_name)
407
+ model = TableTransformerForObjectDetection.from_pretrained(model_name)
408
+ image = Image.open(image_path).convert("RGB")
409
+ inputs = processor(images=image, return_tensors="pt")
410
+ outputs = model(**inputs)
411
+ target_sizes = torch.tensor([image.size[::-1]])
412
+ results = processor.post_process_object_detection(outputs, threshold=0.85, target_sizes=target_sizes)[0]
413
+ tables = []
414
+ for idx, (score, label, box) in enumerate(zip(results["scores"], results["labels"], results["boxes"])):
415
+ x0, y0, x1, y1 = [int(v) for v in box.tolist()]
416
+ label_name = model.config.id2label.get(int(label), str(label))
417
+ tables.append({
418
+ "table_id": f"tt_{idx+1}",
419
+ "source": model_name,
420
+ "label": label_name,
421
+ "score": float(score),
422
+ "bbox": [x0, y0, x1, y1],
423
+ })
424
+ return tables
425
+ except Exception as exc:
426
+ logger.warning("Table Transformer failed: %s", exc)
427
+ return []
428
+
429
+
430
+ def _guess_page_type(blocks: Sequence[Block], tables: Sequence[Dict[str, Any]]) -> str:
431
+ texts = " ".join(b.text for b in blocks)
432
+ if tables:
433
+ return "table_page"
434
+ if any("目录" in b.text or "CONTENTS" in b.text.upper() for b in blocks):
435
+ return "toc_slide"
436
+ if re.search(r"20\d{2}年", texts) and len(re.findall(r"20\d{2}年", texts)) >= 3:
437
+ return "timeline_slide"
438
+ if len([b for b in blocks if b.block_type == "numeric_label"]) >= 3:
439
+ return "data_visual_slide"
440
+ if len(blocks) <= 4:
441
+ return "cover_or_simple_slide"
442
+ return "content_slide"
443
+
444
+
445
+ def _title_from_blocks(blocks: Sequence[Block]) -> str:
446
+ titles = [b for b in blocks if b.block_type == "title"]
447
+ if titles:
448
+ return normalize_ocr_text(re.sub(r"\s+", " ", titles[0].text).strip())
449
+ if blocks:
450
+ return re.sub(r"\s+", " ", blocks[0].text.split("\n", 1)[0]).strip()
451
+ return ""
452
+
453
+
454
+ def _reconstruct_table_text(table: Dict[str, Any]) -> str:
455
+ cells = table.get("cells") or []
456
+ lines = []
457
+ for row in cells:
458
+ if isinstance(row, list):
459
+ line = " | ".join((c.get("text") or "").strip() for c in row)
460
+ if line.strip(" |"):
461
+ lines.append(line)
462
+ return "\n".join(lines)
463
+
464
+
465
+ def _fallback_reconstructed_text(blocks: Sequence[Block], tables: Sequence[Dict[str, Any]]) -> str:
466
+ parts: List[str] = []
467
+
468
+ for block in blocks:
469
+ if block.block_type in {"footer", "noise"}:
470
+ continue
471
+
472
+ text = normalize_ocr_text(block.text.strip())
473
+ if not text:
474
+ continue
475
+
476
+ parts.append(text)
477
+
478
+ for table in tables:
479
+ text = table.get("table_text") or _reconstruct_table_text(table)
480
+ text = normalize_ocr_text(text)
481
+ if text:
482
+ parts.append(text)
483
+
484
+ merged = "\n\n".join(p for p in parts if p).strip()
485
+ return normalize_ocr_text(merged)
486
+
487
+
488
+ def _image_to_data_url(image_path: Path) -> str:
489
+ raw = image_path.read_bytes()
490
+ return "data:image/png;base64," + base64.b64encode(raw).decode("ascii")
491
+
492
+
493
+ def vision_llm_enrich_page(image_path: Path, page_payload: Dict[str, Any]) -> Optional[Dict[str, Any]]:
494
+ if not ENABLE_VISION_LLM:
495
+ return None
496
+ api_key = os.getenv("OPENAI_API_KEY", "").strip()
497
+ if not api_key or OpenAI is None:
498
+ return None
499
+ client = OpenAI(api_key=api_key)
500
+ prompt = (
501
+ "你是一个严格的多模态文档页面解析器。"
502
+ "请基于页面图像本身作为第一依据,OCR/布局候选只作为辅助参考。"
503
+ "目标:输出适合后续评审系统使用的高质量结构化 JSON。"
504
+
505
+ "\n\n要求:"
506
+ "\n1. 优先恢复页面的自然阅读顺序。"
507
+ "\n2. 不要保留明显错误的 OCR 垃圾文本。"
508
+ "\n3. 如果页面中存在竖排字、装饰文字、旋转文本、图形中的碎字、难以连成自然句子的散字,请忽略它们。"
509
+ "\n4. reconstructed_text 只保留对页面语义真正有贡献的内容。"
510
+ "\n5. 对标题、正文、表格、关键标签分别理解,但不要编造。"
511
+ "\n6. 如果 OCR 候选与图像冲突,以图像理解为准。"
512
+ "\n7. 不要把单字散列、坐标噪声、页码、无意义重复片段写进 reconstructed_text。"
513
+
514
+ "\n\n输出 JSON 格式:"
515
+ '{"page_type":"...",'
516
+ '"title":"...",'
517
+ '"reading_order_notes":["..."],'
518
+ '"regions":[{"role":"title/text/table/chart/label","text":"..."}],'
519
+ '"reconstructed_text":"..."}'
520
+ )
521
+ try:
522
+ response = client.chat.completions.create(
523
+ model=VISION_MODEL,
524
+ temperature=0.0,
525
+ response_format={"type": "json_object"},
526
+ messages=[
527
+ {"role": "system", "content": "你是严谨的页面结构解析器,只能依据可见内容回答。"},
528
+ {
529
+ "role": "user",
530
+ "content": [
531
+ {"type": "text", "text": prompt + "\n\nOCR/布局候选:\n" + json.dumps(page_payload, ensure_ascii=False)[:12000]},
532
+ {"type": "image_url", "image_url": {"url": _image_to_data_url(image_path)}},
533
+ ],
534
+ },
535
+ ],
536
+ max_tokens=1800,
537
+ )
538
+ raw = response.choices[0].message.content or "{}"
539
+ return json.loads(raw)
540
+ except Exception as exc:
541
+ logger.warning("vision LLM enrichment failed on %s: %s", image_path, exc)
542
+ return None
543
+
544
+
545
+ def build_page_semantic(pdf_path: Path, image_path: Path, page_index: int) -> PageSemantic:
546
+ width, height = _load_image_size(image_path)
547
+ ocr_words = _ocr_words(image_path)
548
+ lines = _group_words_into_lines(ocr_words)
549
+ blocks = _merge_lines_to_blocks(lines, page_w=width, page_h=height)
550
+ ordered_blocks = [b for b in recover_reading_order(blocks) if b.block_type != "noise"]
551
+ pdf_text = _extract_pdf_text_page(pdf_path, page_index)
552
+ ocr_text = normalize_ocr_text("\n".join(" ".join(w.text for w in line) for line in lines).strip())
553
+
554
+ tables = detect_tables_with_pdfplumber(pdf_path, page_index)
555
+ transformer_tables = detect_tables_with_transformer(image_path)
556
+ if transformer_tables:
557
+ tables.extend(transformer_tables)
558
+
559
+ page_type = _guess_page_type(ordered_blocks, tables)
560
+ title = _title_from_blocks(ordered_blocks)
561
+
562
+ base_payload = {
563
+ "page_index": page_index + 1,
564
+ "page_type": page_type,
565
+ "title": title,
566
+ "blocks": [asdict(b) for b in ordered_blocks],
567
+ "tables": tables,
568
+ "pdf_text": pdf_text,
569
+ "ocr_text": ocr_text,
570
+ }
571
+ vision_summary = vision_llm_enrich_page(image_path, base_payload)
572
+
573
+ # Vision-LLM 优先:把 page_type / title / reconstructed_text / regions 作为主结果
574
+ if vision_summary and isinstance(vision_summary, dict):
575
+ reconstructed_text = normalize_ocr_text(
576
+ vision_summary.get("reconstructed_text", "")
577
+ )
578
+ else:
579
+ reconstructed_text = _fallback_reconstructed_text(ordered_blocks, tables)
580
+
581
+ reconstructed_text = normalize_ocr_text(reconstructed_text)
582
+
583
+ source_notes = []
584
+ if pdf_text:
585
+ source_notes.append("pdf_text_available")
586
+ if ocr_text:
587
+ source_notes.append("ocr_layout_available")
588
+ if vision_summary:
589
+ source_notes.append("vision_llm_enriched")
590
+ if transformer_tables:
591
+ source_notes.append("table_transformer_detected")
592
+ elif tables:
593
+ source_notes.append("pdfplumber_tables_detected")
594
+
595
+ return PageSemantic(
596
+ page_index=page_index + 1,
597
+ image_path=str(image_path),
598
+ width=width,
599
+ height=height,
600
+ pdf_text=pdf_text,
601
+ ocr_text=ocr_text,
602
+ page_type=(vision_summary or {}).get("page_type", page_type),
603
+ title=(vision_summary or {}).get("title", title),
604
+ blocks=[asdict(b) for b in ordered_blocks],
605
+ tables=tables,
606
+ reconstructed_text=reconstructed_text,
607
+ vision_summary=vision_summary,
608
+ source_notes=source_notes,
609
+ )
610
+
611
+
612
+ def build_document_semantics(pdf_path: Path, out_dir: Path, dpi: int = 220) -> Dict[str, Any]:
613
+ image_dir = out_dir / "page_images"
614
+ image_paths = render_pdf_pages(pdf_path, image_dir=image_dir, dpi=dpi)
615
+ pages: List[Dict[str, Any]] = []
616
+ for idx, image_path in enumerate(image_paths):
617
+ page_sem = build_page_semantic(pdf_path=pdf_path, image_path=image_path, page_index=idx)
618
+ pages.append(asdict(page_sem))
619
+ reconstructed_pages = [p.get("reconstructed_text", "").strip() for p in pages]
620
+ reconstructed_full_text = "\n\n".join([p for p in reconstructed_pages if p])
621
+ return {
622
+ "proposal_file": str(pdf_path),
623
+ "num_pages": len(pages),
624
+ "pages": pages,
625
+ "reconstructed_full_text": reconstructed_full_text,
626
+ }
627
+
628
+
629
+ def save_document_semantics(doc_semantics: Dict[str, Any], out_dir: Path) -> Dict[str, str]:
630
+ out_dir.mkdir(parents=True, exist_ok=True)
631
+ page_sem_path = out_dir / "page_semantics.json"
632
+ page_sem_path.write_text(json.dumps(doc_semantics, ensure_ascii=False, indent=2), encoding="utf-8")
633
+ reconstructed_text_path = out_dir / "reconstructed_full_text.txt"
634
+ reconstructed_text_path.write_text(doc_semantics.get("reconstructed_full_text", ""), encoding="utf-8")
635
+ return {
636
+ "page_semantics_path": str(page_sem_path),
637
+ "reconstructed_full_text_path": str(reconstructed_text_path),
638
+ "page_images_dir": str(out_dir / "page_images"),
639
+ }
640
+
641
+
642
+ def semantic_pages_for_stage1(page_semantics: Dict[str, Any]) -> List[Dict[str, Any]]:
643
+ pages = page_semantics.get("pages") or []
644
+ semantic_units: List[Dict[str, Any]] = []
645
+ for page in pages:
646
+ page_idx = page.get("page_index")
647
+ title = (page.get("title") or "").strip()
648
+ page_type = page.get("page_type") or "content_slide"
649
+ text_parts: List[str] = []
650
+ if title:
651
+ text_parts.append(f"[PAGE_TITLE] {title}")
652
+ text_parts.append(f"[PAGE_TYPE] {page_type}")
653
+ reconstructed_text = (page.get("reconstructed_text") or "").strip()
654
+ if reconstructed_text:
655
+ text_parts.append(reconstructed_text)
656
+ for table in page.get("tables") or []:
657
+ ttext = table.get("table_text") or _reconstruct_table_text(table)
658
+ if ttext:
659
+ text_parts.append(f"[TABLE]\n{ttext}")
660
+ unit_text = "\n\n".join([p for p in text_parts if p]).strip()
661
+ semantic_units.append({
662
+ "page_index": page_idx,
663
+ "title": title,
664
+ "page_type": page_type,
665
+ "text": unit_text,
666
+ "source": "page_semantics",
667
+ })
668
+ return semantic_units
src/tools/llm_answering.py ADDED
@@ -0,0 +1,1456 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ llm_answering.py · Proposal-aware LLM Answering (ChatGPT + DeepSeek, no web search)
4
+
5
+ 新版职责:
6
+ - 基于「维度抽取管线」生成的提案事实(dimensions_from_facts)+ 生成的问题集,
7
+ 为每个维度的问题生成“强关联该提案”的结构化回答。
8
+ - 不再依赖外部 Web 检索;只使用:
9
+ • 提案维度事实(summary/key_points/risks/mitigations/numbers 等)
10
+ • LLM 自身通识做解释,但禁止脑补新实验/新数字/新机构
11
+ - 输出结构保持兼容:
12
+ data/refined_answers/{pid}/all_refined_items.json
13
+ data/refined_answers/{pid}/chatgpt_raw.json
14
+ data/refined_answers/{pid}/deepseek_raw.json
15
+
16
+ 本版新增:
17
+ - 明确要求回答中包含“行业基准 / baseline 对比”和“常见坑 & 证据要求”两类内容;
18
+ - 这些通识部分必须写成“行业普遍情况/一般建议”,不能写成项目已经达成的事实;
19
+ - 新增字段 general_insights:专门装“行业基准 / 常见坑 / 证据要求”这类专家经验层;
20
+ - 修复 answer 分点格式问题,避免出现 “1. 1. xxx” 这种双重编号。
21
+ """
22
+
23
+ import os
24
+ import sys
25
+ import json
26
+ import time
27
+ import re
28
+ import random
29
+ import argparse
30
+ from pathlib import Path
31
+ from datetime import datetime
32
+ from typing import Dict, Any, List, Tuple, Optional
33
+
34
+ from dotenv import load_dotenv
35
+
36
+ # ========== 环境 & 路径 ==========
37
+ load_dotenv()
38
+ ROOT = Path(__file__).resolve().parents[1] # .../src
39
+ DATA_DIR = ROOT / "data"
40
+ EXTRACTED_DIR = DATA_DIR / "extracted"
41
+ PARSED_DIR = DATA_DIR / "parsed"
42
+ CONFIG_QS_DEFAULT = DATA_DIR / "config" / "question_sets" / "generated_questions.json"
43
+ OUT_REFINED = DATA_DIR / "refined_answers"
44
+ from src.prompting.domain_adaptive import load_domain_profile
45
+
46
+ DIM_ORDER = ["team", "objectives", "strategy", "innovation", "feasibility"]
47
+
48
+ # ========== SDK ==========
49
+ try:
50
+ from openai import OpenAI as OpenAIClient
51
+ except Exception: # pragma: no cover
52
+ OpenAIClient = None
53
+
54
+ # ========== 变体 & provider 能力 ==========
55
+ VARIANTS = ["default", "risk", "implementation"]
56
+ TEMP_BY_VARIANT = {"default": 0.25, "risk": 0.35, "implementation": 0.30}
57
+
58
+ PROVIDER_CAPS = {
59
+ "openai": {"json_mode": True, "batch_ok": True},
60
+ "deepseek": {"json_mode": False, "batch_ok": False},
61
+ }
62
+
63
+
64
+ def provider_caps(provider: str) -> Dict[str, Any]:
65
+ return PROVIDER_CAPS.get(provider, {"json_mode": True, "batch_ok": True})
66
+
67
+
68
+ CONF_MIN, CONF_MAX = 0.50, 0.92
69
+ MAX_LIST_LEN = 10
70
+
71
+ # ========== 工具函数 ==========
72
+ def read_json(p: Path) -> Any:
73
+ return json.loads(Path(p).read_text(encoding="utf-8"))
74
+
75
+
76
+ def write_json(p: Path, obj: Any) -> None:
77
+ p.parent.mkdir(parents=True, exist_ok=True)
78
+ p.write_text(json.dumps(obj, ensure_ascii=False, indent=2), encoding="utf-8")
79
+
80
+
81
+ def now_str() -> str:
82
+ return datetime.now().strftime("%Y-%m-%d %H:%M:%S")
83
+
84
+
85
+ def detect_latest_pid() -> str:
86
+ if not EXTRACTED_DIR.exists():
87
+ return "unknown"
88
+ cands = [d for d in EXTRACTED_DIR.iterdir() if d.is_dir()]
89
+ if not cands:
90
+ return "unknown"
91
+ cands.sort(key=lambda x: x.stat().st_mtime, reverse=True)
92
+ return cands[0].name
93
+
94
+ def _flatten_list_field(block: Dict[str, Any], keys: List[str], limit: int = 12) -> List[str]:
95
+ items: List[str] = []
96
+ for k in keys:
97
+ v = block.get(k)
98
+ if not v:
99
+ continue
100
+ if isinstance(v, str):
101
+ items.extend([x.strip() for x in re.split(r"[;\n]", v) if x.strip()])
102
+ elif isinstance(v, list):
103
+ for x in v:
104
+ if isinstance(x, str):
105
+ s = x.strip()
106
+ if s:
107
+ items.append(s)
108
+ uniq, seen = [], set()
109
+ for x in items:
110
+ key = x.strip()
111
+ if not key or key in seen:
112
+ continue
113
+ seen.add(key)
114
+ uniq.append(key)
115
+ if len(uniq) >= limit:
116
+ break
117
+ return uniq
118
+
119
+
120
+ def _build_dim_context_text(dim: str, block: Dict[str, Any]) -> str:
121
+ """
122
+ 把从 facts 管线来的一个维度 block 转成适合放进 prompt 的 context 文本。
123
+ 控制长度,优先 summary / key_points / risks / mitigations / numbers。
124
+ """
125
+ if not isinstance(block, dict):
126
+ block = {}
127
+
128
+ summary = str(block.get("summary") or "").strip()
129
+ key_points = _flatten_list_field(block, ["key_points", "keypoints", "key_facts", "bullets"], limit=10)
130
+ risks = _flatten_list_field(block, ["risks", "risk_points"], limit=8)
131
+ mitigations = _flatten_list_field(block, ["mitigations", "mitigation_points"], limit=8)
132
+ numbers = _flatten_list_field(block, ["numbers", "key_numbers"], limit=8)
133
+
134
+ parts: List[str] = []
135
+ if summary:
136
+ parts.append(f"【{dim} 概览】{summary}")
137
+ if key_points:
138
+ parts.append("【关键要点】" + ";".join(key_points))
139
+ if risks:
140
+ parts.append("【主要风险/不确定性】" + ";".join(risks))
141
+ if mitigations:
142
+ parts.append("【已有��解措施】" + ";".join(mitigations))
143
+ if numbers:
144
+ parts.append("【关键量化信息】" + ";".join(numbers))
145
+
146
+ text = "\n".join(parts)
147
+ if len(text) > 2000:
148
+ text = text[:2000]
149
+ return text
150
+
151
+
152
+ def load_dimension_context(pid: str, dim_file: Optional[Path]) -> Dict[str, str]:
153
+ """
154
+ 返回:{dim: context_text}
155
+ 若找不到文件或结构异常,所有维度返回空字符串(模型会退化成通识回答)
156
+ """
157
+ ctx: Dict[str, str] = {d: "" for d in DIM_ORDER}
158
+ if dim_file is None or not dim_file.exists():
159
+ return ctx
160
+
161
+ try:
162
+ raw = read_json(dim_file)
163
+ except Exception:
164
+ return ctx
165
+
166
+ if isinstance(raw, dict) and "dimensions" in raw and isinstance(raw["dimensions"], dict):
167
+ root = raw["dimensions"]
168
+ else:
169
+ root = raw if isinstance(raw, dict) else {}
170
+
171
+ for dim in DIM_ORDER:
172
+ block = root.get(dim) or {}
173
+ ctx[dim] = _build_dim_context_text(dim, block)
174
+ return ctx
175
+
176
+
177
+ # ========== 问题集 & 监管提示 ==========
178
+ def get_q_list(block: Any) -> List[str]:
179
+ if isinstance(block, dict) and isinstance(block.get("questions"), list):
180
+ return [q for q in block["questions"] if isinstance(q, str)]
181
+ if isinstance(block, list):
182
+ return [q for q in block if isinstance(q, str)]
183
+ return []
184
+
185
+
186
+ def _load_reg_hints(qs_cfg: Dict[str, Any], dim: str, limit: int = 8) -> List[str]:
187
+ """
188
+ 从问题集里抓一点监管/术语 hints,但只作为“可选方向提示”,不强依赖。
189
+ """
190
+ block = qs_cfg.get(dim, {}) or {}
191
+ hints = block.get("search_hints") or block.get("reg_hints") or []
192
+ out: List[str] = []
193
+ if isinstance(hints, list):
194
+ for h in hints:
195
+ if not isinstance(h, str):
196
+ continue
197
+ s = h.strip()
198
+ if not s:
199
+ continue
200
+ out.append(s)
201
+ uniq, seen = [], set()
202
+ for x in out:
203
+ if x in seen:
204
+ continue
205
+ seen.add(x)
206
+ uniq.append(x)
207
+ if len(uniq) >= limit:
208
+ break
209
+ return uniq
210
+
211
+ # ========== Prompt 相关 ==========
212
+ SYSTEM_CN = (
213
+ "你是一名严格、通用、证据优先的项目评审专家。"
214
+ "你的任务是:在完整阅读给定的【提案事实】之后,围绕问题进行“与该提案强相关”的专业分析。"
215
+ "原则:"
216
+ "1)必须优先基于【提案事实】给出结论;"
217
+ "2)不得发明提案中未出现的新试验、新数据、新机构或具体数字;"
218
+ "3)当你在【提案事实】中找不到某一类信息时,只能说“当前材料中未看到关于 X 的具体说明”,"
219
+ " 并优先写成“已有 A/B,但在 C/D 方面细节不足”;严禁使用“完全没有分析”“未进行任何评估”"
220
+ " “未提供任何信息”等绝对化表述;"
221
+ "4)可以使用行业通识解释这些事实的意义,但不得虚构提案中没有出现的具体编号、样本量、组织名称或数字细节;"
222
+ "5)回答必须结构化、条理清晰,适合作为专家评审报告的组成部分;"
223
+ "6)你需要在回答中补充“行业基准/baseline 对比”和“类似项目常见的坑与证据要求”等通识内容,"
224
+ " 但这些通识部分必须明确表述为“行业普遍情况/一般建议”,不得写成项目已经达成的事实;"
225
+ )
226
+
227
+ def _schema_structured() -> str:
228
+ return """
229
+ 请严格返回 JSON 对象,键名固定:
230
+ {
231
+ "answer": "分点形式的主回答(中文,条理清晰,含项目现状+问题+改进方向等)",
232
+ "claims": ["关键可验证结论1","关键可验证结论2","..."],
233
+ "evidence_hints": ["哪条提案事实/段落支撑对应结论,或需进一步核查的线索"],
234
+ "general_insights": ["行业基准/类似项目常见做法/常见坑与证据要求(通识,不代表本项目已完成;建议最后一条给出统一免责声明)"],
235
+ "topic_tags": ["维度内的小主题/标签"],
236
+ "confidence": 0.0,
237
+ "caveats": "限制/注意事项(若无可空)"
238
+ }
239
+ """.strip()
240
+
241
+
242
+ def _variant_instructions(variant_id: str) -> str:
243
+ if variant_id == "risk":
244
+ return (
245
+ "变体:risk(风险视角)——重点:\n"
246
+ "- 优先识别不确定性、缺失信息、潜在合规或技术风险;\n"
247
+ "- 对每个主要风险给出“风险来源 + 可能影响 + 建议补充材料”;\n"
248
+ "- 若提案事实中未覆盖关键环节,要显式指出“信息缺口”;\n"
249
+ "- 在 general_insights 中,总结同类项目在该维度常见的风险模式、监管关注点和证据要求,"
250
+ " 明确说明这些是行业通识,不代表本项目已经满足。\n"
251
+ )
252
+ if variant_id == "implementation":
253
+ return (
254
+ "变体:implementation(落地视角)——重点:\n"
255
+ "- 列出 4–8 条“接下来 3–12 个月可执行的具体动作”,每条包含:动作主体/对象 + 具体步骤 + 预期产出;\n"
256
+ "- 动作必须与提案当前状态匹配,不得假设已完成的工作;\n"
257
+ "- 可以提及需要收集/补充的证据或文档类型(而不是凭空给出结论);\n"
258
+ "- 在 general_insights 中,总结同类项目在该维度的主流落地路径、里程碑拆解和常见踩坑点。\n"
259
+ )
260
+ return (
261
+ "变体:default(综合视角)——重点:\n"
262
+ "- 结合该维度的“当前方案/优势/问题/建议”,给出整体评价;\n"
263
+ "- 既要指出做得好的地方,也要指出存在的不足或不确定性;\n"
264
+ "- 至少包含:现状概括、优势点、主要问题、改进方向四类内容;\n"
265
+ "- 在 answer 中,你需要显式区分:\n"
266
+ " a) 基于【提案事实】得出的本项目现状与问题;\n"
267
+ " b) 行业基准 / baseline 对比(同类项目通常达到什么水平、需要哪些能力/数据/里程碑);\n"
268
+ " c) 同类项目常见的坑和证据要求/监管关注点;\n"
269
+ "- 在 claims 中,优先写“提案已经说明了什么/在哪些方面信息仍然有限”,"
270
+ " 避免使用“项目未提供/没有进行任何……”这类绝对否定,如果只能判断信息有限,"
271
+ " 应写成“现有材料未看到关于 X 的更详细说明”。\n"
272
+ "- 在 general_insights 字段中,对 b) 和 c) 做更加抽象的行业通识总结,"
273
+ " 明确使用“通常/一般而言/在同类项目中”等措辞,避免暗示本项目已经达成这些条件。\n"
274
+ )
275
+
276
+ def build_single_prompt(
277
+ dimension: str,
278
+ question: str,
279
+ proposal_context: str,
280
+ reg_hints: List[str],
281
+ variant_id: str,
282
+ ) -> str:
283
+ reg_txt = ";".join(reg_hints) if reg_hints else "无特别提示"
284
+ return f"""
285
+ 维度:{dimension}
286
+
287
+ [提案事实](只能在这里面引用具体细节;如无相关信息请据实指出):
288
+ {proposal_context or "(该维度的提案事实为空,仅可做通识性分析)"}
289
+
290
+ [可选术语/方法方向提示](非必须引用,如与提案无关可忽略):
291
+ {reg_txt}
292
+
293
+ 题目:
294
+ {question.strip()}
295
+
296
+ {_variant_instructions(variant_id)}
297
+
298
+ 回答要求:
299
+ - 语言:中文(专有名词可保留英文);
300
+ - 结构:answer 必须以 3–8 条分点给出,每条前加“1. 2. 3.” 等编号,每条尽量控制在一到两句话;
301
+ - 内容结构建议(非强制格式,但需覆盖):\n
302
+ 1)先基于[提案事实]总结本项目在该维度的“现状 + 优势 + 主要问题”;\n
303
+ 2)再用 1–3 条要点,对标“行业基准 / baseline”,说明同类项目通常在该维度需要达到什么水平(注意:这是行业通识,不代表本项目已达到);\n
304
+ 3)再用 1–3 条要点,总结“同类项目常见的坑、证据要求、领域常见关注点”;\n
305
+ - 关联度:优先基于[提案事实]分析该项目的真实情况;对于[提案事实]中没有的信息,只能用“缺失/需补充”的方式描述,不得假设已经存在;\n
306
+ - 若[提案事实]中已经给出某一方面的部分信息(例如已有部分证据但缺少细化说明、已有对象列表但缺少对应比较),"
307
+ " 必须写成“已有……但在……方面仍缺乏具体细节”,不得笼统说“未提供相关分析/未进行对应评估”;\n
308
+ - 通识与事实的区分:凡是基于行业经验的内容,需要在句中用“通常/一般而言/在同类项目中”等词标识清楚,避免写成好像本项目已经完成这些工作;\n
309
+ - general_insights 字段:请单独列出 3–8 条不依赖本项目具体事实的“行业基准/常见坑/证据要求”要点,这些内容应可用于评估任何类似项目,且必须表述为通识建议;最后一条建议写成类似 “以上为行业通识建议,不代表本项目已经达成相关要求。” 这样的统一免责声明;\n
310
+ - 审慎性:避免下结论式的绝对语气,多用“可能/建议/需要确认”等方式,并尽量指出需要的佐证材料类型;\n
311
+ - 仅输出一个 JSON 对象,不得有任何额外说明文字或 Markdown 围栏。
312
+
313
+ {_schema_structured()}
314
+ """.strip()
315
+
316
+
317
+ def build_batch_prompt(
318
+ dimension: str,
319
+ questions: List[str],
320
+ proposal_context: str,
321
+ reg_hints: List[str],
322
+ variant_id: str,
323
+ ) -> str:
324
+ reg_txt = ";".join(reg_hints) if reg_hints else "无特别提示"
325
+ q_block = "\n".join([f"{i+1}. {q}" for i, q in enumerate(questions)])
326
+ return f"""
327
+ 你将针对同一维度下的多道问题,基于同一份【提案事实】给出与该提案高度相关的回答。
328
+ 请严格以对象数组 JSON 返回结果,格式为:{{"answers":[<对象1>,<对象2>,...]}},数组长度必须与题目数一致。
329
+
330
+ 维度:{dimension}
331
+
332
+ [提案事实](只能在这里面引用具体细节;如无相关信息请据实指出):
333
+ {proposal_context or "(该维度的提案事实为空,仅可��通识性分析)"}
334
+
335
+ [可选术语/方法方向提示](非必须引用,如与提案无关可忽略):
336
+ {reg_txt}
337
+
338
+ 问题列表:
339
+ {q_block}
340
+
341
+ {_variant_instructions(variant_id)}
342
+
343
+ 统一回答要求:
344
+ - 每道题的 answer:3–8 条分点,每条前加数字编号;条数不足时宁可只写 3–4 条扎实要点,也不要凑模板口号;
345
+ - 建议在 answer 中显式覆盖三类信息:\n
346
+ a) 仅基于[提案事实]得出的本项目现状/优势/问题;\n
347
+ b) 行业基准 / baseline 对比(同类项目通常需要的团队能力、数据规模、里程碑等——需标明是行业通识);\n
348
+ c) 同类项目在该维度常见的坑、证据要求、领域常见关注点;\n
349
+ - 每道题的 claims:2–6 条可以被事后核对的结论,优先基于[提案事实];若信息不足,请将“信息缺口”本身写入 claims;
350
+ - 每道题的 evidence_hints:指明“哪类提案事实/哪一段内容/哪类文档可以支撑这些结论”,避免写空泛模板;
351
+ - 每道题的 general_insights:3–8 条不依赖本项目具体事实的“行业基准/常见坑/证据要求”要点,只能写成行业普遍情况/一般建议,不得暗示本项目已经达成;建议其中最后一条写成统一的免责声明,例如 “以上为行业通识建议,并不代表本项目已经满足相关条件。”;\n
352
+ - 若[提案事实]中已经给出某一方面的部分信息(例如已有部分证据、对象清单、风险表等),"
353
+ " 只能评价为“现有描述在……方面仍不够细化/缺乏定量对比”,不得一概写成“项目未提供相关分析/对应评估”等绝对否定;\n
354
+ - 不得发明提案中不存在的新实验/新数据/新机构/具体注册号;对未给出的信息,只能以“需补充/需确认”的方式表达;\n
355
+ - 仅输出一个 JSON 对象,不得有额外文字或 Markdown 围栏。
356
+
357
+ {_schema_structured()}
358
+ """.strip()
359
+
360
+
361
+ def build_refine_prompt(candidate_obj: Dict[str, Any], proposal_context: str, dimension: str) -> str:
362
+ original = json.dumps(candidate_obj, ensure_ascii=False, indent=2)
363
+ return f"""
364
+ 请在不引入任何超出【提案事实】的新信息的前提下,对下面的结构化回答做一次快速自我复核:
365
+ - 删改过于武断或缺乏依据的强结论;
366
+ - 若某条结论在【提案事实】中找不到依据,请改写为“需要补充的材料/信息”;
367
+ - 优化 answer 的分点表达,让每条更具体、更可执行,但不要发明新试验/新数据;
368
+ - 检查 general_insights:确保里面只包含“行业基准/类似项目常见做法/常见坑与证据要求”等通识内容,不得把本项目的具体事实写进 general_insights;\n
369
+ - 保持 JSON 结构和字段名完全不变(包括 general_insights)。
370
+
371
+ 维度:{dimension}
372
+
373
+ [提案事实]:
374
+ {proposal_context or "(该维度的提案事实为空,仅可做轻度通识性修正)"}
375
+
376
+ 原候选:
377
+ {original}
378
+ """.strip()
379
+
380
+
381
+ # ========== LLM 调用基础 ==========
382
+ def _with_retry(fn, max_tries: int = 4, base: float = 0.8):
383
+ for i in range(max_tries):
384
+ try:
385
+ return fn()
386
+ except Exception:
387
+ if i == max_tries - 1:
388
+ raise
389
+ time.sleep(base * (2 ** i) + random.random() * 0.2)
390
+
391
+
392
+ ERR_PATTERNS = [
393
+ r"\[?ERROR[:\]]",
394
+ r"\bHTTP\s*4\d{2}\b",
395
+ r"\bHTTP\s*5\d{2}\b",
396
+ r"insufficient[_\s-]?quota",
397
+ r"invalid[_\s-]?api[_\s-]?key",
398
+ r"request\s+timed\s*out",
399
+ r"rate\s*limit",
400
+ r"payment\s*required",
401
+ r"bad gateway",
402
+ r"service unavailable",
403
+ r"connection (?:reset|refused)",
404
+ ]
405
+ _err_re = re.compile("|".join(ERR_PATTERNS), re.I)
406
+
407
+
408
+ def is_error_text(text: str) -> bool:
409
+ t = (text or "").strip()
410
+ if not t:
411
+ return True
412
+ return bool(_err_re.search(t))
413
+
414
+
415
+ def _chat_completion_json(
416
+ client,
417
+ model: str,
418
+ system_text: str,
419
+ user_text: str,
420
+ max_tokens: int,
421
+ temperature: float,
422
+ force_json: bool,
423
+ ):
424
+ def call(json_mode: bool):
425
+ kwargs = dict(
426
+ model=model,
427
+ messages=[{"role": "system", "content": system_text}, {"role": "user", "content": user_text}],
428
+ temperature=float(temperature),
429
+ max_tokens=int(max_tokens),
430
+ )
431
+ if json_mode:
432
+ kwargs["response_format"] = {"type": "json_object"}
433
+ return client.chat.completions.create(**kwargs)
434
+
435
+ try:
436
+ if force_json:
437
+ resp = _with_retry(lambda: call(json_mode=True))
438
+ out = (resp.choices[0].message.content or "").strip()
439
+ if is_error_text(out):
440
+ raise RuntimeError("provider_error_json_mode")
441
+ return out
442
+ else:
443
+ resp = _with_retry(lambda: call(json_mode=False))
444
+ out = (resp.choices[0].message.content or "").strip()
445
+ if is_error_text(out):
446
+ raise RuntimeError("provider_error_text_mode")
447
+ return out
448
+ except Exception:
449
+ if force_json:
450
+ resp = _with_retry(lambda: call(json_mode=False))
451
+ out = (resp.choices[0].message.content or "").strip()
452
+ if is_error_text(out):
453
+ raise RuntimeError("provider_error_text_mode")
454
+ return out
455
+ raise
456
+
457
+
458
+ def _safe_parse_json_plus(txt: str) -> Optional[Any]:
459
+ if is_error_text(txt):
460
+ return None
461
+
462
+ t = (txt or "").replace("\ufeff", "").strip()
463
+ t = re.sub(r"^\s*```(?:json)?\s*\n?", "", t, flags=re.IGNORECASE)
464
+ t = re.sub(r"\n?\s*```\s*$", "", t, flags=re.IGNORECASE)
465
+ t = t.replace("\xa0", " ")
466
+
467
+ try:
468
+ obj = json.loads(t)
469
+ if isinstance(obj, list):
470
+ return {"answers": obj}
471
+ return obj
472
+ except Exception:
473
+ pass
474
+
475
+ m = re.search(r"(\{.*\}|\[.*\])", t, re.S)
476
+ if m:
477
+ frag = m.group(1)
478
+ try:
479
+ tmp = json.loads(frag)
480
+ if isinstance(tmp, list):
481
+ return {"answers": tmp}
482
+ return tmp
483
+ except Exception:
484
+ pass
485
+
486
+ # 尝试给常见字段名补双引号,包括 general_insights
487
+ cand = re.sub(
488
+ r"(\banswer|claims|evidence_hints|general_insights|topic_tags|confidence|caveats\b)\s*:",
489
+ r'"\1":',
490
+ t,
491
+ )
492
+ try:
493
+ return json.loads(cand)
494
+ except Exception:
495
+ return None
496
+
497
+
498
+ # ========== 模型初始化 ==========
499
+ def init_openai():
500
+ if OpenAIClient is None:
501
+ return None, None
502
+ key = os.getenv("OPENAI_API_KEY", "").strip()
503
+ if not key:
504
+ return None, None
505
+ model = os.getenv("OPENAI_MODEL", "gpt-4o-mini").strip()
506
+ client = OpenAIClient(api_key=key)
507
+ print(f"✅ 已加载 OpenAI 模型:{model}")
508
+ return client, model
509
+
510
+
511
+ def init_deepseek():
512
+ if OpenAIClient is None:
513
+ return None, None
514
+ key = os.getenv("DEEPSEEK_API_KEY", "").strip()
515
+ if not key:
516
+ return None, None
517
+ base = os.getenv("DEEPSEEK_API_BASE", "https://api.deepseek.com/v1").strip()
518
+ model = os.getenv("DEEPSEEK_MODEL", "deepseek-chat").strip()
519
+ client = OpenAIClient(api_key=key, base_url=base)
520
+ print(f"✅ 已加载 DeepSeek 模型:{model}")
521
+ return client, model
522
+
523
+
524
+ # ========== 答案规范化 & 后处理 ==========
525
+ def _norm_str(s: Any) -> str:
526
+ return " ".join(str(s or "").strip().split())
527
+
528
+
529
+ def _uniq_cut(lst: List[Any], k: int = MAX_LIST_LEN) -> List[str]:
530
+ seen, out = set(), []
531
+ for x in lst:
532
+ t = _norm_str(x)
533
+ if not t:
534
+ continue
535
+ key = t.casefold()
536
+ if key in seen:
537
+ continue
538
+ seen.add(key)
539
+ out.append(t)
540
+ if len(out) >= k:
541
+ break
542
+ return out
543
+
544
+
545
+ RE_DATE = re.compile(r"\b(20\d{2}|19\d{2})([-/.]|年)\d{1,2}([-/\.日]|月)\d{1,2}\b|\b(Q[1-4]\s*-\s*20\d{2})\b", re.I)
546
+ RE_MONEY = re.compile(
547
+ r"\b(\$|USD|EUR|CNY|RMB|CAD)\s*\d{2,}(,\d{3})*(\.\d+)?\b|\b\d+(\.\d+)?\s*(million|billion|万|亿)\b",
548
+ re.I,
549
+ )
550
+ RE_TRIAL = re.compile(r"\bNCT\d{8}\b|\bEUCTR-\d{4}-\d{6}-\d{2}\b", re.I)
551
+ RE_DOI = re.compile(r"\b10\.\d{4,9}/[-._;()/:A-Z0-9]+\b", re.I)
552
+ RE_PATENT = re.compile(r"\b(US|EP|CN)\d{5,}\b|\bWO\d{7,}\b", re.I)
553
+ RE_ISO = re.compile(r"\bISO\s?\d{4,5}(-\d+)?\b", re.I)
554
+ RE_STDNUM = re.compile(r"\bEN\s?\d{3,5}\b|\bASTM\s?[A-Z]?\d{2,5}\b", re.I)
555
+ RE_ID_ANY = re.compile(r"(注册号|登记号|批准文号|备案号)", re.I)
556
+
557
+
558
+ def _is_redline(text: str) -> bool:
559
+ t = text or ""
560
+ if RE_DATE.search(t):
561
+ return True
562
+ if RE_MONEY.search(t):
563
+ return True
564
+ if RE_TRIAL.search(t):
565
+ return True
566
+ if RE_DOI.search(t):
567
+ return True
568
+ if RE_PATENT.search(t):
569
+ return True
570
+ if RE_ISO.search(t):
571
+ return True
572
+ if RE_STDNUM.search(t):
573
+ return True
574
+ if RE_ID_ANY.search(t):
575
+ return True
576
+ return False
577
+
578
+
579
+ def _scrub_claims_to_hints(claims: List[Any], hints: List[Any]) -> Tuple[List[str], List[str], List[str]]:
580
+ """
581
+ 把明显带编号/金额的句子从 claims 挪到 evidence_hints,并记录到 moved_facts。
582
+ 不再额外注水模板;只做清洗和重分类。
583
+ """
584
+ new_claims: List[str] = []
585
+ new_hints: List[str] = [str(x).strip() for x in (hints or []) if str(x).strip()]
586
+ moved_facts: List[str] = []
587
+
588
+ for c in claims or []:
589
+ s = _norm_str(c)
590
+ if not s:
591
+ continue
592
+ if _is_redline(s):
593
+ new_hints.append(s)
594
+ moved_facts.append(s)
595
+ else:
596
+ new_claims.append(s)
597
+
598
+ return _uniq_cut(new_claims, MAX_LIST_LEN), _uniq_cut(new_hints, MAX_LIST_LEN), _uniq_cut(moved_facts, MAX_LIST_LEN)
599
+
600
+
601
+ def _to_bullets(answer: str) -> Tuple[str, int]:
602
+ """
603
+ 把任意文本整理成 1. 2. 3. 形式的分点;不过度强制数量,只要 >=1 即可。
604
+ 修复双重编号问题:无论是按行还是按句拆分,都会先去掉已有编号/符号再统一加“1. 2. 3.”。
605
+ """
606
+ raw = (answer or "").strip()
607
+ if not raw:
608
+ return "", 0
609
+
610
+ text = raw.replace("\r\n", "\n").strip()
611
+
612
+ def _normalize_item(line: str) -> str:
613
+ s = line.strip()
614
+ # 去掉 Markdown 项目符号
615
+ s = re.sub(r"^\s*[-*•●▪]+\s*", "", s)
616
+ # 去掉前导数字编号(1. / 1) / 1、 等)
617
+ s = re.sub(r"^\s*\d+[\.\).、]\s*", "", s)
618
+ # 压缩空白
619
+ s = re.sub(r"\s+", " ", s)
620
+ s = s.strip()
621
+ if not s:
622
+ return ""
623
+
624
+ # 丢掉只剩数字或“数字+空格”的内容(例如 "1"、"2"、"1 2" 等)
625
+ s_nospace = s.replace(" ", "")
626
+ if s_nospace.isdigit() and len(s_nospace) <= 4:
627
+ return ""
628
+
629
+ return s
630
+
631
+ # 先按行拆分(利用 LLM 原始的换行结构)
632
+ lines = [ln for ln in text.split("\n") if ln.strip()]
633
+ items: List[str] = []
634
+ for ln in lines:
635
+ norm = _normalize_item(ln)
636
+ if norm:
637
+ items.append(norm)
638
+
639
+ # 如果按行只有 0–1 条,说明可能是整段一坨 -> 再按句号/分号拆一轮
640
+ if len(items) <= 1:
641
+ chunks = re.split(r"[。;;.!??]\s*", text)
642
+ items = []
643
+ for ch in chunks:
644
+ norm = _normalize_item(ch)
645
+ if norm:
646
+ items.append(norm)
647
+
648
+ if not items:
649
+ # 实在拆不出有效内容,就保留原文
650
+ return raw, 0
651
+
652
+ # 控制条数上限,避免 answer 过长
653
+ items = items[:8]
654
+
655
+ numbered = [f"{i+1}. {seg}" for i, seg in enumerate(items)]
656
+ return "\n".join(numbered), len(items)
657
+
658
+ def _calibrate_conf(x: Any) -> float:
659
+ try:
660
+ v = float(x)
661
+ except Exception:
662
+ v = 0.65
663
+ if v < CONF_MIN:
664
+ v = CONF_MIN
665
+ if v > CONF_MAX:
666
+ v = CONF_MAX
667
+ return round(v, 2)
668
+
669
+
670
+ def _normalize_candidate_obj(obj: Any) -> Dict[str, Any]:
671
+ if not isinstance(obj, dict):
672
+ obj = {}
673
+ out: Dict[str, Any] = {}
674
+
675
+ # ❗ 保留原始换行,不再用 _norm_str 压缩,避免破坏 LLM 已经分好的行
676
+ raw_answer = obj.get("answer", "")
677
+
678
+ # ====== 新增:专门处理 list / dict 形式的 answer,避免 json.dumps 成一坨 ======
679
+ if isinstance(raw_answer, list):
680
+ # LLM 有时会给 ["1\n2. ...", "3\n4. ..."] 这种
681
+ pieces: List[str] = []
682
+ for elem in raw_answer:
683
+ if elem is None:
684
+ continue
685
+ # 再兜一层 list(极端情况)
686
+ if isinstance(elem, list):
687
+ for sub in elem:
688
+ s = str(sub or "").strip()
689
+ if s:
690
+ pieces.append(s)
691
+ else:
692
+ s = str(elem or "").strip()
693
+ if s:
694
+ pieces.append(s)
695
+ raw_answer = "\n".join(pieces) # 变成多行文本,交给 _to_bullets 按行拆
696
+
697
+ elif isinstance(raw_answer, dict):
698
+ # 如果以后 LLM 返回 {"bullets":[...]} 之类,优先拉出里面的 list
699
+ for key in ("bullets", "points", "items"):
700
+ val = raw_answer.get(key)
701
+ if isinstance(val, list):
702
+ pieces = [str(x or "").strip() for x in val if str(x or "").strip()]
703
+ raw_answer = "\n".join(pieces)
704
+ break
705
+ else:
706
+ # 实在没有结构化 list,再兜底 dump 成字符串
707
+ raw_answer = json.dumps(raw_answer, ensure_ascii=False)
708
+
709
+ # 其他情况:本来就是字符串/数字,直接转成字符串
710
+ out["answer"] = str(raw_answer or "")
711
+
712
+ raw_claims = obj.get("claims", [])
713
+ raw_hints = obj.get("evidence_hints", [])
714
+ raw_tags = obj.get("topic_tags", [])
715
+ raw_gi = obj.get("general_insights", [])
716
+
717
+ if isinstance(raw_claims, (str, int, float)):
718
+ raw_claims = [raw_claims]
719
+ if isinstance(raw_hints, (str, int, float)):
720
+ raw_hints = [raw_hints]
721
+ if isinstance(raw_tags, (str, int, float)):
722
+ raw_tags = [raw_tags]
723
+ if isinstance(raw_gi, (str, int, float)):
724
+ raw_gi = [raw_gi]
725
+
726
+ out["claims"] = [str(x).strip() for x in (raw_claims or []) if str(x).strip()]
727
+ out["evidence_hints"] = [str(x).strip() for x in (raw_hints or []) if str(x).strip()]
728
+ out["topic_tags"] = [str(x).strip() for x in (raw_tags or []) if str(x).strip()]
729
+ out["general_insights"] = [str(x).strip() for x in (raw_gi or []) if str(x).strip()]
730
+ try:
731
+ out["confidence"] = float(obj.get("confidence", 0.65))
732
+ except Exception:
733
+ out["confidence"] = 0.65
734
+ out["caveats"] = _norm_str(obj.get("caveats", ""))
735
+ return out
736
+
737
+
738
+ def _validate_candidate_dict(obj: Any) -> bool:
739
+ if not isinstance(obj, dict):
740
+ return False
741
+ if not isinstance(obj.get("answer"), str) or not obj.get("answer").strip():
742
+ return False
743
+ if not isinstance(obj.get("claims"), list):
744
+ return False
745
+ if not isinstance(obj.get("evidence_hints"), list):
746
+ return False
747
+ if not isinstance(obj.get("topic_tags"), list):
748
+ return False
749
+ if not isinstance(obj.get("general_insights"), list):
750
+ return False
751
+ return True
752
+
753
+
754
+ def _build_topic_tags(tags: List[Any], dimension: str, answer: str) -> List[str]:
755
+ base = [str(t).lower().strip() for t in (tags or []) if str(t).strip()]
756
+ extra: List[str] = []
757
+
758
+ dim_token = dimension.lower().strip()
759
+ if dim_token:
760
+ extra.append(dim_token)
761
+
762
+ words = re.findall(r"[A-Za-z]+|[\u4e00-\u9fff]{2,8}", answer)
763
+ freq: Dict[str, int] = {}
764
+ for w in words:
765
+ wl = w.lower()
766
+ if len(wl) < 2:
767
+ continue
768
+ freq[wl] = freq.get(wl, 0) + 1
769
+ common = sorted(freq.items(), key=lambda kv: kv[1], reverse=True)[:6]
770
+ extra.extend([w for w, _ in common])
771
+
772
+ return _uniq_cut(base + extra, MAX_LIST_LEN)
773
+
774
+
775
+ def _quick_score(cand: Dict[str, Any]) -> Dict[str, Any]:
776
+ """
777
+ 给后续选优一个简单分数:
778
+ - 分点条数
779
+ - claims 数量
780
+ - confidence
781
+ (general_insights 目前只在下游使用,这里不额外打分,避免逻辑过重)
782
+ """
783
+ bullets = len([ln for ln in str(cand.get("answer", "")).splitlines() if ln.strip()])
784
+ claims_n = len(cand.get("claims") or [])
785
+ conf = float(cand.get("confidence", 0.65))
786
+
787
+ score = 0.0
788
+ if bullets >= 3:
789
+ score += 0.25
790
+ if 3 <= bullets <= 8:
791
+ score += 0.20
792
+ if claims_n >= 2:
793
+ score += 0.25
794
+ if claims_n >= 4:
795
+ score += 0.10
796
+ score += max(0.0, min(0.20, (conf - CONF_MIN) / (CONF_MAX - CONF_MIN + 1e-6) * 0.20))
797
+
798
+ cand["quick_score"] = round(score, 3)
799
+ return cand
800
+
801
+
802
+ def _finalize_candidate(
803
+ obj: Dict[str, Any],
804
+ provider: str,
805
+ model: str,
806
+ variant_id: str,
807
+ sample_id: int,
808
+ dimension: str,
809
+ ) -> Dict[str, Any]:
810
+ base = _normalize_candidate_obj(obj)
811
+ answer_bullets, n_bullets = _to_bullets(base["answer"])
812
+ base["answer"] = answer_bullets
813
+
814
+ new_claims, new_hints, moved_facts = _scrub_claims_to_hints(base.get("claims", []), base.get("evidence_hints", []))
815
+ base["claims"] = new_claims
816
+ base["evidence_hints"] = new_hints
817
+ base["facts_redlined"] = moved_facts
818
+
819
+ # general_insights 只做去重+截断,不做红线搬移(允许包含“通常需要 NCT 编号”这类行业通识)
820
+ base["general_insights"] = _uniq_cut(base.get("general_insights", []), MAX_LIST_LEN)
821
+
822
+ base["topic_tags"] = _build_topic_tags(base.get("topic_tags", []), dimension, base["answer"])
823
+ base["confidence"] = _calibrate_conf(base.get("confidence", 0.65))
824
+
825
+ if not base.get("caveats"):
826
+ base["caveats"] = "结论需结合原始提案全文与支撑材料进一步核查。"
827
+
828
+ base["provider"] = provider
829
+ base["model"] = model
830
+ base["variant_id"] = variant_id
831
+ base["sample_id"] = int(sample_id)
832
+ base["generated_at"] = now_str()
833
+
834
+ base["diag"] = {
835
+ "bullet_count": n_bullets,
836
+ "claims_count": len(base["claims"]),
837
+ "hints_count": len(base["evidence_hints"]),
838
+ "general_insights_count": len(base["general_insights"]),
839
+ }
840
+
841
+ base = _quick_score(base)
842
+ return base
843
+
844
+
845
+ def _simhash_key(s: str) -> str:
846
+ s = re.sub(r"\s+", " ", (s or "").strip().lower())
847
+ return s[:256]
848
+
849
+
850
+ def dedup_nearby(candidates: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
851
+ seen = set()
852
+ kept: List[Dict[str, Any]] = []
853
+ for c in candidates:
854
+ key = _simhash_key(c.get("answer", ""))
855
+ if not key:
856
+ continue
857
+ if key in seen:
858
+ continue
859
+ seen.add(key)
860
+ kept.append(c)
861
+ return kept
862
+
863
+
864
+ # ========== 问答主逻辑(batch / single) ==========
865
+ def ask_model_batch(
866
+ provider: str,
867
+ client,
868
+ model: str,
869
+ dimension: str,
870
+ q_list: List[str],
871
+ proposal_context: str,
872
+ reg_hints: List[str],
873
+ variant_id: str,
874
+ max_tokens: int,
875
+ ) -> List[Dict[str, Any]]:
876
+ """
877
+ 对支持 batch 的模型:同一维度下多题一起问;
878
+ 若解析失败或数组长度不对,由上层改走逐题兜底。
879
+ 返回:长度 == len(q_list) 的 candidate 列表。
880
+ """
881
+ prompt = build_batch_prompt(dimension, q_list, proposal_context, reg_hints, variant_id)
882
+ try:
883
+ txt = _chat_completion_json(
884
+ client,
885
+ model,
886
+ SYSTEM_CN,
887
+ prompt,
888
+ max_tokens=max_tokens,
889
+ temperature=TEMP_BY_VARIANT.get(variant_id, 0.3),
890
+ force_json=provider_caps(provider)["json_mode"],
891
+ )
892
+ except Exception:
893
+ return []
894
+
895
+ low = (txt or "").strip().lower()
896
+ if any(k in low for k in ("incorrect api key", "invalid api key", "rate limit", "quota", "access denied")):
897
+ return []
898
+
899
+ obj = _safe_parse_json_plus(txt)
900
+ if not isinstance(obj, dict):
901
+ return []
902
+ arr = obj.get("answers")
903
+ if not isinstance(arr, list) or len(arr) != len(q_list):
904
+ return []
905
+
906
+ out: List[Dict[str, Any]] = []
907
+ for idx, it in enumerate(arr, 1):
908
+ cand_norm = _normalize_candidate_obj(it)
909
+ if not _validate_candidate_dict(cand_norm):
910
+ out.append({})
911
+ continue
912
+ finalized = _finalize_candidate(
913
+ cand_norm,
914
+ provider=provider,
915
+ model=model,
916
+ variant_id=variant_id,
917
+ sample_id=idx,
918
+ dimension=dimension,
919
+ )
920
+ out.append(finalized)
921
+ return out
922
+
923
+
924
+ def ask_model_single(
925
+ provider: str,
926
+ client,
927
+ model: str,
928
+ dimension: str,
929
+ question: str,
930
+ proposal_context: str,
931
+ reg_hints: List[str],
932
+ variant_id: str,
933
+ max_tokens: int,
934
+ ) -> Dict[str, Any]:
935
+ prompt = build_single_prompt(dimension, question, proposal_context, reg_hints, variant_id)
936
+ try:
937
+ txt = _chat_completion_json(
938
+ client,
939
+ model,
940
+ SYSTEM_CN,
941
+ prompt,
942
+ max_tokens=max_tokens,
943
+ temperature=TEMP_BY_VARIANT.get(variant_id, 0.3),
944
+ force_json=provider_caps(provider)["json_mode"],
945
+ )
946
+ except Exception:
947
+ return {"error": True, "answer": ""}
948
+
949
+ obj = _safe_parse_json_plus(txt)
950
+ if not isinstance(obj, dict):
951
+ return {"error": True, "answer": ""}
952
+
953
+ cand_norm = _normalize_candidate_obj(obj)
954
+ if not _validate_candidate_dict(cand_norm):
955
+ return {"error": True, "answer": ""}
956
+
957
+ return _finalize_candidate(
958
+ cand_norm,
959
+ provider=provider,
960
+ model=model,
961
+ variant_id=variant_id,
962
+ sample_id=1,
963
+ dimension=dimension,
964
+ )
965
+
966
+
967
+ def refine_candidate(
968
+ candidate: Dict[str, Any],
969
+ client,
970
+ model: str,
971
+ dimension: str,
972
+ proposal_context: str,
973
+ provider: str,
974
+ max_tokens: int = 600,
975
+ ) -> Dict[str, Any]:
976
+ try:
977
+ rp = build_refine_prompt(candidate, proposal_context, dimension)
978
+ txt = _chat_completion_json(
979
+ client,
980
+ model,
981
+ SYSTEM_CN,
982
+ rp,
983
+ max_tokens=max_tokens,
984
+ temperature=0.2,
985
+ force_json=provider_caps(provider)["json_mode"],
986
+ )
987
+ obj = _safe_parse_json_plus(txt)
988
+ if isinstance(obj, dict):
989
+ cand_norm = _normalize_candidate_obj(obj)
990
+ if not _validate_candidate_dict(cand_norm):
991
+ return candidate
992
+ return _finalize_candidate(
993
+ cand_norm,
994
+ provider=candidate.get("provider", provider),
995
+ model=candidate.get("model", model),
996
+ variant_id=candidate.get("variant_id", "default"),
997
+ sample_id=candidate.get("sample_id", 1),
998
+ dimension=dimension,
999
+ )
1000
+ return candidate
1001
+ except Exception:
1002
+ return candidate
1003
+
1004
+
1005
+ # ========== 维度级问答 ==========
1006
+ def print_dim_banner(provider_name: str, dim: str, total: int, mode: str):
1007
+ bar = "=" * 12
1008
+ print(f"\n{bar} [{provider_name}] 维度:{dim} | 题目数:{total} | 模式:{mode} {bar}", flush=True)
1009
+
1010
+
1011
+ def print_q_progress(provider_name: str, dim: str, idx: int, total: int, qtext: str):
1012
+ preview = qtext.strip().replace("\n", " ")
1013
+ if len(preview) > 80:
1014
+ preview = preview[:80] + "..."
1015
+ print(f"[{provider_name}] ({dim}) Q{idx}/{total} ▶ {preview}", flush=True)
1016
+
1017
+
1018
+ def chunked(lst: List[Any], n: int):
1019
+ for i in range(0, len(lst), n):
1020
+ yield i, lst[i : i + n]
1021
+
1022
+
1023
+ def answer_dimension(
1024
+ provider: str,
1025
+ client,
1026
+ model_name: str,
1027
+ dim: str,
1028
+ q_list: List[str],
1029
+ proposal_context: str,
1030
+ reg_hints: List[str],
1031
+ refine: bool,
1032
+ group_size: int,
1033
+ max_tokens: int,
1034
+ ) -> List[Dict[str, Any]]:
1035
+ """
1036
+ 返回:list[
1037
+ {
1038
+ "dimension": dim,
1039
+ "q_index": idx,
1040
+ "question": q,
1041
+ "candidates": [candidate_obj, ...]
1042
+ }, ...
1043
+ ]
1044
+ """
1045
+ out_items: List[Dict[str, Any]] = []
1046
+ provider_name = "ChatGPT" if provider == "openai" else "DeepSeek"
1047
+
1048
+ if not q_list:
1049
+ return out_items
1050
+
1051
+ caps = provider_caps(provider)
1052
+ supports_batch = bool(caps.get("batch_ok", True))
1053
+
1054
+ mode = "批量+变体" if supports_batch else "逐题+变体"
1055
+ print_dim_banner(provider_name, dim, len(q_list), mode)
1056
+
1057
+ # 不支持 batch:逐题 × 多变体
1058
+ if not supports_batch:
1059
+ for idx, q in enumerate(q_list, 1):
1060
+ print_q_progress(provider_name, dim, idx, len(q_list), q)
1061
+ raw_cands: List[Dict[str, Any]] = []
1062
+ for v in VARIANTS:
1063
+ cand = ask_model_single(
1064
+ provider=provider,
1065
+ client=client,
1066
+ model=model_name,
1067
+ dimension=dim,
1068
+ question=q,
1069
+ proposal_context=proposal_context,
1070
+ reg_hints=reg_hints,
1071
+ variant_id=v,
1072
+ max_tokens=min(900, max_tokens),
1073
+ )
1074
+ if isinstance(cand, dict) and not cand.get("error") and cand.get("answer", "").strip():
1075
+ raw_cands.append(cand)
1076
+ cands = dedup_nearby(raw_cands)
1077
+ if refine and cands:
1078
+ cands = [
1079
+ refine_candidate(
1080
+ c,
1081
+ client=client,
1082
+ model=model_name,
1083
+ dimension=dim,
1084
+ proposal_context=proposal_context,
1085
+ provider=provider,
1086
+ max_tokens=min(700, max_tokens),
1087
+ )
1088
+ for c in cands
1089
+ ]
1090
+ for c in cands:
1091
+ c["dimension"] = dim
1092
+ c["q_index"] = idx
1093
+
1094
+ out_items.append(
1095
+ {
1096
+ "dimension": dim,
1097
+ "q_index": idx,
1098
+ "question": q,
1099
+ "candidates": cands,
1100
+ }
1101
+ )
1102
+ return out_items
1103
+
1104
+ # 支持 batch:按 group_size 分批 + 变体
1105
+ group_size = max(1, min(int(group_size), 4))
1106
+ for start_idx, sub_qs in chunked(q_list, group_size):
1107
+ batch_tag = f"{start_idx+1}-{start_idx+len(sub_qs)}"
1108
+ per_variant_results: Dict[str, List[Dict[str, Any]]] = {}
1109
+ batch_failed = False
1110
+
1111
+ for v in VARIANTS:
1112
+ t0 = time.time()
1113
+ arr = ask_model_batch(
1114
+ provider=provider,
1115
+ client=client,
1116
+ model=model_name,
1117
+ dimension=dim,
1118
+ q_list=sub_qs,
1119
+ proposal_context=proposal_context,
1120
+ reg_hints=reg_hints,
1121
+ variant_id=v,
1122
+ max_tokens=max_tokens,
1123
+ )
1124
+ ok = bool(arr) and len(arr) == len(sub_qs)
1125
+ print(
1126
+ f"[{provider_name}] ({dim}) 小批 {batch_tag} · {v:<13} 返回 {len(arr)}/{len(sub_qs)} 条,用时 {time.time()-t0:.1f}s({'OK' if ok else 'FAIL→逐题'})"
1127
+ )
1128
+ if not ok:
1129
+ batch_failed = True
1130
+ break
1131
+ per_variant_results[v] = arr
1132
+
1133
+ if not batch_failed:
1134
+ # 合成每题的 candidates
1135
+ for j, q in enumerate(sub_qs, 1):
1136
+ global_idx = start_idx + j
1137
+ raw_cands = [per_variant_results[v][j - 1] for v in VARIANTS]
1138
+ cands = [c for c in raw_cands if isinstance(c, dict) and c.get("answer", "").strip()]
1139
+ cands = dedup_nearby(cands)
1140
+ if refine and cands:
1141
+ cands = [
1142
+ refine_candidate(
1143
+ c,
1144
+ client=client,
1145
+ model=model_name,
1146
+ dimension=dim,
1147
+ proposal_context=proposal_context,
1148
+ provider=provider,
1149
+ max_tokens=min(700, max_tokens),
1150
+ )
1151
+ for c in cands
1152
+ ]
1153
+ for c in cands:
1154
+ c["dimension"] = dim
1155
+ c["q_index"] = global_idx
1156
+ out_items.append(
1157
+ {
1158
+ "dimension": dim,
1159
+ "q_index": global_idx,
1160
+ "question": q,
1161
+ "candidates": cands,
1162
+ }
1163
+ )
1164
+ continue
1165
+
1166
+ # 批量失败:这一小批改逐题
1167
+ for j, q in enumerate(sub_qs, 1):
1168
+ global_idx = start_idx + j
1169
+ print_q_progress(provider_name, dim, global_idx, len(q_list), q)
1170
+ raw_cands: List[Dict[str, Any]] = []
1171
+ for v in VARIANTS:
1172
+ cand = ask_model_single(
1173
+ provider=provider,
1174
+ client=client,
1175
+ model=model_name,
1176
+ dimension=dim,
1177
+ question=q,
1178
+ proposal_context=proposal_context,
1179
+ reg_hints=reg_hints,
1180
+ variant_id=v,
1181
+ max_tokens=min(900, max_tokens),
1182
+ )
1183
+ if isinstance(cand, dict) and not cand.get("error") and cand.get("answer", "").strip():
1184
+ raw_cands.append(cand)
1185
+ cands = dedup_nearby(raw_cands)
1186
+ if refine and cands:
1187
+ cands = [
1188
+ refine_candidate(
1189
+ c,
1190
+ client=client,
1191
+ model=model_name,
1192
+ dimension=dim,
1193
+ proposal_context=proposal_context,
1194
+ provider=provider,
1195
+ max_tokens=min(700, max_tokens),
1196
+ )
1197
+ for c in cands
1198
+ ]
1199
+ for c in cands:
1200
+ c["dimension"] = dim
1201
+ c["q_index"] = global_idx
1202
+ out_items.append(
1203
+ {
1204
+ "dimension": dim,
1205
+ "q_index": global_idx,
1206
+ "question": q,
1207
+ "candidates": cands,
1208
+ }
1209
+ )
1210
+
1211
+ return out_items
1212
+
1213
+
1214
+ # ========== 多模型合并 ==========
1215
+ def merge_two_models(chatgpt_items: List[Dict[str, Any]], deepseek_items: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
1216
+ def key_of(x: Dict[str, Any]):
1217
+ return (x["dimension"], x["q_index"], x["question"])
1218
+
1219
+ pool: Dict[Any, Dict[str, Any]] = {}
1220
+ for it in chatgpt_items:
1221
+ pool[key_of(it)] = {
1222
+ "dimension": it["dimension"],
1223
+ "q_index": it["q_index"],
1224
+ "question": it["question"],
1225
+ "candidates": list(it.get("candidates", [])),
1226
+ }
1227
+ for it in deepseek_items:
1228
+ k = key_of(it)
1229
+ if k in pool:
1230
+ pool[k]["candidates"].extend(it.get("candidates", []))
1231
+ else:
1232
+ pool[k] = {
1233
+ "dimension": it["dimension"],
1234
+ "q_index": it["q_index"],
1235
+ "question": it["question"],
1236
+ "candidates": list(it.get("candidates", [])),
1237
+ }
1238
+
1239
+ items = list(pool.values())
1240
+ items.sort(
1241
+ key=lambda x: (
1242
+ DIM_ORDER.index(x["dimension"]) if x["dimension"] in DIM_ORDER else 99,
1243
+ x["q_index"],
1244
+ )
1245
+ )
1246
+ return items
1247
+
1248
+
1249
+ # ========== CLI ==========
1250
+ def parse_args():
1251
+ ap = argparse.ArgumentParser(
1252
+ description="Proposal-aware Dual-LLM Answering (ChatGPT + DeepSeek, no web search, dimension-facts aware)"
1253
+ )
1254
+ ap.add_argument(
1255
+ "--proposal-id",
1256
+ type=str,
1257
+ default="",
1258
+ help="指定提案 ID;不填则自动检测 data/extracted 最近目录名",
1259
+ )
1260
+ ap.add_argument(
1261
+ "--qs-file",
1262
+ type=str,
1263
+ default=str(CONFIG_QS_DEFAULT),
1264
+ help="问题集 JSON 路径(通常由 generate_questions.py 生成)",
1265
+ )
1266
+ ap.add_argument(
1267
+ "--dim-file",
1268
+ type=str,
1269
+ default="",
1270
+ help="维度 JSON 路径;默认使用 data/extracted/{pid}/dimensions_v2.json",
1271
+ )
1272
+ ap.add_argument(
1273
+ "--refine",
1274
+ type=int,
1275
+ default=1,
1276
+ help="是否进行轻量自我复核(0/1)",
1277
+ )
1278
+ ap.add_argument(
1279
+ "--max_tokens",
1280
+ type=int,
1281
+ default=2200,
1282
+ )
1283
+ ap.add_argument(
1284
+ "--group-size",
1285
+ type=int,
1286
+ default=3,
1287
+ help="每批问题数(支持 batch 的模型才生效,建议 2–4)",
1288
+ )
1289
+ ap.add_argument(
1290
+ "--seed",
1291
+ type=int,
1292
+ default=None,
1293
+ )
1294
+ return ap.parse_args()
1295
+
1296
+
1297
+ def main():
1298
+ args = parse_args()
1299
+ if args.seed is not None:
1300
+ try:
1301
+ random.seed(int(args.seed))
1302
+ except Exception:
1303
+ pass
1304
+
1305
+ pid = args.proposal_id.strip() or detect_latest_pid()
1306
+ if pid == "unknown":
1307
+ print("⚠️ 未检测到 data/extracted 下的提案目录,将使用占位 pid=unknown。")
1308
+ print(f"🧩 proposal_id = {pid}")
1309
+
1310
+ qs_path = Path(args.qs_file)
1311
+ if not qs_path.exists():
1312
+ raise FileNotFoundError(f"未找到问题集:{qs_path}")
1313
+ qs_cfg = read_json(qs_path)
1314
+
1315
+ missing = [d for d in DIM_ORDER if not get_q_list(qs_cfg.get(d, []))]
1316
+ if missing:
1317
+ raise RuntimeError(
1318
+ f"问题集缺少维度或该维度题目为空:{missing}(请检查 {qs_path} 或 generate_questions.py 输出)"
1319
+ )
1320
+
1321
+ # ========= 维度上下文:固定读取 extracted/{pid}/dimensions_v2.json =========
1322
+ if args.dim_file.strip():
1323
+ dim_file = Path(args.dim_file.strip())
1324
+ else:
1325
+ dim_file = EXTRACTED_DIR / pid / "dimensions_v2.json"
1326
+
1327
+ if not dim_file.exists():
1328
+ raise FileNotFoundError(
1329
+ f"未找到维度文件:{dim_file} ;请先运行 build_dimensions_from_facts.py 生成 dimensions_v2.json"
1330
+ )
1331
+
1332
+ print(f"📄 使用维度上下文文件:{dim_file}")
1333
+ dim_context_map = load_dimension_context(pid, dim_file)
1334
+
1335
+ domain_profile = load_domain_profile(EXTRACTED_DIR / pid / "domain_profile.json")
1336
+
1337
+ reg_hints_map: Dict[str, List[str]] = {}
1338
+ terminology = domain_profile.get("terminology", []) if isinstance(domain_profile, dict) else []
1339
+ methods = domain_profile.get("methods", []) if isinstance(domain_profile, dict) else []
1340
+ risks = domain_profile.get("risks", []) if isinstance(domain_profile, dict) else []
1341
+ domain_hints = [*terminology[:4], *methods[:2], *risks[:2]]
1342
+ for dim in DIM_ORDER:
1343
+ reg_hints_map[dim] = _load_reg_hints(qs_cfg, dim, limit=8)
1344
+ reg_hints_map[dim] = list(dict.fromkeys(reg_hints_map[dim] + domain_hints))[:8]
1345
+
1346
+ oa_client, oa_model = init_openai()
1347
+ ds_client, ds_model = init_deepseek()
1348
+ if (oa_client is None or oa_model is None) and (ds_client is None or ds_model is None):
1349
+ raise RuntimeError("未检测到可用模型:请在 .env 配置 OPENAI_API_KEY 与/或 DEEPSEEK_API_KEY。")
1350
+
1351
+ dims = DIM_ORDER[:]
1352
+ chatgpt_items_all: List[Dict[str, Any]] = []
1353
+ deepseek_items_all: List[Dict[str, Any]] = []
1354
+
1355
+ # ChatGPT
1356
+ if oa_client and oa_model:
1357
+ print("🧠 ChatGPT 答题中 ...")
1358
+ for dim in dims:
1359
+ q_list = get_q_list(qs_cfg.get(dim, []))
1360
+ ctx = dim_context_map.get(dim, "")
1361
+ reg_hints = reg_hints_map.get(dim, [])
1362
+ items = answer_dimension(
1363
+ provider="openai",
1364
+ client=oa_client,
1365
+ model_name=oa_model,
1366
+ dim=dim,
1367
+ q_list=q_list,
1368
+ proposal_context=ctx,
1369
+ reg_hints=reg_hints,
1370
+ refine=bool(args.refine),
1371
+ group_size=int(args.group_size),
1372
+ max_tokens=int(args.max_tokens),
1373
+ )
1374
+ chatgpt_items_all.extend(items)
1375
+
1376
+ out_path = OUT_REFINED / pid / "chatgpt_raw.json"
1377
+ write_json(
1378
+ out_path,
1379
+ {
1380
+ "meta": {
1381
+ "model": oa_model,
1382
+ "provider": "openai",
1383
+ "generated_at": now_str(),
1384
+ "pid": pid,
1385
+ },
1386
+ "items": chatgpt_items_all,
1387
+ },
1388
+ )
1389
+ print(f"✅ ChatGPT 结果 -> {out_path}")
1390
+ else:
1391
+ print("⚠️ 跳过 ChatGPT:未配置 OPENAI_API_KEY。")
1392
+
1393
+ # DeepSeek
1394
+ if ds_client and ds_model:
1395
+ print("🧠 DeepSeek 答题中 ...")
1396
+ for dim in dims:
1397
+ q_list = get_q_list(qs_cfg.get(dim, []))
1398
+ ctx = dim_context_map.get(dim, "")
1399
+ reg_hints = reg_hints_map.get(dim, [])
1400
+ items = answer_dimension(
1401
+ provider="deepseek",
1402
+ client=ds_client,
1403
+ model_name=ds_model,
1404
+ dim=dim,
1405
+ q_list=q_list,
1406
+ proposal_context=ctx,
1407
+ reg_hints=reg_hints,
1408
+ refine=bool(args.refine),
1409
+ group_size=int(args.group_size),
1410
+ max_tokens=int(args.max_tokens),
1411
+ )
1412
+ deepseek_items_all.extend(items)
1413
+
1414
+ out_path = OUT_REFINED / pid / "deepseek_raw.json"
1415
+ write_json(
1416
+ out_path,
1417
+ {
1418
+ "meta": {
1419
+ "model": ds_model,
1420
+ "provider": "deepseek",
1421
+ "generated_at": now_str(),
1422
+ "pid": pid,
1423
+ },
1424
+ "items": deepseek_items_all,
1425
+ },
1426
+ )
1427
+ print(f"✅ DeepSeek 结果 -> {out_path}")
1428
+ else:
1429
+ print("⚠️ 跳过 DeepSeek:未配置 DEEPSEEK_API_KEY。")
1430
+
1431
+ merged_items = merge_two_models(chatgpt_items_all, deepseek_items_all)
1432
+ merged = {
1433
+ "meta": {
1434
+ "pid": pid,
1435
+ "generated_at": now_str(),
1436
+ "schema": "refined_items.v2.proposal_aware_with_general_insights",
1437
+ "args": {
1438
+ "refine": bool(args.refine),
1439
+ "max_tokens": int(args.max_tokens),
1440
+ "group_size": int(args.group_size),
1441
+ },
1442
+ "models": {
1443
+ "chatgpt": {"model": oa_model, "provider": "openai"} if chatgpt_items_all else None,
1444
+ "deepseek": {"model": ds_model, "provider": "deepseek"} if deepseek_items_all else None,
1445
+ },
1446
+ },
1447
+ "items": merged_items,
1448
+ }
1449
+ out_path = OUT_REFINED / pid / "all_refined_items.json"
1450
+ write_json(out_path, merged)
1451
+ print(f"📦 合并结果 -> {out_path}")
1452
+ print("🎯 完成。")
1453
+
1454
+
1455
+ if __name__ == "__main__": # pragma: no cover
1456
+ main()
src/tools/metric_checker.py ADDED
@@ -0,0 +1,218 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ from __future__ import annotations
3
+
4
+ import re
5
+ from collections import Counter
6
+ from typing import Any, Dict, List, Sequence
7
+
8
+
9
+ def _clean_text(text: str) -> str:
10
+ return re.sub(r"\s+", " ", text or "").strip()
11
+
12
+
13
+ def _unique_keep_order(items: List[str], max_items: int = 20) -> List[str]:
14
+ seen = set()
15
+ out = []
16
+ for item in items:
17
+ key = item.strip().lower()
18
+ if not key or key in seen:
19
+ continue
20
+ seen.add(key)
21
+ out.append(item.strip())
22
+ if len(out) >= max_items:
23
+ break
24
+ return out
25
+
26
+
27
+ def extract_numeric_spans(text: str) -> List[str]:
28
+ """
29
+ Extract common quantitative expressions from proposal text.
30
+ Covers:
31
+ - integers / decimals
32
+ - percentages
33
+ - money / financing
34
+ - years / year-ranges
35
+ - count-like targets
36
+ """
37
+ text = _clean_text(text)
38
+
39
+ patterns = [
40
+ r"\d+(?:\.\d+)?%", # 50%
41
+ r"\d+(?:\.\d+)?\s*%", # 50 %
42
+ r"\d+(?:\.\d+)?万", # 5500万
43
+ r"\d+(?:\.\d+)?亿", # 3亿
44
+ r"\d+(?:\.\d+)?万元", # 500万元
45
+ r"\d+(?:\.\d+)?亿元", # 3亿元
46
+ r"\d{4}\s*[-–—]\s*\d{4}", # 2018-2020
47
+ r"\d{4}年(?:\d{1,2}月)?", # 2019年 / 2019年10月
48
+ r"第[一二三四五六七八九十0-9]+年", # 第一年
49
+ r"\d+(?:\.\d+)?倍", # 2倍
50
+ r"\d+(?:\.\d+)?种", # 20种
51
+ r"\d+(?:\.\d+)?个", # 4个
52
+ r"\d+(?:\.\d+)?项", # 70项
53
+ r"\d+(?:\.\d+)?辆", # 5万辆
54
+ r"\d+(?:\.\d+)?家", # 10家
55
+ r"\d+(?:\.\d+)?台", # 3台
56
+ r"\d+(?:\.\d+)?套", # 4套
57
+ r"\d+(?:\.\d+)?人", # 20人
58
+ r"\d+(?:\.\d+)?次", # 3次
59
+ r"\d+(?:\.\d+)?(?:\.\d+)?", # fallback numbers
60
+ ]
61
+
62
+ spans: List[str] = []
63
+ for pat in patterns:
64
+ spans.extend(re.findall(pat, text))
65
+
66
+ return _unique_keep_order(spans, max_items=40)
67
+
68
+
69
+ def detect_metric_signals(text: str) -> Dict[str, Any]:
70
+ """
71
+ Detect whether the document contains useful measurable evidence.
72
+ """
73
+ text = _clean_text(text)
74
+ lower = text.lower()
75
+
76
+ numeric_spans = extract_numeric_spans(text)
77
+
78
+ signal_patterns = {
79
+ "has_numbers": r"\d",
80
+ "has_money": r"(融资|预算|资金|投入|销售额|利润|成本|收入|亿元?|万元?|million|billion|revenue|profit|cost)",
81
+ "has_timeline": r"(20\d{2}|timeline|roadmap|阶段|起步|腾飞|验证|年|月)",
82
+ "has_percentages": r"\d+(?:\.\d+)?\s*%",
83
+ "has_targets": r"(目标|达到|覆盖|完成|形成|实现|sales|target|milestone|deliverable)",
84
+ "has_benchmark_terms": r"(领先|首个|第一|benchmark|sota|state of the art|对比|优于|提升)",
85
+ "has_validation_terms": r"(验证|测试|试验|仿真|实验|prototype|validation|test|pilot)",
86
+ "has_financial_table_terms": r"(销售额|利润|成本|投入|融资计划|财务预测|现金流|revenue|profit|cost)",
87
+ "has_scale_terms": r"(规模化|场景|覆盖|批量|市场占有率|用户数|deployment|adoption|scale)",
88
+ }
89
+
90
+ flags = {name: bool(re.search(pattern, text, flags=re.IGNORECASE)) for name, pattern in signal_patterns.items()}
91
+
92
+ score = 0
93
+ score += 2 if flags["has_numbers"] else 0
94
+ score += 1 if flags["has_money"] else 0
95
+ score += 1 if flags["has_timeline"] else 0
96
+ score += 1 if flags["has_percentages"] else 0
97
+ score += 1 if flags["has_targets"] else 0
98
+ score += 1 if flags["has_benchmark_terms"] else 0
99
+ score += 1 if flags["has_validation_terms"] else 0
100
+ score += 1 if flags["has_financial_table_terms"] else 0
101
+ score += 1 if flags["has_scale_terms"] else 0
102
+
103
+ if score >= 8:
104
+ evidence_level = "strong"
105
+ elif score >= 5:
106
+ evidence_level = "moderate"
107
+ else:
108
+ evidence_level = "weak"
109
+
110
+ missing = []
111
+ if not flags["has_numbers"]:
112
+ missing.append("no explicit numeric evidence")
113
+ if not flags["has_targets"]:
114
+ missing.append("no clear measurable targets")
115
+ if not flags["has_timeline"]:
116
+ missing.append("no clear timeline or milestone evidence")
117
+ if not flags["has_validation_terms"]:
118
+ missing.append("no explicit validation or testing evidence")
119
+
120
+ return {
121
+ "flags": flags,
122
+ "numeric_spans": numeric_spans[:30],
123
+ "numeric_count": len(numeric_spans),
124
+ "evidence_level": evidence_level,
125
+ "missing_metric_signals": missing,
126
+ }
127
+
128
+
129
+ def detect_possible_numeric_conflicts(text: str) -> List[Dict[str, Any]]:
130
+ """
131
+ Lightweight conflict detector.
132
+ It does not prove contradictions mathematically,
133
+ but surfaces repeated metric categories with divergent values.
134
+ """
135
+ text = _clean_text(text)
136
+
137
+ categories = {
138
+ "financing": r"(\d+(?:\.\d+)?(?:万|亿|万元|亿元))",
139
+ "scenarios": r"(\d+(?:\.\d+)?种(?:以上)?应用场景)",
140
+ "sales": r"(预计年销售额\s*\d+(?:\.\d+)?)",
141
+ "profit": r"(预计年利润\s*-?\d+(?:\.\d+)?)",
142
+ "years": r"(20\d{2}\s*[-–—]\s*\d{4}|20\d{2}年(?:\d{1,2}月)?)",
143
+ }
144
+
145
+ conflicts: List[Dict[str, Any]] = []
146
+
147
+ for category, pattern in categories.items():
148
+ matches = re.findall(pattern, text)
149
+ uniq = _unique_keep_order(matches, max_items=10)
150
+ if len(uniq) >= 3:
151
+ conflicts.append({
152
+ "category": category,
153
+ "values": uniq,
154
+ "note": "multiple values detected; review for consistency",
155
+ })
156
+
157
+ return conflicts
158
+
159
+
160
+ def build_metric_report(full_text: str, pages: Sequence[Dict[str, Any]]) -> Dict[str, Any]:
161
+ """
162
+ Stage 1.5 output artifact.
163
+ """
164
+ text = _clean_text(full_text)
165
+ metric_signals = detect_metric_signals(text)
166
+ conflicts = detect_possible_numeric_conflicts(text)
167
+
168
+ page_hits = []
169
+ for page in pages:
170
+ page_text = _clean_text(page.get("text", ""))
171
+ page_metrics = extract_numeric_spans(page_text)
172
+ if page_metrics:
173
+ page_hits.append({
174
+ "page_index": int(page.get("page_index", -1)),
175
+ "metric_count": len(page_metrics),
176
+ "sample_metrics": page_metrics[:8],
177
+ })
178
+
179
+ page_hits = sorted(page_hits, key=lambda x: x["metric_count"], reverse=True)[:8]
180
+
181
+ return {
182
+ "stage": "1.5_metric_checker",
183
+ "metric_signals": metric_signals,
184
+ "possible_conflicts": conflicts,
185
+ "top_metric_pages": page_hits,
186
+ }
187
+
188
+
189
+ def build_metric_prompt_suffix(metric_report: Dict[str, Any]) -> str:
190
+ """
191
+ Convert metric detection result into prompt guidance for downstream review.
192
+ """
193
+ signals = metric_report.get("metric_signals", {})
194
+ flags = signals.get("flags", {})
195
+ evidence_level = signals.get("evidence_level", "weak")
196
+ missing = signals.get("missing_metric_signals", [])
197
+ numeric_spans = signals.get("numeric_spans", [])[:10]
198
+ conflicts = metric_report.get("possible_conflicts", [])
199
+
200
+ lines = []
201
+ lines.append("Metric-check guidance:")
202
+ lines.append(f"- Quantitative evidence level: {evidence_level}.")
203
+ lines.append(f"- Numeric evidence detected: {'yes' if flags.get('has_numbers') else 'no'}.")
204
+ lines.append(f"- Timeline evidence detected: {'yes' if flags.get('has_timeline') else 'no'}.")
205
+ lines.append(f"- Validation evidence detected: {'yes' if flags.get('has_validation_terms') else 'no'}.")
206
+ lines.append(f"- Financial evidence detected: {'yes' if flags.get('has_money') else 'no'}.")
207
+
208
+ if numeric_spans:
209
+ lines.append(f"- Example metric spans: {'; '.join(numeric_spans)}.")
210
+
211
+ if missing:
212
+ lines.append(f"- Missing metric signals: {'; '.join(missing)}.")
213
+ lines.append("- If a dimension relies on quantitative justification but explicit metrics are missing, mention this clearly as a limitation.")
214
+
215
+ if conflicts:
216
+ lines.append("- Possible numeric consistency issues were detected; review repeated values carefully before making strong claims.")
217
+
218
+ return " ".join(lines)
src/tools/post_processing.py ADDED
@@ -0,0 +1,1595 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ post_processing.py · Structured Candidate Post-Processing (v3.2 · no-web-hard-gates)
4
+
5
+ 适配新版 llm_answering 输出:
6
+ - 期望输入 schema: "llm_answering.v2" 或 "refined_items.v2.proposal_aware_with_general_insights"
7
+ - 顶层结构:{"meta": {...}, "items": [ {dimension, q_index, question, candidates: [...]}, ... ]}
8
+
9
+ 输出:
10
+ - src/data/refined_answers/<pid>/postproc/metrics.json
11
+ - src/data/refined_answers/<pid>/postproc/selected_by_question.json
12
+ - src/data/refined_answers/<pid>/postproc/final_payload.json
13
+ - src/data/refined_answers/<pid>/postproc/report.md
14
+ - src/data/refined_answers/<pid>/postproc/drops_debug.json
15
+ """
16
+
17
+ import os
18
+ import re
19
+ import json
20
+ import math
21
+ import string
22
+ import argparse
23
+ from pathlib import Path
24
+ from datetime import datetime
25
+ from collections import defaultdict, Counter
26
+ from functools import lru_cache
27
+ import re as _re
28
+
29
+ # ============================ 路径与默认 ============================
30
+
31
+ ROOT = Path(__file__).resolve().parents[1] # .../src
32
+ DATA_DIR = ROOT / "data"
33
+ REFINED_ROOT = DATA_DIR / "refined_answers"
34
+ CONF_DIR = DATA_DIR / "config" / "postproc"
35
+ QS_CONF_PATH = DATA_DIR / "config" / "question_sets" / "generated_questions.json"
36
+
37
+ # —— 与 answering 对齐的权威词表(多辖区/标准) ——
38
+ # ⚠️ 已移除 "hc",避免把普通文本误计为 Health Canada;保留 "health canada"
39
+ AUTHORITY_TOKENS = [
40
+ "fda","ema","ich q8","ich q9","ich q10","21 cfr part 11",
41
+ "iso 13485","iso 14971","iso 27001","who","gamp5","gamp 5", "pic/s",
42
+ "clinicaltrials.gov","eudract","nct","doi","orcid","pubmed","scopus",
43
+ "uspto","epo","cnipa",
44
+ # 多辖区/缩写
45
+ "nmpa","cfda","mhra","pmda","tga","health canada","nice",
46
+ "eudralex","mdr","ivdr",
47
+ "iec 62304","iec 62366","iso 62304","iso 62366",
48
+ "gcp","gmp","glp","gxp",
49
+ # 隐私/安全/合规
50
+ "gdpr","hipaa","phipa","pipeda","nist","soc 2","iso 27017","iso 27018"
51
+ ]
52
+
53
+ COVERAGE_BANK = {
54
+ "regulatory": ["fda","ema","ich","21 cfr","iso","pic/s","gmp","gcp","glp","gxp",
55
+ "nmpa","mhra","pmda","tga","health canada","mdr","ivdr",
56
+ "gdpr","pipeda","phipa","hipaa","nist","soc 2","iso 27017","iso 27018"],
57
+ "trial": ["clinicaltrials.gov","eudract","nct"],
58
+ "publication": ["pubmed","scopus","doi","orcid"],
59
+ "patent": ["uspto","epo","cnipa","wo"],
60
+ "repo": ["github","gitlab","model card","data card"]
61
+ }
62
+
63
+ # === BEGIN PATCH · alias + tokenizer ===
64
+ # 压缩串 → 标准空格写法(用于 21CFRPart11 / ISO13485 / ICHQ10 等)
65
+ ALIAS_MAP = {
66
+ # 法规/标准归一
67
+ "21cfrpart11": "21 cfr part 11",
68
+ "21cfr11": "21 cfr part 11",
69
+ "iso13485": "iso 13485",
70
+ "iso14971": "iso 14971",
71
+ "iso27001": "iso 27001",
72
+ "iso27017": "iso 27017",
73
+ "iso27018": "iso 27018",
74
+ "iec62304": "iec 62304",
75
+ "iec62366": "iec 62366",
76
+ "gamp5": "gamp 5",
77
+ "ichq8": "ich q8",
78
+ "ichq9": "ich q9",
79
+ "ichq10": "ich q10",
80
+ "pics": "pic/s",
81
+
82
+ # 站点/域名归一
83
+ "clinicaltrialsgov": "clinicaltrials gov",
84
+ }
85
+
86
+ # 对齐加权:权威 token 给更高权重
87
+ AUTHORITY_KEYWORDS = {
88
+ "fda": 2.0, "ema": 2.0, "ich": 2.0, "q8": 1.4, "q9": 1.4, "q10": 1.6,
89
+ "iso": 2.0, "13485": 2.0, "14971": 2.0, "27001": 1.6, "27017": 1.4, "27018": 1.4,
90
+ "21": 1.1, "cfr": 2.0, "part": 1.1, "11": 1.1,
91
+ "clinicaltrials": 2.0, "gov": 1.0, "eudract": 2.0,
92
+ "pubmed": 2.0, "scopus": 2.0, "orcid": 1.5,
93
+ "uspto": 2.0, "epo": 2.0, "cnipa": 2.0,
94
+ "gdpr": 2.0, "hipaa": 2.0, "phipa": 2.0, "pipeda": 2.0,
95
+ "gxp": 1.4, "gmp": 1.4, "glp": 1.4, "gcp": 1.4, "doi": 1.4,
96
+ }
97
+
98
+ # ==== 中文高权词(权重可按需微调)====
99
+ AUTHORITY_KEYWORDS_ZH = {
100
+ "国家药监局": 2.0, "药监局": 1.8, "药监": 1.6, "nmpa": 2.0,
101
+ "注册": 1.6, "临床": 1.6, "注册号": 1.6, "备案": 1.2,
102
+ "发表": 1.4, "论文": 1.6, "期刊": 1.4, "doi": 1.6, "pubmed": 2.0,
103
+ "专利": 1.8, "发明专利": 2.0, "uspto": 2.0, "epo": 2.0, "cnipa": 2.0,
104
+ "合规": 1.6, "隐私": 1.4, "安全": 1.4, "gxp": 1.4, "gmp": 1.4, "gcp": 1.4, "glp": 1.4,
105
+ "上市": 1.6, "量产": 1.4, "认证": 1.4, "iso": 2.0, "iec": 1.6,
106
+ "药品审评中心": 1.8, "cde": 1.6, "卫健委": 1.4, "nhc": 1.4,
107
+ "注册证": 1.6, "批准文号": 1.6, "临床试验登记": 1.6
108
+ }
109
+
110
+ STOPWORDS_ALIGN = {
111
+ "the","and","of","to","for","in","on","by","with","a","an","is","are",
112
+ "及","与","的","和","在","对","为","以及"
113
+ }
114
+
115
+ _RE_NON_ALNUM = _re.compile(r"[^a-z0-9]+", _re.IGNORECASE)
116
+
117
+ def _apply_aliases(s: str) -> str:
118
+ """统一空格/点号,做数字↔字母断词,并用 ALIAS_MAP 展开压缩串。"""
119
+ if not s:
120
+ return ""
121
+ base = str(s).lower()
122
+ base = base.replace("\u00a0", " ").replace("\u3000", " ")
123
+
124
+ # 先生成去标点“紧凑串”做 alias 命中
125
+ compact = _RE_NON_ALNUM.sub("", base)
126
+ if compact in ALIAS_MAP:
127
+ base = ALIAS_MAP[compact]
128
+
129
+ # 数字-字母边界���空格:21CFRPart11 -> 21 CFR Part 11
130
+ base = _re.sub(r"([a-z])([0-9])", r"\1 \2", base)
131
+ base = _re.sub(r"([0-9])([a-z])", r"\1 \2", base)
132
+
133
+ # ICHQ10 → ich q10(兜底)
134
+ base = _re.sub(r"\b(ich)\s*q\s*(\d+)\b", r"\1 q\2", base)
135
+ # clinicaltrials.gov → clinicaltrials gov
136
+ base = base.replace("clinicaltrials.gov", "clinicaltrials gov")
137
+ return base
138
+
139
+ def _tokens_for_alignment(s: str):
140
+ r"""切出 [a-z]+ / \d+ / 中文块 词元,并构造 bigrams。"""
141
+ if not s:
142
+ return set(), set()
143
+ s = _apply_aliases(s)
144
+
145
+ # 保留原有:把非字母数字替换为空格(英文/数字通道)
146
+ s_en = _RE_NON_ALNUM.sub(" ", s)
147
+
148
+ # 英文/数字 token
149
+ toks_en = [t for t in _re.findall(r"[a-z]+|\d+", s_en) if t not in STOPWORDS_ALIGN]
150
+
151
+ # —— 中文通道:直接抽取连续的中日韩统一表意文字(至少两个字避免噪音)——
152
+ toks_zh = _re.findall(r"[\u4e00-\u9fff]{2,}", s)
153
+
154
+ # 合并
155
+ toks = toks_en + toks_zh
156
+
157
+ unigrams = set(toks)
158
+ bigrams = set()
159
+ for i in range(len(toks)-1):
160
+ bigrams.add(toks[i] + " " + toks[i+1])
161
+ return unigrams, bigrams
162
+
163
+
164
+ def _weighted_overlap(q_uni, q_bi, c_uni, c_bi):
165
+ """基于问题/提示 tokens 与候选 tokens 的带权重重合度,bigrams 轻度加成。"""
166
+ if not q_uni:
167
+ return 0.0
168
+
169
+ def w(t):
170
+ # 既支持英文权重,也支持中文高权词
171
+ if t in AUTHORITY_KEYWORDS:
172
+ return AUTHORITY_KEYWORDS[t]
173
+ if t in AUTHORITY_KEYWORDS_ZH:
174
+ return AUTHORITY_KEYWORDS_ZH[t]
175
+ return 1.0
176
+
177
+ den = sum(w(t) for t in q_uni)
178
+ hit = sum(w(t) for t in (q_uni & c_uni))
179
+ bonus = 0.0
180
+ if q_bi and c_bi:
181
+ bi_hit = len(q_bi & c_bi)
182
+ bonus = min(0.30, 0.10 * bi_hit)
183
+ return min(1.0, (hit / max(1e-9, den)) * (1.0 + bonus))
184
+ # === END PATCH ===
185
+
186
+
187
+ DEFAULT_CONF = {
188
+ # ========= 题内标准化参数 =========
189
+ "length_ref_chars": 280,
190
+ "claims_ref": 3,
191
+ "evidence_ref": 4,
192
+ "jaccard_threshold": 0.30,
193
+
194
+ # ========= 权重配置 =========
195
+ # 说明:在“无 web 检索”模式下,evidence/authority/coverage 不再参与主评分
196
+ # 👉 降低一致性权重,增加“对齐 + claims”的权重,让强维度更容易拉开
197
+ "consistency_weight": 0.10,
198
+ "fields_weight": {
199
+ "length": 0.18, # 原 0.20
200
+ "claims": 0.27, # 原 0.25(多写要点的题更拉分)
201
+ "evidence_count": 0.0,
202
+ "evidence_authority": 0.0,
203
+ "evidence_coverage": 0.0,
204
+ "structure": 0.20,
205
+ "alignment": 0.30, # 原 0.25(更看重和问题/维度的贴合度)
206
+ "calibrated_confidence": 0.05 # 原 0.10(置信度稍微降一点)
207
+ },
208
+
209
+ # ========= 维度权重 =========
210
+ "dimension_weight": {
211
+ "team": 1.00,
212
+ "objectives": 1.00,
213
+ "strategy": 1.00,
214
+ "innovation": 1.10,
215
+ "feasibility": 1.20
216
+ },
217
+
218
+ # ========= 扣分项 =========
219
+ # 👉 所有 penalty 稍微变“软”一点,避免把分数全往 0.4 附近压扁
220
+ "penalties": {
221
+ "contradiction": 0.05, # 原 0.08
222
+ "overclaim": 0.03, # 原 0.05
223
+ "understructure": 0.03, # 原 0.05
224
+ "redline_residual": 0.06, # 原 0.08
225
+ "dimension_drift": 0.04 # 原 0.06
226
+ },
227
+
228
+ # ========= 过滤阈值(支持软窗口) =========
229
+ # 注:权威/覆盖相关阈值只保留作兼容字段,不再触发硬过滤
230
+ "filters": {
231
+ # 证据相关:不再做硬门槛,仅统计
232
+ "min_evidence_count": 0,
233
+
234
+ # 基础结构要求
235
+ "min_bullet_lines": 3,
236
+ "min_median_bullet_len": 6,
237
+ "min_structured_score": 0.05,
238
+ "soft_window": True,
239
+
240
+ # 对齐阈值:仅用于过滤严重跑题的垃圾回答
241
+ "min_alignment_for_keep": 0.10,
242
+
243
+ # 兼容字段(不再参与过滤)
244
+ "min_auth_hits_for_keep": 0,
245
+ "min_coverage_bins_for_keep": 0,
246
+ "min_authority_ratio": 0.0,
247
+ "min_coverage_ratio": 0.0,
248
+ "dyn_align_relax_trigger": 0.0,
249
+ "dyn_align_relax_delta": 0.0
250
+ },
251
+
252
+ # ========= 一致性修正参数 =========
253
+ "consistency_correction": {
254
+ "k_contradiction": 0.15,
255
+ "beta": {
256
+ "jaccard_sweetspot": [0.25, 0.65],
257
+ "provider_calibration": {
258
+ "deepseek": {"beta_delta": -0.01},
259
+ "openai": {"beta_delta": 0.00},
260
+ "default": {"beta_delta": 0.00}
261
+ }
262
+ }
263
+ },
264
+
265
+ # ========= 维度差异化放宽(仅过滤时用) =========
266
+ "dimension_specific": {
267
+ "objectives": {"min_bullet_lines": 3, "min_alignment_for_keep": 0.18},
268
+ "strategy": {"min_bullet_lines": 3, "min_alignment_for_keep": 0.24},
269
+ "innovation": {"min_bullet_lines": 3, "min_alignment_for_keep": 0.18},
270
+ "feasibility":{"min_bullet_lines": 3, "min_alignment_for_keep": 0.12}
271
+ },
272
+
273
+ # ========= 报告输出 =========
274
+ "bar_symbols": 20,
275
+ "adv_topk": 5,
276
+ "report_width": 92,
277
+ "unknown_warn_ratio": 0.10
278
+ }
279
+
280
+ PUNC = set(string.punctuation + ",。;:!?、()【】《》…—-·")
281
+ DIM_ORDER = ["team", "objectives", "strategy", "innovation", "feasibility"]
282
+
283
+ RE_DATE = re.compile(r"\b(20\d{2}|19\d{2})([-/.]|年)\d{1,2}([-/\.日]|月)\d{1,2}\b|\b(Q[1-4]\s*-\s*20\d{2})\b", re.I)
284
+ RE_MONEY = re.compile(r"\b(\$|USD|EUR|CNY|RMB|CAD)\s*\d{2,}(,\d{3})*(\.\d+)?\b|\b\d+(\.\d+)?\s*(million|billion|万|亿)\b", re.I)
285
+ RE_TRIAL = re.compile(r"\bNCT\d{8}\b|\bEUCTR-\d{4}-\d{6}-\d{2}\b", re.I)
286
+ RE_DOI = re.compile(r"\b10\.\d{4,9}/[-._;()/:A-Z0-9]+\b", re.I)
287
+ RE_PATENT = re.compile(r"\b(US|EP|CN)\d{5,}\b|\bWO\d{7,}\b", re.I)
288
+ RE_ISO = re.compile(r"\b(ISO|IEC)\s?\d{4,5}(-\d+)?\b", re.I) # 兼容 IEC
289
+ RE_STDNUM = re.compile(r"\bEN\s?\d{3,5}\b|\bASTM\s?[A-Z]?\d{2,5}\b", re.I)
290
+ RE_ID_ANY = re.compile(r"(注册号|登记号|批准文号|备案号)", re.I)
291
+
292
+ # ====== 行清洗与软拼接 ======
293
+
294
+ DOMAIN_FIXES = [
295
+ (re.compile(r"clinicaltrials\s*[\.\-]?\s*(\d+\s*[\.\-]?\s*)?gov", re.I), "ClinicalTrials.gov"),
296
+ (re.compile(r"eudract\s*[\.\-]?\s*eu", re.I), "EudraCT EU"),
297
+ (re.compile(r"pubmed\s*[\.\-]?\s*ncbi\s*[\.\-]?\s*nlm\s*[\.\-]?\s*nih\s*[\.\-]?\s*gov", re.I), "PubMed"),
298
+ ]
299
+ BULLET_NORM = re.compile(r"^\s*(?:[-*•·■▪︎▶️●]|(\d+)[\.\)]|[((]?[一二三四五六七八九十][)).、])\s*")
300
+
301
+ def _soft_join(text: str) -> str:
302
+ if not text:
303
+ return ""
304
+ s = text
305
+ s = re.sub(r"(\w)-\n(\w)", r"\1\2", s)
306
+ s = re.sub(r"([A-Za-z0-9])\n([A-Za-z0-9])", r"\1 \2", s)
307
+ s = re.sub(r"\s*\.\s*(gov|com|org|net|io)\b", r".\1", s, flags=re.I)
308
+ for pat, rep in DOMAIN_FIXES:
309
+ s = pat.sub(rep, s)
310
+ return s
311
+
312
+ def _normalize_bullets(text: str) -> str:
313
+ lines = [ln.rstrip() for ln in (text or "").splitlines()]
314
+ out = []
315
+ for ln in lines:
316
+ base = BULLET_NORM.sub("", ln).strip()
317
+ if not base:
318
+ continue
319
+ out.append(base)
320
+ return "\n".join(out)
321
+
322
+ # ========= 两套清洗 =========
323
+
324
+ def sanitize_for_scoring(raw: str) -> str:
325
+ s = raw or ""
326
+ s = _soft_join(s)
327
+ s = re.sub(r"[ \t]+", " ", s)
328
+ return s.strip()
329
+
330
+ def sanitize_for_display(raw: str) -> str:
331
+ s = raw or ""
332
+ s = _soft_join(s)
333
+ s = re.sub(r"[ \t]+", " ", s)
334
+ s = re.sub(r"\b([A-Za-z])(?:\s+[A-Za-z]){1,3}\b",
335
+ lambda m: m.group(0).replace(" ", ""), s)
336
+ s = _normalize_bullets(s)
337
+ return s.strip()
338
+
339
+ def sanitize_answer(raw: str) -> str:
340
+ return sanitize_for_display(raw)
341
+
342
+ # ====== 占位行检测 ======
343
+ PLACEHOLDER_LINE = re.compile(
344
+ r"""^(
345
+ [\-\–—=~\._]{2,}$ |
346
+ [\(\)\[\]\{\}]$ |
347
+ \d+\s*[\.\)]\s*\d?$ |
348
+ (?:原则|区间|示例|条款)(?:\s*/\s*(?:原则|区间))?$ |
349
+ [A-Za-z]$ |
350
+ [\u3000\s]*$
351
+ )""",
352
+ re.X
353
+ )
354
+
355
+ def _placeholder_ratio(cleaned_text: str) -> float:
356
+ if not cleaned_text:
357
+ return 1.0
358
+ lines = [ln.strip() for ln in cleaned_text.splitlines() if ln.strip()]
359
+ if not lines:
360
+ return 1.0
361
+ bad = sum(1 for ln in lines if PLACEHOLDER_LINE.match(ln))
362
+ return bad / max(1, len(lines))
363
+
364
+ # ============================ 工具函数(带缓存) ============================
365
+
366
+ def load_config():
367
+ cfg_path = CONF_DIR / "config.json"
368
+ if cfg_path.exists():
369
+ try:
370
+ user_cfg = json.loads(cfg_path.read_text(encoding="utf-8"))
371
+ return _merge_conf(DEFAULT_CONF, user_cfg)
372
+ except Exception:
373
+ pass
374
+ return DEFAULT_CONF
375
+
376
+ def _merge_conf(base, user):
377
+ out = dict(base)
378
+ for k, v in (user or {}).items():
379
+ if isinstance(v, dict) and isinstance(base.get(k), dict):
380
+ nv = dict(base[k]); nv.update(v); out[k] = nv
381
+ else:
382
+ out[k] = v
383
+ return out
384
+
385
+ def now_str():
386
+ return datetime.now().strftime("%Y-%m-%d %H:%M:%S")
387
+
388
+ def detect_latest_pid() -> str:
389
+ if not REFINED_ROOT.exists():
390
+ return ""
391
+ cands = []
392
+ for d in REFINED_ROOT.iterdir():
393
+ if d.is_dir() and (d / "all_refined_items.json").exists():
394
+ cands.append((d.name, (d / "all_refined_items.json").stat().st_mtime))
395
+ cands.sort(key=lambda x: x[1], reverse=True)
396
+ return cands[0][0] if cands else ""
397
+
398
+ def read_json(p: Path):
399
+ return json.loads(p.read_text(encoding="utf-8"))
400
+
401
+ def write_json(p: Path, obj):
402
+ p.parent.mkdir(parents=True, exist_ok=True)
403
+ p.write_text(json.dumps(obj, ensure_ascii=False, indent=2), encoding="utf-8")
404
+
405
+ def norm01(x, lo, hi):
406
+ """
407
+ 把 x 在线性映射到 [0,1] 之间:
408
+ - hi <= lo 时直接返回 0
409
+ - 小于区间下界截到 0,大于上界截到 1
410
+ """
411
+ if hi <= lo:
412
+ return 0.0
413
+ v = (x - lo) / (hi - lo)
414
+ if v < 0:
415
+ return 0.0
416
+ if v > 1:
417
+ return 1.0
418
+ return float(v)
419
+
420
+ def safe_float(x, default=0.6):
421
+ try:
422
+ f = float(x)
423
+ if math.isnan(f):
424
+ return default
425
+ return max(0.0, min(1.0, f))
426
+ except Exception:
427
+ return default
428
+
429
+ @lru_cache(maxsize=8192)
430
+ def _sanitize_cached(raw: str) -> str:
431
+ return sanitize_for_display(raw)
432
+
433
+ @lru_cache(maxsize=8192)
434
+ def _word_tokens_cached(text: str):
435
+ """
436
+ 词级 token:英文/数字用 [a-z0-9]+,中文取连续中日韩统一表意文字(长度≥2)。
437
+ 与语义对齐逻辑保持一致,不做词干化与停用词处理。
438
+ """
439
+ s = (sanitize_for_scoring(text) or "").lower()
440
+ # 英文/数字 token
441
+ en = re.findall(r"[a-z0-9]+", s)
442
+ # 中文 token:连续汉字(≥2,降噪;如需更激进可改为 {1,})
443
+ zh = re.findall(r"[\u4e00-\u9fff]{2,}", s)
444
+ return tuple(t for t in en + zh if t)
445
+
446
+ _tokenize_cached = _word_tokens_cached
447
+
448
+ def tokenize(text: str):
449
+ return list(_word_tokens_cached(text))
450
+
451
+
452
+ def jaccard(a_tokens, b_tokens):
453
+ if not a_tokens or not b_tokens:
454
+ return 0.0
455
+ A, B = set(a_tokens), set(b_tokens)
456
+ inter = len(A & B)
457
+ uni = len(A | B)
458
+ return inter / uni if uni else 0.0
459
+
460
+ def looks_structured(ans: str) -> float:
461
+ cleaned = sanitize_for_scoring(ans)
462
+ lines = [ln.strip() for ln in (cleaned or "").splitlines() if ln.strip()]
463
+ if not lines:
464
+ return 0.0
465
+ bulletish = re.compile(r"^(\d+[\.\)]\s+|[-*•·■▪︎▶️●]\s+|[((]?[一二三四五六七八九十][)).、]\s+|[A-Za-z]\))")
466
+ bullets = sum(1 for ln in lines if bulletish.match(ln))
467
+ ratio_b = bullets / max(1, len(lines))
468
+ shortish = sum(1 for ln in lines if 6 <= len(ln) <= 64)
469
+ ratio_s = shortish / max(1, len(lines))
470
+ return max(0.0, min(1.0, 0.4 * ratio_b + 0.6 * ratio_s))
471
+
472
+ def overclaim_score(ans: str) -> float:
473
+ terms = ["必须", "保证", "绝对", "完全", "零风险", "一定", "毫无", "不可能出错", "100%"]
474
+ cnt = sum((ans or "").count(t) for t in terms)
475
+ length = max(1, len(ans or ""))
476
+ dens = cnt / length
477
+ return max(0.0, min(1.0, dens * 200))
478
+
479
+ def contradiction_pair(a: str, b: str) -> float:
480
+ neg = ["不", "无", "否", "不可", "禁止", "避免", "不得"]
481
+ pos = ["可", "能", "允许", "建议", "推荐", "可行", "通过"]
482
+ a_neg = sum((sanitize_for_scoring(a) or "").count(t) for t in neg)
483
+ b_neg = sum((sanitize_for_scoring(b) or "").count(t) for t in neg)
484
+ a_pos = sum((sanitize_for_scoring(a) or "").count(t) for t in pos)
485
+ b_pos = sum((sanitize_for_scoring(b) or "").count(t) for t in pos)
486
+ jac = jaccard(_tokenize_cached(sanitize_for_scoring(a)), _tokenize_cached(sanitize_for_scoring(b)))
487
+ diff = abs((a_pos - a_neg) - (b_pos - b_neg))
488
+ base = norm01(diff, 0, 20)
489
+ penal = 1.0 - jac
490
+ return max(0.0, min(1.0, base * penal))
491
+
492
+ def bar(value: float, n: int = 20) -> str:
493
+ n = max(1, n)
494
+ k = max(0, min(n, int(round(float(value) * n))))
495
+ return "█" * k + "░" * (n - k)
496
+
497
+ def authority_ratio(hints: list) -> float:
498
+ if not hints:
499
+ return 0.0
500
+ hits = 0
501
+ for h in hints:
502
+ s = (h or "").lower()
503
+ if any(tok in s for tok in AUTHORITY_TOKENS):
504
+ hits += 1
505
+ return hits / max(1, len(hints))
506
+
507
+ def coverage_score(hints: list) -> float:
508
+ if not hints:
509
+ return 0.0
510
+ cats = set()
511
+ for h in hints:
512
+ s = (h or "").lower()
513
+ for cat, toks in COVERAGE_BANK.items():
514
+ if any(tok in s for tok in toks):
515
+ cats.add(cat)
516
+ return len(cats) / len(COVERAGE_BANK)
517
+
518
+ def has_redline(text: str) -> bool:
519
+ t = text or ""
520
+ return any(p.search(t) for p in (RE_DATE, RE_MONEY, RE_TRIAL, RE_DOI, RE_PATENT, RE_ISO, RE_STDNUM, RE_ID_ANY))
521
+
522
+ # ============================ 去串味 + 权威导向 ============================
523
+
524
+ OTHER_DIMS = set(DIM_ORDER)
525
+
526
+ def _strip_cross_dim_tags(dim: str, tags: list) -> list:
527
+ keep, seen = [], set()
528
+ for t in tags or []:
529
+ s = str(t or "").strip()
530
+ if not s:
531
+ continue
532
+ low = s.lower()
533
+ if low in OTHER_DIMS and low != dim.lower():
534
+ continue
535
+ if low in seen:
536
+ continue
537
+ seen.add(low)
538
+ keep.append(s)
539
+ return keep
540
+
541
+ def _authority_hints_from_qs(qs_cfg: dict, dim: str, limit: int = 8) -> list:
542
+ try:
543
+ block = qs_cfg.get(dim, {}) or {}
544
+ hints = block.get("search_hints", []) or []
545
+ hints = _strip_cross_dim_tags(dim, hints)
546
+ def score(h):
547
+ s = str(h).lower()
548
+ return -sum(1 for t in AUTHORITY_TOKENS if t in s)
549
+ hints = list({h: None for h in hints}.keys())
550
+ hints.sort(key=score)
551
+ return hints[:limit]
552
+ except Exception:
553
+ return []
554
+
555
+ # ===== 语义化对齐(替换原先的字面包含比对) =====
556
+ # === BEGIN PATCH · improved alignment ===
557
+ def _alignment_ratio(dim: str, auth_hints: list, answer: str, topic_tags: list, evidence_hints: list,
558
+ question: str = "", claims: list = None) -> float:
559
+ """
560
+ 语义对齐评分(支持中英):把候选语料(answer + topic_tags + evidence_hints*2 [+ question] [+ claims])
561
+ 与每条 auth_hint 做 token 对齐。
562
+ - 若某条 hint 切词后为空(常见于中文),跳过该条;
563
+ - 若最终“有效切词的 hints 条数”为 0,则视为“无提示词场景”,给中性保底分。
564
+ """
565
+ claims = claims or []
566
+ # 候选语料:把 evidence_hints 重复一次,提高权重;question/claims 可控注入
567
+ pool = [
568
+ sanitize_for_scoring(answer or ""),
569
+ " ".join([str(t) for t in (topic_tags or [])]),
570
+ " ".join([str(h) for h in (evidence_hints or [])]),
571
+ " ".join([str(h) for h in (evidence_hints or [])])
572
+ ]
573
+ if question:
574
+ pool.append(str(question))
575
+ if claims:
576
+ pool.append(" ".join([str(c) for c in claims if str(c).strip()]))
577
+
578
+ corpus = " \n ".join(pool)
579
+ c_uni, c_bi = _tokens_for_alignment(corpus)
580
+
581
+ # 无提示词:给一个中性保底
582
+ if not auth_hints:
583
+ return 0.35 if c_uni else 0.0
584
+
585
+ cover_hits = 0
586
+ scores = []
587
+ valid_hints = 0 # 关键:仅统计切得出 token 的 hint
588
+
589
+ for h in auth_hints:
590
+ q_uni, q_bi = _tokens_for_alignment(str(h))
591
+ if not q_uni and not q_bi:
592
+ continue
593
+ valid_hints += 1
594
+ s = _weighted_overlap(q_uni, q_bi, c_uni, c_bi)
595
+ scores.append(s)
596
+ if s >= 0.15:
597
+ cover_hits += 1
598
+
599
+ # 若所有 hints 切完都无有效 token,则当作“无 hint 场景”,给保底分
600
+ if valid_hints == 0:
601
+ return 0.5 if c_uni else 0.0
602
+
603
+ if not scores:
604
+ return 0.0
605
+
606
+ coverage = cover_hits / max(1, valid_hints)
607
+ mean_hit = sum(scores) / len(scores)
608
+ return max(0.0, min(1.0, 0.5 * coverage + 0.5 * mean_hit))
609
+ # === END PATCH ===
610
+
611
+
612
+ def _dimension_drift_score(dim: str, answer: str, topic_tags: list, evidence_hints: list) -> float:
613
+ tags = " ".join([str(t) for t in (topic_tags or [])]).lower()
614
+ others = sorted(OTHER_DIMS - {dim})
615
+ hit = sum(1 for od in others if od in tags)
616
+ weak_corpus = (" ".join([
617
+ sanitize_for_scoring(answer or ""),
618
+ " ".join([str(h) for h in (evidence_hints or [])])
619
+ ])).lower()
620
+ weak = sum(1 for od in others if od in weak_corpus)
621
+ score = 0.6 * (hit / max(1, len(others))) + 0.4 * (weak / max(1, len(others)))
622
+ return max(0.0, min(1.0, score))
623
+
624
+ # =============== 证据短语 / 通识要点抽取(报告用) ===============
625
+
626
+ def _top_evidence_phrases(hints: list, topk: int = 3):
627
+ if not hints:
628
+ return []
629
+ def _norm(h: str):
630
+ s = re.sub(r"\s+", " ", (h or "")).strip()
631
+ return s[:120]
632
+ candidates = []
633
+ for h in hints:
634
+ s = _norm(str(h))
635
+ if not s:
636
+ continue
637
+ score = 0
638
+ low = s.lower()
639
+ for tok in AUTHORITY_TOKENS:
640
+ if tok in low:
641
+ score += 1
642
+ candidates.append((s, score))
643
+ counter = Counter([c[0] for c in candidates])
644
+ ranked = sorted(
645
+ counter.items(),
646
+ key=lambda x: (-max([sc for (txt, sc) in candidates if txt == x[0]]), -x[1], x[0])
647
+ )
648
+ return [r[0] for r in ranked[:topk]]
649
+
650
+ def _uniq_general_insights(gi_list, topk: int = 10, max_len: int = 220):
651
+ """
652
+ 将各问答里的 general_insights 聚合成维度级“行业经验层”:
653
+ - 去重
654
+ - 统一空白
655
+ - 控制长度
656
+ """
657
+ if not gi_list:
658
+ return []
659
+ seen = set()
660
+ out = []
661
+ for g in gi_list:
662
+ s = re.sub(r"\s+", " ", str(g or "").strip())
663
+ if not s:
664
+ continue
665
+ if s in seen:
666
+ continue
667
+ seen.add(s)
668
+ if len(s) > max_len:
669
+ s = s[:max_len].rstrip() + "…"
670
+ out.append(s)
671
+ if len(out) >= topk:
672
+ break
673
+ return out
674
+
675
+ # ============================ 打分逻辑 ============================
676
+
677
+ def _strong_alignment_bonus(ans: str, evids: list) -> float:
678
+ if not ans and not evids:
679
+ return 0.0
680
+ text = (sanitize_for_scoring(ans) or "") + " " + " ".join([str(x) for x in (evids or [])])
681
+ low = text.lower()
682
+ strong = 0
683
+ if ("clinicaltrials.gov" in low or "eudract" in low) and re.search(r"\b(NCT\d{8}|EUCTR-\d{4}-\d{6}-\d{2})\b", low, re.I):
684
+ strong += 1
685
+ if ("pubmed" in low or "doi" in low) and re.search(r"\b10\.\d{4,9}/[-._;()/:A-Z0-9]+", low, re.I):
686
+ strong += 1
687
+ if ("uspto" in low or "epo" in low or "cnipa" in low) and re.search(r"\b(US|EP|CN)\d{5,}|WO\d{7,}\b", low, re.I):
688
+ strong += 1
689
+ if ("fda" in low or "ema" in low or "21 cfr" in low or "iso" in low) and re.search(r"\b(20\d{2}|19\d{2})\b", low, re.I):
690
+ strong += 1
691
+ return min(0.10, 0.03 * strong)
692
+
693
+ def score_candidate(ans_item: dict, cfg: dict, peer_tokens_list=None, dim: str = "", auth_hints: list = None, question: str = "") -> dict:
694
+ ans = (ans_item.get("answer") or "").strip()
695
+ claims = ans_item.get("claims") or []
696
+ evids = ans_item.get("evidence_hints") or []
697
+ tags = ans_item.get("topic_tags") or []
698
+ conf = safe_float(ans_item.get("confidence"), 0.6)
699
+ diag = ans_item.get("diag") or {}
700
+
701
+ s_len = norm01(len(sanitize_for_scoring(ans)), 0, cfg["length_ref_chars"])
702
+ s_clm = norm01(len([c for c in claims if isinstance(c, str) and c.strip()]), 0, cfg["claims_ref"])
703
+ s_evc = norm01(len([e for e in evids if isinstance(e, str) and e.strip()]), 0, cfg["evidence_ref"])
704
+ s_eva = authority_ratio(evids)
705
+ s_evg = coverage_score(evids)
706
+ s_str = looks_structured(ans)
707
+
708
+ # —— 对齐(计分阶段只注入 claims,不注入 question;可通过 A/B 关闭 claims 注入)——
709
+ use_qc = not cfg.get("_ablate_no_question_claims", False)
710
+ s_aln = _alignment_ratio(dim, auth_hints or [], ans, tags, evids,
711
+ question=(question if use_qc else ""),
712
+ claims=(claims if use_qc else []))
713
+
714
+ s_cal = conf
715
+
716
+ # 这里 auth_boost / cov_boost 仅用于 diagnostics,不再进入主权重
717
+ auth_boost = min(1.0, float(diag.get("auth_hits", 0)) / 5.0) if isinstance(diag.get("auth_hits", 0), (int, float)) else 0.0
718
+ cov_boost = min(1.0, float(len(set(diag.get("coverage_bins") or []))) / 4.0)
719
+ s_eva = max(s_eva, auth_boost)
720
+ s_evg = max(s_evg, cov_boost)
721
+
722
+ s_aln = min(1.0, s_aln + _strong_alignment_bonus(ans, evids))
723
+
724
+ if peer_tokens_list:
725
+ me = list(_tokenize_cached(sanitize_for_scoring(ans)))
726
+ sims = [jaccard(me, tks) for tks in peer_tokens_list if tks is not None]
727
+ s_con = sum(sims) / max(1, len(sims))
728
+ else:
729
+ s_con = 0.5
730
+
731
+ pen = 0.0
732
+ oc = overclaim_score(ans)
733
+ if oc > 0.15:
734
+ pen += cfg["penalties"]["overclaim"]
735
+ if s_str < 0.20:
736
+ pen += cfg["penalties"]["understructure"]
737
+ if any(has_redline(str(c)) for c in claims):
738
+ pen += cfg["penalties"]["redline_residual"]
739
+
740
+ drift_raw = _dimension_drift_score(dim, ans, tags, evids)
741
+ if drift_raw > 0:
742
+ scale = 0.4 if drift_raw <= 0.2 else (0.7 if drift_raw <= 0.5 else 1.0)
743
+ pen += min(cfg["penalties"]["dimension_drift"] * scale, drift_raw * cfg["penalties"]["dimension_drift"])
744
+ if (diag.get("cross_dim") is True):
745
+ pen += min(cfg["penalties"]["dimension_drift"], 0.02)
746
+
747
+ fw = cfg["fields_weight"]
748
+ fields_part = (
749
+ fw["length"] * s_len +
750
+ fw["claims"] * s_clm +
751
+ fw["evidence_count"] * s_evc +
752
+ fw["evidence_authority"] * s_eva +
753
+ fw["evidence_coverage"] * s_evg +
754
+ fw["structure"] * s_str +
755
+ fw["alignment"] * s_aln +
756
+ fw["calibrated_confidence"] * s_cal
757
+ )
758
+
759
+ total = (1.0 - cfg["consistency_weight"]) * fields_part + cfg["consistency_weight"] * s_con
760
+ total = max(0.0, min(1.0, total - pen))
761
+
762
+ alpha = max(0.0, min(1.0, fields_part - pen))
763
+
764
+ if (dim == "innovation") and (ans_item.get("diag", {}).get("repro_signal") is True):
765
+ total = min(1.0, total + 0.01)
766
+ alpha = min(1.0, alpha + 0.01)
767
+
768
+ return {
769
+ "scores": {
770
+ "length": s_len,
771
+ "claims": s_clm,
772
+ "evidence_count": s_evc,
773
+ "evidence_authority": s_eva,
774
+ "evidence_coverage": s_evg,
775
+ "structure": s_str,
776
+ "alignment": s_aln,
777
+ "calibrated_confidence": s_cal,
778
+ "consistency": s_con
779
+ },
780
+ "penalties": {
781
+ "overclaim": oc,
782
+ "dimension_drift": drift_raw,
783
+ "applied": pen
784
+ },
785
+ "alpha": alpha,
786
+ "total": total
787
+ }
788
+
789
+ # —— 更强坏候选过滤:返回 (bool_bad, reason)
790
+ def _bad_candidate_with_reason(c: dict, cfg: dict, dim_name: str, auth_hints: list, q_text: str = ""):
791
+ ans_raw = (c.get("answer") or "").strip()
792
+ ans = sanitize_for_scoring(ans_raw)
793
+
794
+ # 1. 明确错误 / 过短
795
+ if c.get("error") is True:
796
+ return True, "error"
797
+ if len(ans) < 12:
798
+ return True, "too_short"
799
+
800
+ # 2. 占位/垃圾行比例:只有当非常高时才视为垃圾
801
+ ph_ratio = _placeholder_ratio(sanitize_for_display(ans_raw))
802
+ if ph_ratio > 0.60:
803
+ return True, "placeholder_noise"
804
+
805
+ # 3. 维度特定参数
806
+ dim_conf = (cfg.get("dimension_specific") or {}).get(dim_name, {})
807
+ min_struct = cfg["filters"]["min_structured_score"]
808
+ min_bullets = dim_conf.get("min_bullet_lines", cfg["filters"]["min_bullet_lines"])
809
+ min_align = float(dim_conf.get("min_alignment_for_keep", cfg["filters"]["min_alignment_for_keep"]))
810
+
811
+ if dim_name == "unknown":
812
+ # unknown 维度对齐阈值稍微低一点
813
+ min_align = min(0.15, min_align)
814
+
815
+ # 4. 结构评分:极差直接丢弃;略差在 soft_window 下放行
816
+ s_struct = looks_structured(ans_raw)
817
+ if s_struct < min_struct:
818
+ if cfg["filters"].get("soft_window", True) and s_struct >= 0.5 * min_struct:
819
+ pass
820
+ else:
821
+ return True, "poor_structure"
822
+
823
+ # 5. 去重后的条目行(用于 bullet 数量 & 中位长度)
824
+ _raw_lines = [ln.strip() for ln in sanitize_for_display(ans_raw).splitlines() if ln.strip()]
825
+ seen = set()
826
+ bullet_lines = []
827
+ for ln in _raw_lines:
828
+ if ln not in seen:
829
+ seen.add(ln)
830
+ bullet_lines.append(ln)
831
+
832
+ # 6. 基本分点要求:不再依赖 evidence/authority 做动态放宽
833
+ if len(bullet_lines) < min_bullets:
834
+ if cfg["filters"].get("soft_window", True) and len(bullet_lines) >= max(1, min_bullets - 1):
835
+ pass
836
+ else:
837
+ return True, "few_bullets"
838
+
839
+ # 7. 行中位长度:过滤口号式 / 极短句堆砌
840
+ if bullet_lines:
841
+ lens = sorted(len(x) for x in bullet_lines)
842
+ med = lens[len(lens)//2]
843
+ if med < cfg["filters"]["min_median_bullet_len"]:
844
+ if cfg["filters"].get("soft_window", True) and med >= max(2, cfg["filters"]["min_median_bullet_len"] - 2):
845
+ pass
846
+ else:
847
+ return True, "too_short_lines"
848
+
849
+ # 8. 语义对齐检查:只用来丢弃严重跑题的 candidate,不再看 evidence/authority
850
+ use_qc = not cfg.get("_ablate_no_question_claims", False)
851
+ aln = _alignment_ratio(
852
+ dim_name,
853
+ auth_hints or [],
854
+ ans_raw,
855
+ c.get("topic_tags") or [],
856
+ c.get("evidence_hints") or [],
857
+ question=(q_text if use_qc else ""),
858
+ claims=(c.get("claims") or [] if use_qc else [])
859
+ )
860
+
861
+ dyn_min_align = float(min_align)
862
+ if aln < dyn_min_align:
863
+ # 若结构还不错,或者只是略低一点,在 soft_window 下放行
864
+ if cfg["filters"].get("soft_window", True) and (aln >= dyn_min_align * 0.6 or s_struct >= 0.5):
865
+ pass
866
+ else:
867
+ return True, "weak_alignment"
868
+
869
+ return False, ""
870
+
871
+ def _fallback_pick(cands: list, dim: str, auth_hints: list):
872
+ def _alignment_ratio_local(answer: str, evids: list):
873
+ # 仅作为兜底时的快速对齐(共性 token 命中)
874
+ corpus = (sanitize_for_scoring(answer or "")) + " " + " ".join([str(h) for h in (evids or [])])
875
+ total = len(auth_hints) or 1
876
+ hit = 0
877
+ low = corpus.lower()
878
+ for h in auth_hints or []:
879
+ if (h or "").strip().lower() in low:
880
+ hit += 1
881
+ return hit / total
882
+
883
+ def _fallback_score(c):
884
+ ans = (c.get("answer") or "")
885
+ evids = c.get("evidence_hints") or []
886
+ s_str = looks_structured(ans)
887
+ s_eva = authority_ratio(evids)
888
+ s_evg = coverage_score(evids)
889
+ s_evc = norm01(len([e for e in evids if str(e).strip()]), 0, 4)
890
+ s_aln = _alignment_ratio_local(ans, evids)
891
+ red_p = 0.0
892
+ if len((c.get("facts_redlined") or [])) >= 3:
893
+ red_p = 0.08
894
+ # 兜底场景:结构 + 对齐 为主,evidence 指标只给很轻的权重
895
+ base = 0.50 * s_str + 0.25 * s_aln + 0.15 * s_evc + 0.05 * s_eva + 0.05 * s_evg
896
+ return max(0.0, base - red_p)
897
+
898
+ ranked = sorted([(i, _fallback_score(c)) for i, c in enumerate(cands)], key=lambda x: -x[1])
899
+ return ranked[0][0] if ranked else 0
900
+
901
+ def _provider_name(cand: dict) -> str:
902
+ pv = (cand.get("provider") or "").strip().lower()
903
+ if pv:
904
+ return pv
905
+ mdl = (cand.get("model") or "").lower()
906
+ if "deepseek" in mdl:
907
+ return "deepseek"
908
+ if "gpt" in mdl or "openai" in mdl or "o" == mdl[:1]:
909
+ return "openai"
910
+ return "default"
911
+
912
+ def _beta_with_sweetspot_and_provider(beta_raw: float, avg_jac: float, avg_ctr: float, cfg: dict, provider: str) -> float:
913
+ lo, hi = cfg["consistency_correction"]["beta"]["jaccard_sweetspot"]
914
+ if avg_jac > hi:
915
+ beta_adj = (1 - 0.05) * beta_raw
916
+ elif avg_jac < lo:
917
+ beta_adj = (1 - 0.05 * (lo - avg_jac) / max(1e-9, lo)) * beta_raw
918
+ else:
919
+ beta_adj = beta_raw
920
+ cal = cfg["consistency_correction"]["beta"]["provider_calibration"].get(
921
+ provider, cfg["consistency_correction"]["beta"]["provider_calibration"]["default"]
922
+ )
923
+ delta = float(cal.get("beta_delta", 0.0))
924
+ beta_final = max(0.0, min(1.0, beta_adj + delta))
925
+ return beta_final
926
+
927
+ def select_best_candidate(cands: list, cfg: dict, dim: str, auth_hints: list, q_text: str = "", last_provider: str = None):
928
+ if not cands:
929
+ return {"best": None, "all": [], "pairwise": {"avg_jaccard": 0.0, "avg_contradiction": 0.0}, "drop_stats": {}}
930
+
931
+ drop_stats = Counter()
932
+ cleaned = []
933
+ for c in cands:
934
+ bad, reason = _bad_candidate_with_reason(c, cfg, dim_name=dim, auth_hints=auth_hints, q_text=q_text)
935
+ if bad:
936
+ drop_stats[reason] += 1
937
+ else:
938
+ cleaned.append(c)
939
+
940
+ if not cleaned:
941
+ idx = _fallback_pick(cands, dim, auth_hints)
942
+ cleaned = [cands[idx]]
943
+
944
+ cands = cleaned
945
+
946
+ tokens = [list(_tokenize_cached(sanitize_for_scoring((c.get("answer") or "")))) for c in cands]
947
+ jac_sum = ctr_sum = 0.0
948
+ pair_cnt = 0
949
+ for i in range(len(cands)):
950
+ for j in range(i + 1, len(cands)):
951
+ pair_cnt += 1
952
+ jv = jaccard(tokens[i], tokens[j])
953
+ cv = contradiction_pair(cands[i].get("answer", ""), cands[j].get("answer", ""))
954
+ jac_sum += jv
955
+ ctr_sum += cv
956
+
957
+ avg_jac = jac_sum / pair_cnt if pair_cnt else 0.0
958
+ avg_ctr = ctr_sum / pair_cnt if pair_cnt else 0.0
959
+
960
+ k1 = cfg["consistency_correction"]["k_contradiction"]
961
+ beta_raw = max(0.0, min(1.0, (1 - k1 * avg_ctr) * (0.5 + 0.5 * avg_jac)))
962
+
963
+ scored = []
964
+ for idx, c in enumerate(cands):
965
+ peer = [tokens[k] if k != idx else None for k in range(len(cands))]
966
+ sc = score_candidate(c, cfg, peer_tokens_list=peer, dim=dim, auth_hints=auth_hints, question=q_text)
967
+ pv = _provider_name(c)
968
+ beta = _beta_with_sweetspot_and_provider(beta_raw, avg_jac, avg_ctr, cfg, provider=pv)
969
+ final = max(0.0, min(1.0, sc.get("alpha", sc.get("total", 0.0)) * beta))
970
+ scored.append((idx, final, sc, beta, pv))
971
+
972
+ def _tie_key(t):
973
+ final = t[1]
974
+ sd = t[2].get("scores", {})
975
+ return (
976
+ -float(final),
977
+ -float(sd.get("evidence_authority", 0.0)),
978
+ -float(sd.get("evidence_coverage", 0.0)),
979
+ -float(sd.get("alignment", 0.0)),
980
+ -float(sd.get("structure", 0.0)),
981
+ -float(sd.get("claims", 0.0)),
982
+ -float(sd.get("length", 0.0)),
983
+ )
984
+
985
+ scored.sort(key=_tie_key)
986
+
987
+ # === Provider 近分轮换策略(新增) ===
988
+ def _pick_with_provider_balance(scored_list, last_pv, margin=0.02):
989
+ """
990
+ 在与top候选分差小于 margin 的范围内,优先选择 provider != last_pv 的候选;
991
+ 若找不到,则保持原top。
992
+ scored_list 元素结构: (idx, final_score, detail_dict, beta, provider)
993
+ """
994
+ if not scored_list:
995
+ return None
996
+ top = scored_list[0]
997
+ if len(scored_list) == 1 or last_pv is None:
998
+ return top
999
+ top_score = top[1]
1000
+ # 只看前三个,避免质量抖动
1001
+ for cand in scored_list[:3]:
1002
+ _, sc_final, _, _, sc_pv = cand
1003
+ if sc_pv != last_pv and (top_score - sc_final) < margin:
1004
+ return cand
1005
+ return top
1006
+
1007
+ picked_tuple = _pick_with_provider_balance(scored, last_provider)
1008
+ if picked_tuple is None:
1009
+ best_idx, final_score, best_detail, best_beta, best_provider = (
1010
+ 0, 0.0, {"scores": {}, "penalties": {}, "alpha": 0.0, "total": 0.0}, 0.0, "default"
1011
+ )
1012
+ else:
1013
+ best_idx, final_score, best_detail, best_beta, best_provider = picked_tuple
1014
+
1015
+ conflict_pen = cfg["penalties"]["contradiction"] * avg_ctr
1016
+ topic_reliability = max(0.0, min(1.0, best_detail.get("total", 0.0) - conflict_pen))
1017
+
1018
+ out_all = []
1019
+ for idx, fscore, detail, b, pv in scored:
1020
+ item = dict(cands[idx])
1021
+ item["_score_detail"] = detail
1022
+ item["_score_total"] = detail.get("total", 0.0)
1023
+ item["_score_alpha"] = detail.get("alpha", 0.0)
1024
+ item["_score_final"] = fscore
1025
+ item["_score_beta"] = b
1026
+ item["_provider_used"] = pv
1027
+ out_all.append(item)
1028
+
1029
+ return {
1030
+ "best": {
1031
+ "index": best_idx,
1032
+ "candidate": cands[best_idx] if cands else None,
1033
+ "score": {
1034
+ "alpha": best_detail.get("alpha", 0.0),
1035
+ "beta": best_beta,
1036
+ "final": final_score,
1037
+ "raw_total": best_detail.get("total", 0.0),
1038
+ "after_topic_conflict": topic_reliability,
1039
+ "avg_pairwise_jaccard": avg_jac,
1040
+ "avg_pairwise_contradiction": avg_ctr
1041
+ },
1042
+ "detail": best_detail,
1043
+ "provider": best_provider
1044
+ },
1045
+ "all": out_all,
1046
+ "pairwise": {"avg_jaccard": avg_jac, "avg_contradiction": avg_ctr},
1047
+ "drop_stats": dict(drop_stats)
1048
+ }
1049
+
1050
+ # ============================ 聚合与报告 ============================
1051
+
1052
+ def aggregate_dimensions(items: list, cfg: dict, qs_cfg: dict):
1053
+ per_question = []
1054
+
1055
+ dim_bucket = defaultdict(list)
1056
+ dropped_reason_bucket = defaultdict(Counter)
1057
+ # —— 用于 provider 近分轮换 ——
1058
+ last_provider_per_dim = {d: None for d in DIM_ORDER + ["unknown"]}
1059
+
1060
+ auth_map = {dim: _authority_hints_from_qs(qs_cfg, dim, limit=8) for dim in DIM_ORDER}
1061
+ auth_map.setdefault("unknown", [])
1062
+
1063
+ provider_stats = Counter()
1064
+ provider_alpha = defaultdict(list)
1065
+ provider_final = defaultdict(list)
1066
+
1067
+ for it in items:
1068
+ dim0 = (it.get("dimension") or "").strip().lower()
1069
+ dim = dim0 if dim0 in DIM_ORDER else "unknown"
1070
+ qidx = it.get("q_index")
1071
+ ques = (it.get("question") or "").strip()
1072
+ cands = it.get("candidates") or []
1073
+
1074
+ picked = select_best_candidate(
1075
+ cands, cfg,
1076
+ dim=dim,
1077
+ auth_hints=auth_map.get(dim, []),
1078
+ q_text=ques,
1079
+ last_provider=last_provider_per_dim.get(dim)
1080
+ )
1081
+ best = picked["best"]
1082
+
1083
+ # 记录本维度上一次选中的 provider,供下一题“近分轮换”微调
1084
+ if best and best.get("candidate"):
1085
+ last_provider_per_dim[dim] = (best.get("provider") or last_provider_per_dim.get(dim))
1086
+
1087
+ if picked.get("drop_stats"):
1088
+ dropped_reason_bucket[dim].update(picked["drop_stats"])
1089
+
1090
+ if best and best["candidate"]:
1091
+ selc = best["candidate"]
1092
+ best_item = {
1093
+ "dimension": dim,
1094
+ "q_index": qidx,
1095
+ "question": ques,
1096
+ "selected": {
1097
+ "provider": selc.get("provider"),
1098
+ "model": selc.get("model"),
1099
+ "variant_id": selc.get("variant_id"),
1100
+ "answer": selc.get("answer"),
1101
+ "claims": selc.get("claims") or [],
1102
+ "evidence_hints": selc.get("evidence_hints") or [],
1103
+ "topic_tags": selc.get("topic_tags") or [],
1104
+ "confidence": safe_float(selc.get("confidence"), 0.6),
1105
+ "alignment_ratio": best["detail"]["scores"].get("alignment", 0.0),
1106
+ "dimension_drift": best["detail"]["penalties"].get("dimension_drift", 0.0),
1107
+ "facts_redlined": selc.get("facts_redlined", []) or [],
1108
+ # ★ 新增:保留通识经验层,供后续 ai_expert_opinion 使用
1109
+ "general_insights": selc.get("general_insights") or []
1110
+ },
1111
+ "score": best["score"],
1112
+ "score_detail": best["detail"],
1113
+ "pairwise": picked["pairwise"],
1114
+ "all_candidates": picked["all"],
1115
+ "auth_hints_used": auth_map.get(dim, []),
1116
+ "drop_stats": picked["drop_stats"], # ← 新增
1117
+ }
1118
+ per_question.append(best_item)
1119
+ dim_bucket[dim].append(best_item)
1120
+
1121
+ pv = (picked.get("best") or {}).get("provider") or "default"
1122
+ provider_stats[pv] += 1
1123
+ provider_alpha[pv].append(float(best["detail"].get("alpha", 0.0)))
1124
+ provider_final[pv].append(float(best["score"].get("final", 0.0)))
1125
+ else:
1126
+ per_question.append({
1127
+ "dimension": dim, "q_index": qidx, "question": ques,
1128
+ "selected": None,
1129
+ "score": {"alpha": 0.0, "beta": 0.0, "final": 0.0,
1130
+ "raw_total": 0.0, "after_topic_conflict": 0.0,
1131
+ "avg_pairwise_jaccard": 0.0, "avg_pairwise_contradiction": 0.0},
1132
+ "score_detail": {},
1133
+ "pairwise": {"avg_jaccard": 0.0, "avg_contradiction": 0.0},
1134
+ "all_candidates": [],
1135
+ "auth_hints_used": auth_map.get(dim, []),
1136
+ "drop_stats": picked["drop_stats"] # ← 新增
1137
+ })
1138
+
1139
+ per_dimension = {}
1140
+ dims_for_report = list(DIM_ORDER)
1141
+ if "unknown" in dim_bucket:
1142
+ dims_for_report.append("unknown")
1143
+
1144
+ drop_reasons_global = Counter()
1145
+ for d, cnts in dropped_reason_bucket.items():
1146
+ drop_reasons_global.update(cnts)
1147
+
1148
+ for dim in dims_for_report:
1149
+ qs = sorted([x for x in per_question if x["dimension"] == dim], key=lambda z: z["q_index"])
1150
+ if not qs:
1151
+ per_dimension[dim] = {
1152
+ "avg": 0.0, "n": 0,
1153
+ "avg_alignment": 0.0, "avg_drift": 0.0,
1154
+ "strengths": [], "risks": [], "snippets": [],
1155
+ "auth_hints": auth_map.get(dim, []),
1156
+ "redlined_samples": [],
1157
+ "dropped_reasons": {},
1158
+ "top_evidence_phrases": [],
1159
+ "general_insights": [],
1160
+ "explain": {
1161
+ "top_contributors": [],
1162
+ "top_penalties": []
1163
+ }
1164
+ }
1165
+ continue
1166
+
1167
+ # ✅ 安全取每题 final 分数(缺就当 0)
1168
+ scores_final = []
1169
+ for q in qs:
1170
+ score_block = q.get("score") or {}
1171
+ val = score_block.get("final", score_block.get("after_topic_conflict", 0.0))
1172
+ try:
1173
+ scores_final.append(float(val))
1174
+ except (TypeError, ValueError):
1175
+ scores_final.append(0.0)
1176
+
1177
+ # ✅ 新逻辑:均值 + 最大值 加权,拉开维度差异
1178
+ if scores_final:
1179
+ mean_sc = sum(scores_final) / len(scores_final)
1180
+ max_sc = max(scores_final)
1181
+ # 强维度通常会有几题明显高分;弱维度整体偏平
1182
+ avg = 0.6 * mean_sc + 0.4 * max_sc
1183
+ else:
1184
+ mean_sc = 0.0
1185
+ max_sc = 0.0
1186
+ avg = 0.0
1187
+
1188
+ strengths, risks, snippets = [], [], []
1189
+ aln_vals, drf_vals = [], []
1190
+ redlined_samples = []
1191
+
1192
+ evid_pool = []
1193
+ gi_pool = [] # ★ 新增:聚合 general_insights
1194
+ contrib_counter = Counter()
1195
+ penalty_counter = Counter()
1196
+
1197
+ for q in qs:
1198
+ sel = q.get("selected")
1199
+ if sel:
1200
+ eva = authority_ratio(sel.get("evidence_hints") or [])
1201
+ evg = coverage_score(sel.get("evidence_hints") or [])
1202
+ aln = float(sel.get("alignment_ratio", 0.0))
1203
+ drf = float(sel.get("dimension_drift", 0.0))
1204
+
1205
+ aln_vals.append(aln)
1206
+ drf_vals.append(drf)
1207
+
1208
+ # 优势判断:不再强依赖权威/覆盖,只要 claims + 对齐即可
1209
+ if len(sel.get("claims") or []) >= 2 and aln >= 0.40:
1210
+ strengths.append(
1211
+ f"Q{q['q_index']}:要点充分,结构/对齐较好(auth={eva:.2f}, cover={evg:.2f}, align={aln:.2f})"
1212
+ )
1213
+ if drf > 0.0:
1214
+ risks.append(f"Q{q['q_index']}:跨维度串味迹象(drift={drf:.2f}),建议人工复核维度边界")
1215
+ if (q.get("score_detail") or {}).get("penalties", {}).get("applied", 0.0) > 0.0:
1216
+ risks.append("存在过度断言/结构不足/红线残留等扣分(请抽检)")
1217
+
1218
+ ans_txt = sel.get("answer") or ""
1219
+ clean_snip = sanitize_for_display(ans_txt)
1220
+ snippets.append(f"Q{q['q_index']}:{(clean_snip[:160] + '…') if clean_snip else '(无)'}")
1221
+
1222
+ for s in (sel.get("facts_redlined") or [])[:2]:
1223
+ if s not in redlined_samples and len(redlined_samples) < 8:
1224
+ redlined_samples.append(s)
1225
+
1226
+ evid_pool.extend(sel.get("evidence_hints") or [])
1227
+ gi_pool.extend(sel.get("general_insights") or []) # ★ 新增
1228
+
1229
+ sd = q.get("score_detail", {}).get("scores", {})
1230
+ rank_pairs = sorted(sd.items(), key=lambda x: -float(x[1]))[:2]
1231
+ for k, _ in rank_pairs:
1232
+ contrib_counter[k] += 1
1233
+ pen = q.get("score_detail", {}).get("penalties", {})
1234
+ if pen.get("applied", 0.0) > 0.0:
1235
+ if pen.get("dimension_drift", 0.0) > 0:
1236
+ penalty_counter["dimension_drift"] += 1
1237
+ if pen.get("overclaim", 0.0) > 0.15:
1238
+ penalty_counter["overclaim"] += 1
1239
+ if sd.get("structure", 1.0) < 0.2:
1240
+ penalty_counter["understructure"] += 1
1241
+ else:
1242
+ risks.append(f"Q{q['q_index']}:无有效答案")
1243
+ snippets.append(f"Q{q['q_index']}:(无)")
1244
+
1245
+ top_evid = _top_evidence_phrases(evid_pool, topk=3)
1246
+ gi_agg = _uniq_general_insights(gi_pool, topk=10)
1247
+
1248
+ per_dimension[dim] = {
1249
+ "avg": avg,
1250
+ "n": len(qs),
1251
+ "avg_alignment": (sum(aln_vals) / max(1, len(aln_vals))) if aln_vals else 0.0,
1252
+ "avg_drift": (sum(drf_vals) / max(1, len(drf_vals))) if drf_vals else 0.0,
1253
+ "strengths": strengths[:cfg["adv_topk"]],
1254
+ "risks": risks[:cfg["adv_topk"]],
1255
+ "snippets": snippets[:6],
1256
+ "auth_hints": auth_map.get(dim, []),
1257
+ "redlined_samples": redlined_samples,
1258
+ "dropped_reasons": dict(dropped_reason_bucket.get(dim, {})),
1259
+ "top_evidence_phrases": top_evid,
1260
+ "general_insights": gi_agg,
1261
+ "explain": {
1262
+ "top_contributors": [k for k, _ in contrib_counter.most_common(3)],
1263
+ "top_penalties": [k for k, _ in penalty_counter.most_common(3)]
1264
+ }
1265
+ }
1266
+
1267
+ # overall
1268
+ dim_score = 0.0
1269
+ weight_sum = 0.0
1270
+ total_q_count = len(per_question)
1271
+ unknown_q_count = len([1 for q in per_question if q["dimension"] == "unknown"])
1272
+ for dim, info in per_dimension.items():
1273
+ if dim not in DIM_ORDER:
1274
+ continue
1275
+ w = (cfg.get("dimension_weight") or {}).get(dim, 1.0)
1276
+ dim_score += w * float(info.get("avg", 0.0))
1277
+ weight_sum += w
1278
+ overall_score = dim_score / max(1e-9, weight_sum if weight_sum > 0 else 1.0)
1279
+
1280
+ all_conf, all_jac, all_ctr, cnt = [], 0.0, 0.0, 0
1281
+ for q in per_question:
1282
+ if q.get("selected"):
1283
+ all_conf.append(safe_float(q["selected"].get("confidence"), 0.6))
1284
+ all_jac += float(q.get("pairwise", {}).get("avg_jaccard", 0.0))
1285
+ all_ctr += float(q.get("pairwise", {}).get("avg_contradiction", 0.0))
1286
+ cnt += 1
1287
+ mean_conf = sum(all_conf) / max(1, len(all_conf))
1288
+ mean_jac = all_jac / max(1, cnt)
1289
+ mean_ctr = all_ctr / max(1, cnt)
1290
+ overall_confidence = max(0.0, min(1.0, (0.5 * mean_conf + 0.3 * mean_jac + 0.2 * (1 - mean_ctr))))
1291
+
1292
+ provider_summary = {}
1293
+ for pv, n in provider_stats.items():
1294
+ provider_summary[pv] = {
1295
+ "selected_count": n,
1296
+ "avg_alpha": sum(provider_alpha[pv]) / max(1, len(provider_alpha[pv])),
1297
+ "avg_final": sum(provider_final[pv]) / max(1, len(provider_final[pv]))
1298
+ }
1299
+
1300
+ drop_g = dict(drop_reasons_global)
1301
+ placeholder_cnt = drop_g.get("placeholder_noise", 0)
1302
+ few_bullets_cnt = drop_g.get("few_bullets", 0)
1303
+
1304
+ dim_median_bullets = {}
1305
+ for dim in per_dimension:
1306
+ qs_d = [q for q in per_question if q["dimension"] == dim and q.get("selected")]
1307
+ nums = []
1308
+ for q in qs_d:
1309
+ cleaned = sanitize_for_display(q["selected"]["answer"] or "")
1310
+ nums.append(len([ln for ln in cleaned.splitlines() if ln.strip()]))
1311
+ if nums:
1312
+ nums.sort()
1313
+ dim_median_bullets[dim] = nums[len(nums)//2]
1314
+ else:
1315
+ dim_median_bullets[dim] = 0
1316
+
1317
+ overall = {
1318
+ "overall_score": overall_score,
1319
+ "overall_confidence": overall_confidence,
1320
+ "mean_pairwise_jaccard": mean_jac,
1321
+ "mean_pairwise_contradiction": mean_ctr,
1322
+ "unknown_ratio": (unknown_q_count / max(1, total_q_count)),
1323
+ "drop_reasons_global": dict(drop_reasons_global),
1324
+ "provider_stats": provider_summary,
1325
+ "placeholder_ratio_global": placeholder_cnt / max(1, sum(drop_g.values())) if drop_g else 0.0,
1326
+ "few_bullets_ratio_global": few_bullets_cnt / max(1, sum(drop_g.values())) if drop_g else 0.0,
1327
+ "dim_median_bullets": dim_median_bullets
1328
+ }
1329
+
1330
+ return per_question, per_dimension, overall
1331
+
1332
+ def build_report_md(pid: str, meta: dict, per_dim: dict, overall: dict, cfg: dict) -> str:
1333
+ lines = []
1334
+ lines.append(f"# 项目后处理报告 · {pid}")
1335
+ lines.append("")
1336
+ lines.append(f"- 生成时间:{now_str()}")
1337
+ if meta:
1338
+ m = {k: meta.get(k) for k in ("generated_at", "pid", "schema") if k in meta}
1339
+ if "args" in meta:
1340
+ m["args"] = meta["args"]
1341
+ lines.append(f"- 元信息:{json.dumps(m, ensure_ascii=False)}")
1342
+ lines.append("")
1343
+ lines.append("## 总览")
1344
+ sc = overall["overall_score"]
1345
+ cf = overall["overall_confidence"]
1346
+ lines.append(f"- 综合评分(0~1):**{sc:.3f}** {bar(sc, cfg['bar_symbols'])}")
1347
+ lines.append(f"- 综合信心度(0~1):**{cf:.3f}** {bar(cf, cfg['bar_symbols'])}")
1348
+ lines.append(f"- 全局一致性(平均 Jaccard):**{overall['mean_pairwise_jaccard']:.3f}**")
1349
+ lines.append(f"- 全局冲突度(平均):**{overall['mean_pairwise_contradiction']:.3f}**")
1350
+
1351
+ if "drop_reasons_global" in overall and overall["drop_reasons_global"]:
1352
+ drg = overall["drop_reasons_global"]
1353
+ disp = ", ".join([f"{k}:{v}" for k, v in sorted(drg.items(), key=lambda x: (-x[1], x[0]))])
1354
+ lines.append(f"- 候选被丢弃原因(全局Top):{disp}")
1355
+ lines.append(f"- 估计 placeholder 噪声占比:{overall.get('placeholder_ratio_global', 0.0):.1%}")
1356
+ lines.append(f"- 估计 few_bullets 占比:{overall.get('few_bullets_ratio_global', 0.0):.1%}")
1357
+
1358
+ if "unknown" in per_dim and per_dim["unknown"]["n"] > 0:
1359
+ unk_n = per_dim["unknown"]["n"]
1360
+ unk_ratio = overall.get("unknown_ratio", 0.0)
1361
+ lines.append(f"- 警示:存在 **{unk_n}** 道题落在 `unknown` 维度(占比 {unk_ratio:.1%}),建议复核问题集与维度抽取。")
1362
+ if unk_ratio > cfg.get("unknown_warn_ratio", 0.10):
1363
+ lines.append(f"- **严重提示**:unknown 占比超过 {int(cfg.get('unknown_warn_ratio', 0.10)*100)}% 的警戒阈值。")
1364
+
1365
+ pvstats = overall.get("provider_stats", {})
1366
+ if pvstats:
1367
+ parts = []
1368
+ for k, v in pvstats.items():
1369
+ parts.append(f"{k}: 选中{v['selected_count']} | α均值={v['avg_alpha']:.3f} | final均值={v['avg_final']:.3f}")
1370
+ lines.append(f"- Provider 统计:{'; '.join(parts)}")
1371
+
1372
+ lines.append("")
1373
+ lines.append("## 维度分解")
1374
+
1375
+ dims_in_report = list(DIM_ORDER) + [d for d in per_dim.keys() if d not in DIM_ORDER]
1376
+
1377
+ for dim in dims_in_report:
1378
+ if dim not in per_dim:
1379
+ continue
1380
+ info = per_dim[dim]
1381
+ avg = info["avg"]
1382
+ lines.append(f"### {dim} · 评分 {avg:.3f} {bar(avg, cfg['bar_symbols'])}")
1383
+ if info.get("auth_hints"):
1384
+ lines.append(f"- 参考方向(非事实):{'; '.join(info['auth_hints'])}")
1385
+ lines.append(f"- 对齐均值/漂移均值:**{info.get('avg_alignment', 0.0):.2f} / {info.get('avg_drift', 0.0):.2f}**")
1386
+ if "explain" in info:
1387
+ ex = info["explain"] or {}
1388
+ if ex.get("top_contributors"):
1389
+ lines.append(f"- 主要得分贡献因子:{', '.join(ex['top_contributors'])}")
1390
+ if ex.get("top_penalties"):
1391
+ lines.append(f"- 主要扣分因子:{', '.join(ex['top_penalties'])}")
1392
+ if info.get("top_evidence_phrases"):
1393
+ lines.append("- **Top 证据短语(权威命中优先)**:")
1394
+ for s in info["top_evidence_phrases"]:
1395
+ lines.append(f" - {s}")
1396
+ gi_list = info.get("general_insights") or []
1397
+ if gi_list:
1398
+ lines.append("- **行业通识要点(general_insights,通用建议,不代表本项目已达成)**:")
1399
+ for s in gi_list[:5]:
1400
+ lines.append(f" - {s}")
1401
+ if info["strengths"]:
1402
+ lines.append("- **优势**:")
1403
+ for s in info["strengths"]:
1404
+ lines.append(f" - {s}")
1405
+ if info["risks"]:
1406
+ lines.append("- **风险**:")
1407
+ for r in info["risks"]:
1408
+ lines.append(f" - {r}")
1409
+ if info["snippets"]:
1410
+ lines.append("- **代表性片段**:")
1411
+ for sn in info["snippets"]:
1412
+ lines.append(f" - {sn}")
1413
+ redlined_samples = info.get("redlined_samples") or []
1414
+ if redlined_samples:
1415
+ lines.append("- **已转为检索线索的原句(抽样)**:")
1416
+ for s in redlined_samples[:6]:
1417
+ lines.append(f" - {s}")
1418
+ drop_r = info.get("dropped_reasons") or {}
1419
+ if drop_r:
1420
+ lines.append("- **被丢弃候选统计**(原因:次数):")
1421
+ disp = ", ".join([f"{k}:{v}" for k, v in sorted(drop_r.items(), key=lambda x: (-x[1], x[0]))])
1422
+ lines.append(f" - {disp}")
1423
+ lines.append(f"- 中位条目数(估计):{overall.get('dim_median_bullets', {}).get(dim, 0)}")
1424
+ lines.append("")
1425
+ lines.append("> 注:本报告仅基于候选答案的结构化指标(长度/要点/证据提示/权威度/覆盖度/结构/一致性/置信度/维度对齐等),"
1426
+ "以及 LLM 给出的“行业通识建议”(general_insights)。通识建议仅为行业基准参考,并不代表本项目已经实现或满足相关要求。"
1427
+ "定稿前建议人工抽检与项目原文/事实证据对齐。")
1428
+ return "\n".join(lines)
1429
+
1430
+ # ============================ 主流程 ============================
1431
+
1432
+ def main():
1433
+ ap = argparse.ArgumentParser(description="Post-processing for structured candidates (no LLM calls).")
1434
+ ap.add_argument("--pid", type=str, default="", help="提案ID,不填则自动选择 refined_answers 下最新目录")
1435
+ ap.add_argument("--input", type=str, default="", help="可选:all_refined_items.json 的绝对/相对路径")
1436
+
1437
+ # === A/B 开关 ===
1438
+ ap.add_argument("--ablate_no_question_claims", action="store_true",
1439
+ help="关闭对齐语料中的 question/claims 注入(用于消融实验)")
1440
+ ap.add_argument("--ablate_no_dyn_relax", action="store_true",
1441
+ help="关闭基于权威/覆盖触发的 min_alignment 放宽(已废弃,仅保留兼容)")
1442
+
1443
+ args = ap.parse_args()
1444
+
1445
+ cfg = load_config()
1446
+
1447
+ # 把 A/B 开关塞入 cfg,便于下游函数访问
1448
+ cfg["_ablate_no_question_claims"] = bool(args.ablate_no_question_claims)
1449
+ cfg["_ablate_no_dyn_relax"] = bool(args.ablate_no_dyn_relax)
1450
+
1451
+ # 读取问题集(用于权威/题材 hints 对齐)
1452
+ if QS_CONF_PATH.exists():
1453
+ try:
1454
+ qs_cfg = read_json(QS_CONF_PATH)
1455
+ except Exception:
1456
+ qs_cfg = {}
1457
+ else:
1458
+ qs_cfg = {}
1459
+
1460
+ if args.input:
1461
+ refined_path = Path(args.input)
1462
+ if not refined_path.exists():
1463
+ raise FileNotFoundError(f"未找到输入文件:{refined_path}")
1464
+ pid = refined_path.parent.name
1465
+ else:
1466
+ pid = args.pid.strip() or detect_latest_pid()
1467
+ if not pid:
1468
+ raise RuntimeError("未检测到 refined_answers 下的最新项目目录,也未提供 --pid / --input")
1469
+ refined_path = REFINED_ROOT / pid / "all_refined_items.json"
1470
+ if not refined_path.exists():
1471
+ raise FileNotFoundError(f"未找到文件:{refined_path}")
1472
+
1473
+ data = read_json(refined_path)
1474
+ meta = (data.get("meta") or {})
1475
+ schema = meta.get("schema", "")
1476
+
1477
+ # 新版支持 llm_answering.v2 以及 refined_items.v2.proposal_aware_with_general_insights
1478
+ allowed_schemas = {"", "llm_answering.v2", "refined_items.v2.proposal_aware_with_general_insights"}
1479
+ if schema and schema not in allowed_schemas:
1480
+ print(f"⚠️ 警告:输入 schema={schema}(预期 {allowed_schemas} 之一),将继续处理。")
1481
+
1482
+ # items / questions 兼容读取(新版 llm_answering 使用 items)
1483
+ items = data.get("items") or data.get("questions") or []
1484
+
1485
+ bad = []
1486
+ for i, it in enumerate(items):
1487
+ if "dimension" not in it or "q_index" not in it or "question" not in it or "candidates" not in it:
1488
+ bad.append(i)
1489
+ if bad:
1490
+ raise ValueError(f"输入 items 中存在缺失必要字段的条目:索引 {bad[:10]} ...,请检查 llm_answering 输出结构。")
1491
+
1492
+ per_question, per_dimension, overall = aggregate_dimensions(items, cfg, qs_cfg)
1493
+
1494
+ out_dir = REFINED_ROOT / pid / "postproc"
1495
+ out_dir.mkdir(parents=True, exist_ok=True)
1496
+
1497
+ metrics = {
1498
+ "meta": {
1499
+ "pid": pid,
1500
+ "generated_at": now_str(),
1501
+ "source_file": str(refined_path),
1502
+ "schema": schema or "",
1503
+ "args": meta.get("args", {})
1504
+ },
1505
+ "config_used": cfg,
1506
+ "overall": overall,
1507
+ "dimensions": per_dimension,
1508
+ "questions": per_question
1509
+ }
1510
+ write_json(out_dir / "metrics.json", metrics)
1511
+
1512
+ # 逐题选中结果(便于人工抽检)
1513
+ write_json(out_dir / "selected_by_question.json", per_question)
1514
+
1515
+ # final_payload(下游报告的统一入口)
1516
+ final_payload = {
1517
+ "meta": {"pid": pid, "generated_at": now_str()},
1518
+ "dimensions": {}
1519
+ }
1520
+ for dim in DIM_ORDER:
1521
+ qs = [q for q in per_question if q["dimension"] == dim and q.get("selected")]
1522
+ final_payload["dimensions"][dim] = {
1523
+ "score": round(float(per_dimension.get(dim, {}).get("avg", 0.0)) * 100, 1),
1524
+ "qas": [
1525
+ {
1526
+ "q": q["question"],
1527
+ "answer": q["selected"]["answer"],
1528
+ "claims": q["selected"]["claims"],
1529
+ "evidence_hints": q["selected"]["evidence_hints"],
1530
+ "topic_tags": q["selected"].get("topic_tags", []),
1531
+ "provider": q["selected"].get("provider"),
1532
+ "model": q["selected"].get("model"),
1533
+ "alignment": q["selected"].get("alignment_ratio", 0.0),
1534
+ "dimension_drift": q["selected"].get("dimension_drift", 0.0),
1535
+ "confidence": q["selected"]["confidence"],
1536
+ "caveats": "",
1537
+ # ★ 新增:逐问通识经验层
1538
+ "general_insights": q["selected"].get("general_insights", [])
1539
+ } for q in qs
1540
+ ],
1541
+ "rationales": (per_dimension.get(dim, {}).get("strengths") or [])[:3],
1542
+ # ★ 新增:维度级聚合通识经验层,供 ai_expert_opinion 作为“行业经验层”使用
1543
+ "general_insights": per_dimension.get(dim, {}).get("general_insights", [])
1544
+ }
1545
+ write_json(out_dir / "final_payload.json", final_payload)
1546
+
1547
+ report_md = build_report_md(pid, meta, per_dimension, overall, cfg)
1548
+ (out_dir / "report.md").write_text(report_md, encoding="utf-8")
1549
+
1550
+ # 丢弃原因调试(仅写逐题的精简信息)
1551
+ write_json(
1552
+ out_dir / "drops_debug.json",
1553
+ [
1554
+ {
1555
+ "dimension": q.get("dimension"),
1556
+ "q_index": q.get("q_index"),
1557
+ "question": q.get("question"),
1558
+ "drop_stats": q.get("drop_stats", {})
1559
+ }
1560
+ for q in metrics["questions"]
1561
+ if q.get("drop_stats") # 只有存在统计才写入
1562
+ ]
1563
+ )
1564
+
1565
+ print(f"✅ metrics.json -> {out_dir/'metrics.json'}")
1566
+ print(f"✅ selected_by_question.json -> {out_dir/'selected_by_question.json'}")
1567
+ print(f"✅ final_payload.json -> {out_dir/'final_payload.json'}")
1568
+ print(f"✅ report.md -> {out_dir/'report.md'}")
1569
+
1570
+ total_q = len(metrics["questions"])
1571
+ dim_brief = ", ".join(f"{d}:{info['n']}" for d, info in metrics["dimensions"].items())
1572
+ ov = metrics["overall"]
1573
+ print(f"ℹ️ 题目数:{total_q} | 维度题量:{dim_brief}")
1574
+ print(f"ℹ️ 综合评分={ov['overall_score']:.3f} 信心度={ov['overall_confidence']:.3f} "
1575
+ f"Jaccard={ov['mean_pairwise_jaccard']:.3f} 冲突度={ov['mean_pairwise_contradiction']:.3f}")
1576
+ if "unknown" in per_dimension and per_dimension["unknown"]["n"] > 0:
1577
+ unk_ratio = ov.get("unknown_ratio", 0.0)
1578
+ print(f"⚠️ 提示:存在 {per_dimension['unknown']['n']} 道题被归入 unknown 维度(占比 {unk_ratio:.1%}),请检查问题集/维度抽取。")
1579
+
1580
+ drg = ov.get("drop_reasons_global", {})
1581
+ if drg:
1582
+ disp = ", ".join([f"{k}:{v}" for k, v in sorted(drg.items(), key=lambda x: (-x[1], x[0]))])
1583
+ print(f"ℹ️ 丢弃原因(全局Top):{disp}")
1584
+
1585
+ pvstats = ov.get("provider_stats", {})
1586
+ if pvstats:
1587
+ parts = []
1588
+ for k, v in pvstats.items():
1589
+ parts.append(f"{k}: 选中{v['selected_count']} | α均值={v['avg_alpha']:.3f} | final均值={v['avg_final']:.3f}")
1590
+ print(f"ℹ️ Provider 统计:{'; '.join(parts)}")
1591
+
1592
+ print("🎯 后处理完成。")
1593
+
1594
+ if __name__ == "__main__":
1595
+ main()
src/tools/prepare_proposal_text.py ADDED
@@ -0,0 +1,207 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ Stage 0: 提案文本准备(基础 OCR + 多模态页面重建)
4
+ """
5
+ from __future__ import annotations
6
+
7
+ import os
8
+ import json
9
+ import argparse
10
+ from pathlib import Path
11
+ from typing import Dict, Any, List
12
+
13
+ import pdfplumber
14
+ from pdf2image import convert_from_path
15
+ from PIL import Image
16
+ import pytesseract
17
+ pytesseract.pytesseract.tesseract_cmd = "tesseract"
18
+ from docx import Document
19
+
20
+ from .layout_reconstruction import build_document_semantics, save_document_semantics
21
+
22
+ MIN_TEXT_CHARS_PER_PAGE = 30
23
+ TESSERACT_LANG = os.getenv("TESS_LANG", "chi_sim+eng")
24
+
25
+
26
+ def detect_file_type(path: Path) -> str:
27
+ suffix = path.suffix.lower()
28
+ if suffix == ".pdf":
29
+ return "pdf"
30
+ if suffix in [".docx", ".doc"]:
31
+ return "docx"
32
+ if suffix in [".txt", ".md"]:
33
+ return "txt"
34
+ raise ValueError(f"暂不支持的文件类型: {suffix}")
35
+
36
+
37
+ def find_latest_proposal() -> Path:
38
+ base_dir = Path(__file__).resolve().parents[2]
39
+ proposals_dir = base_dir / "src" / "data" / "proposals"
40
+ if not proposals_dir.exists():
41
+ raise FileNotFoundError(f"未找到提案目录: {proposals_dir}")
42
+ candidates = [
43
+ p for p in proposals_dir.iterdir()
44
+ if p.is_file() and p.suffix.lower() in [".pdf", ".docx", ".doc", ".txt", ".md"]
45
+ ]
46
+ if not candidates:
47
+ raise FileNotFoundError(f"提案目录中没有可用文件: {proposals_dir}")
48
+ latest = max(candidates, key=lambda x: x.stat().st_mtime)
49
+ print(f"[INFO] [auto] 选中最新提案文件: {latest}")
50
+ return latest
51
+
52
+
53
+ def ocr_page_from_pdf(pdf_path: Path, page_index: int) -> str:
54
+ try:
55
+ images = convert_from_path(str(pdf_path), first_page=page_index + 1, last_page=page_index + 1)
56
+ except Exception as e:
57
+ print(f"[WARN] convert_from_path 失败 (page {page_index+1}): {e}")
58
+ return ""
59
+ if not images:
60
+ return ""
61
+ image: Image.Image = images[0]
62
+ try:
63
+ return pytesseract.image_to_string(image, lang=TESSERACT_LANG)
64
+ except Exception as e:
65
+ print(f"[WARN] OCR 失败 (page {page_index+1}): {e}")
66
+ return ""
67
+
68
+
69
+ def extract_from_pdf(pdf_path: Path, use_ocr: bool = True):
70
+ pages_text, page_sources = [], []
71
+ with pdfplumber.open(pdf_path) as pdf:
72
+ print(f"[INFO] PDF 页面数: {len(pdf.pages)}")
73
+ for i, page in enumerate(pdf.pages):
74
+ txt = (page.extract_text() or "").strip()
75
+ if txt and len(txt) >= MIN_TEXT_CHARS_PER_PAGE:
76
+ pages_text.append(txt)
77
+ page_sources.append("pdf_text")
78
+ print(f" - 第 {i+1} 页: 使用 pdfplumber 文本,长度 {len(txt)}")
79
+ else:
80
+ if use_ocr:
81
+ print(f" - 第 {i+1} 页: 文本太少({len(txt)} chars),尝试 OCR...")
82
+ ocr_txt = (ocr_page_from_pdf(pdf_path, page_index=i) or "").strip()
83
+ if ocr_txt:
84
+ pages_text.append(ocr_txt)
85
+ page_sources.append("ocr")
86
+ print(f" -> OCR 成功,长度 {len(ocr_txt)}")
87
+ else:
88
+ pages_text.append("")
89
+ page_sources.append("empty")
90
+ print(" -> OCR 也没有提取到文本")
91
+ else:
92
+ pages_text.append(txt)
93
+ page_sources.append("pdf_text_empty")
94
+ return pages_text, page_sources
95
+
96
+
97
+ def extract_from_docx(docx_path: Path):
98
+ doc = Document(str(docx_path))
99
+ paragraphs = [p.text.strip() for p in doc.paragraphs if p.text and p.text.strip()]
100
+ return ["\n".join(paragraphs)], ["docx"]
101
+
102
+
103
+ def extract_from_txt(txt_path: Path):
104
+ content = Path(txt_path).read_text(encoding="utf-8", errors="ignore").strip()
105
+ return [content], ["txt"]
106
+
107
+
108
+ def _build_pages_json(pages_text: List[str], page_sources: List[str]) -> List[Dict[str, Any]]:
109
+ pages_data: List[Dict[str, Any]] = []
110
+ offset = 0
111
+ for i, (txt, src) in enumerate(zip(pages_text, page_sources)):
112
+ char_len = len(txt)
113
+ page_start = offset
114
+ page_end = offset + char_len
115
+ pages_data.append({
116
+ "page_index": i + 1,
117
+ "source": src,
118
+ "char_len": char_len,
119
+ "global_char_start": page_start,
120
+ "global_char_end": page_end,
121
+ "text": txt,
122
+ })
123
+ offset = page_end + 2
124
+ return pages_data
125
+
126
+
127
+ def prepare_text(file_path: Path, proposal_id: str, use_ocr: bool = True):
128
+ file_path = Path(file_path)
129
+ if not file_path.exists():
130
+ raise FileNotFoundError(file_path)
131
+
132
+ file_type = detect_file_type(file_path)
133
+ print(f"[INFO] 开始提取文本: {file_path} (type={file_type})")
134
+
135
+ if file_type == "pdf":
136
+ pages_text, page_sources = extract_from_pdf(file_path, use_ocr=use_ocr)
137
+ elif file_type == "docx":
138
+ pages_text, page_sources = extract_from_docx(file_path)
139
+ elif file_type == "txt":
140
+ pages_text, page_sources = extract_from_txt(file_path)
141
+ else:
142
+ raise ValueError(f"未知文件类型: {file_type}")
143
+
144
+ base_dir = Path(__file__).resolve().parents[2]
145
+ out_dir = base_dir / "src" / "data" / "prepared" / proposal_id
146
+ out_dir.mkdir(parents=True, exist_ok=True)
147
+
148
+ full_text = "\n\n".join(pages_text)
149
+ full_text_path = out_dir / "full_text.txt"
150
+ full_text_path.write_text(full_text, encoding="utf-8")
151
+ print(f"[OK] full_text.txt 写入完成: {full_text_path} (长度 {len(full_text)} 字符)")
152
+
153
+ pages_data = _build_pages_json(pages_text, page_sources)
154
+ pages_json_path = out_dir / "pages.json"
155
+ pages_json_path.write_text(json.dumps(pages_data, ensure_ascii=False, indent=2), encoding="utf-8")
156
+ print(f"[OK] pages.json 写入完成: {pages_json_path}")
157
+
158
+ page_semantics_path = None
159
+ reconstructed_full_text_path = None
160
+ page_images_dir = None
161
+ if file_type == "pdf":
162
+ try:
163
+ print("[INFO] 开始多模态 Stage 0 重建:pdf2image -> layout/table -> page_semantics")
164
+ doc_sem = build_document_semantics(file_path, out_dir=out_dir)
165
+ saved = save_document_semantics(doc_sem, out_dir=out_dir)
166
+ page_semantics_path = saved["page_semantics_path"]
167
+ reconstructed_full_text_path = saved["reconstructed_full_text_path"]
168
+ page_images_dir = saved["page_images_dir"]
169
+ reconstructed_full_text = doc_sem.get("reconstructed_full_text", "").strip()
170
+ if reconstructed_full_text:
171
+ full_text_path.write_text(reconstructed_full_text, encoding="utf-8")
172
+ print(f"[OK] 已用 reconstructed_full_text 覆盖 full_text.txt,长度 {len(reconstructed_full_text)} 字符")
173
+ except Exception as e:
174
+ print(f"[WARN] 多模态 Stage 0 重建失败,继续使用基础文本: {e}")
175
+
176
+ return {
177
+ "proposal_id": proposal_id,
178
+ "file_type": file_type,
179
+ "out_dir": str(out_dir),
180
+ "full_text_path": str(full_text_path),
181
+ "pages_json_path": str(pages_json_path),
182
+ "page_semantics_path": page_semantics_path,
183
+ "reconstructed_full_text_path": reconstructed_full_text_path,
184
+ "page_images_dir": page_images_dir,
185
+ "num_pages": len(pages_text),
186
+ }
187
+
188
+
189
+ def main():
190
+ parser = argparse.ArgumentParser(description="Stage 0: 提案文本准备(含 OCR + 多模态页面重建 + 自动选最新提案)")
191
+ parser.add_argument("--file", required=False, help="提案文件路径(PDF/DOCX/TXT)。不填则自动选最新提案")
192
+ parser.add_argument("--proposal_id", required=False, help="提案 ID(用于输出目录名,不填则用文件名)")
193
+ parser.add_argument("--no_ocr", action="store_true", help="禁用 OCR(仅调试用)")
194
+ args = parser.parse_args()
195
+
196
+ file_path = Path(args.file) if args.file else find_latest_proposal()
197
+ if args.file:
198
+ print(f"[INFO] 使用用户指定文件: {file_path}")
199
+ proposal_id = args.proposal_id or file_path.stem
200
+ info = prepare_text(file_path, proposal_id, use_ocr=not args.no_ocr)
201
+
202
+ print("\n[SUMMARY]")
203
+ print(json.dumps(info, ensure_ascii=False, indent=2))
204
+
205
+
206
+ if __name__ == "__main__":
207
+ main()
src/tools/run_pipeline.py ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # src/tools/run_pipeline.py
2
+ # -*- coding: utf-8 -*-
3
+ import subprocess
4
+ from pathlib import Path
5
+
6
+ # 项目根目录(包含 src/)
7
+ BASE_DIR = Path(__file__).resolve().parents[2]
8
+
9
+
10
+ def run_cmd(cmd: list):
11
+ """在项目根目录下执行一个子命令,并在失败时直接抛出异常。"""
12
+ print("🚀 Running:", " ".join(cmd))
13
+ r = subprocess.run(cmd, cwd=BASE_DIR)
14
+ if r.returncode != 0:
15
+ raise RuntimeError(f"Command failed: {' '.join(cmd)}")
16
+
17
+
18
+ def run_full_pipeline():
19
+ """
20
+ 一键跑完整流程(自动提案模式):
21
+
22
+ 1) prepare_proposal_text.py
23
+ 2) extract_facts_by_chunk.py
24
+ 3) build_dimensions_from_facts.py
25
+ 3.5) profiling/domain_profiler.py
26
+ 4) generate_questions.py
27
+ 5) llm_answering.py
28
+ 6) post_processing.py
29
+ 7) ai_expert_opinion.py
30
+ 8) generate_final_report.py
31
+
32
+ 前提:每个脚本内部都实现了“自动检测最新 proposal”的逻辑,
33
+ 即在不传 --file / --proposal_id / --pid 的情况下能自己找到最新项目。
34
+ """
35
+
36
+ # 1) 准备文本
37
+ run_cmd(["python", "-m", "src.tools.prepare_proposal_text"])
38
+
39
+ # 2) facts 抽取
40
+ run_cmd(["python", "-m", "src.tools.extract_facts_by_chunk"])
41
+
42
+ # 3) 由 facts 构建维度
43
+ run_cmd(["python", "-m", "src.tools.build_dimensions_from_facts"])
44
+
45
+ # 3.5) 域画像
46
+ run_cmd(["python", "-m", "src.profiling.domain_profiler"])
47
+
48
+ # 4) 生成问题
49
+ run_cmd(["python", "-m", "src.tools.generate_questions"])
50
+
51
+ # 5) LLM 回答
52
+ run_cmd(["python", "-m", "src.tools.llm_answering"])
53
+
54
+ # 6) post-processing
55
+ run_cmd(["python", "-m", "src.tools.post_processing"])
56
+
57
+ # 7) AI 专家意见
58
+ run_cmd(["python", "-m", "src.tools.ai_expert_opinion"])
59
+
60
+ # 8) 最终报告
61
+ run_cmd(["python", "-m", "src.tools.generate_final_report"])
62
+
63
+ print("🎯 Full pipeline finished.")
64
+
65
+
66
+ if __name__ == "__main__":
67
+ # 不再需要任何命令行参数,直接按照“最新提案”自动跑一遍
68
+ run_full_pipeline()
src/tools/search_by_dimension.py ADDED
@@ -0,0 +1,486 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """
3
+ 阶段 2:智能语义检索器(v2025.12 ProClean · strong-query / early-stop / robust-io / diagnostics)
4
+ 保持下游兼容:evidence/* 命名/结构不变;新增 *_queries.json(维度投放查询记录)
5
+ """
6
+ import os, sys, re, json, time, argparse
7
+ from pathlib import Path
8
+ from urllib.parse import urlparse
9
+ from collections import Counter, defaultdict
10
+ from dotenv import load_dotenv
11
+
12
+ load_dotenv()
13
+ CURRENT_DIR = Path(__file__).resolve().parent
14
+ SRC_ROOT = CURRENT_DIR.parent
15
+ DATA_DIR = SRC_ROOT / "data"
16
+ PROPOSAL_DIR = DATA_DIR / "extracted"
17
+ EVIDENCE_ROOT = DATA_DIR / "evidence"
18
+ CONFIG_DIR = DATA_DIR / "config"
19
+ PARSED_DIR = DATA_DIR / "parsed"
20
+ EVIDENCE_ROOT.mkdir(parents=True, exist_ok=True)
21
+ if str(SRC_ROOT) not in sys.path: sys.path.insert(0, str(SRC_ROOT))
22
+
23
+ from backend.utils.model_selector import get_llm_client
24
+ from backend.retrievers.web_search import simple_search
25
+
26
+ llm = get_llm_client()
27
+ client = llm["client"]; model_name = llm["model_name"]; provider = llm["provider"]
28
+ print(f"💬 QueryGen 使用 {provider.upper()} 模型:{model_name}")
29
+
30
+ # ===== 基本参数 =====
31
+ IGNORE_DIMS = {"proposal_id", "generated_time", "chunk_count", "coverage_estimate", "meta", "doc_meta", "run_meta"}
32
+ FIRST_ROUND_N = 4
33
+ MAX_RESULTS_PER_QUERY = 5
34
+ MAX_QUERIES_PER_DIM = 14
35
+ SLEEP_BETWEEN_QUERIES = 0.8
36
+ MIN_SAVE_EVIDENCE = 1
37
+
38
+ # 每题“学术域命中”提前停止阈值(维度自定义)
39
+ EARLY_STOP_ACADEMIC = {"strategy":4, "objectives":4, "feasibility":4, "innovation":3, "team":3}
40
+ # 维度层面的软阈值(累计达到后对后续题适度保守)
41
+ DIM_SOFT_EARLY_STOP = {"strategy":10, "objectives":10, "feasibility":10, "innovation":8, "team":8}
42
+
43
+ ACADEMIC_SITES = [
44
+ "arxiv.org","openreview.net","aclweb.org","ieeexplore.ieee.org","dl.acm.org",
45
+ "springer.com","nature.com","sciencedirect.com","wiley.com","tandfonline.com",
46
+ "mdpi.com","frontiersin.org","osf.io","zenodo.org","doi.org"
47
+ ]
48
+
49
+ DIM_HINTS = {
50
+ "team": ["team expertise", "roles and responsibilities", "governance", "affiliation"],
51
+ "objectives": ["problem statement", "scope", "deliverable", "evaluation metric"],
52
+ "strategy": ["method", "technical approach", "implementation pathway", "workflow"],
53
+ "innovation": ["novelty", "differentiation", "prior work", "evidence"],
54
+ "feasibility": ["resources", "timeline", "risk", "dependency", "budget"]
55
+ }
56
+
57
+ def clean_query(q: str, max_len=280):
58
+ q = re.sub(r"[,。?:;!、]", " ", q or "")
59
+ q = re.sub(r"\s+", " ", q.strip())
60
+ return q[:max_len]
61
+
62
+ def uniq(seq):
63
+ seen, out = set(), []
64
+ for x in seq:
65
+ sx = str(x or "").strip()
66
+ if not sx: continue
67
+ lx = sx.lower()
68
+ if lx not in seen:
69
+ out.append(sx); seen.add(lx)
70
+ return out
71
+
72
+ def _extract_bracket_block(t: str, lch: str, rch: str):
73
+ s, e = t.find(lch), t.rfind(rch)
74
+ if s != -1 and e != -1 and e > s:
75
+ frag = t[s:e+1]
76
+ return frag.replace("“","\"").replace("”","\"")
77
+ return None
78
+
79
+ def safe_json_loads(text: str):
80
+ t = (text or "").strip()
81
+ if "```" in t: t = t.replace("```json","").replace("```","").strip()
82
+ for frag in (t, _extract_bracket_block(t,"[","]"), _extract_bracket_block(t,"{","}")):
83
+ if not frag: continue
84
+ try:
85
+ obj = json.loads(frag)
86
+ if isinstance(obj, list): return obj
87
+ if isinstance(obj, dict) and isinstance(obj.get("questions"), list): return obj["questions"]
88
+ except Exception: continue
89
+ return []
90
+
91
+ def collect_entities_numbers_terms(dim_content: dict):
92
+ ents, nums, key_terms = [], [], []
93
+ try:
94
+ people = dim_content.get("entities", {}).get("people", []) or []
95
+ ents += [p.get("name","") for p in people if isinstance(p, dict)]
96
+ ents += dim_content.get("entities", {}).get("orgs", []) or []
97
+ except Exception: pass
98
+ ents = [e for e in ents if isinstance(e,str) and e.strip()]
99
+
100
+ try:
101
+ nums = [str(n.get("value","")) for n in (dim_content.get("numbers") or [])
102
+ if isinstance(n, dict) and n.get("value")]
103
+ except Exception: pass
104
+ nums = [n for n in nums if n]
105
+
106
+ try:
107
+ kt = dim_content.get("key_terms") or []
108
+ if isinstance(kt, list): key_terms = [str(k) for k in kt if k]
109
+ except Exception: pass
110
+ return ents[:8], nums[:6], key_terms[:12]
111
+
112
+ # ---- 回退模板(保留原有)----
113
+ FALLBACK_TEMPLATES = {
114
+ "team": [
115
+ '"{PERSON}" profile role affiliation 2019..2026',
116
+ '"{ORG}" team lab group leadership 2019..2026',
117
+ '"{PERSON}" biography experience project 2019..2026'
118
+ ],
119
+ "objectives": [
120
+ '"{KEY}" objective scope deliverable 2020..2026',
121
+ '"{KEY}" evaluation metric milestone 2020..2026',
122
+ '"{KEY}" problem statement requirement 2020..2026'
123
+ ],
124
+ "strategy": [
125
+ '"{KEY}" method workflow implementation 2020..2026',
126
+ '"{KEY}" technical approach validation 2020..2026',
127
+ '"{KEY}" architecture design process 2020..2026'
128
+ ],
129
+ "innovation": [
130
+ '"{KEY}" novelty differentiation prior work 2020..2026',
131
+ '"{KEY}" patent benchmark evidence 2020..2026',
132
+ '"{KEY}" state of the art comparison 2020..2026'
133
+ ],
134
+ "feasibility": [
135
+ '"{KEY}" resource timeline risk dependency 2020..2026',
136
+ '"{KEY}" budget implementation constraint 2020..2026',
137
+ '"{KEY}" evaluation challenge mitigation 2020..2026'
138
+ ]
139
+ }
140
+
141
+ def _inject_fallbacks(dimension: str, entities: list, keywords: list):
142
+ toks = uniq((entities or []) + (keywords or []))
143
+ outs = []
144
+ for tpl in FALLBACK_TEMPLATES.get(dimension.lower(), []):
145
+ if "{PERSON}" in tpl and entities:
146
+ outs.append(tpl.replace("{PERSON}", entities[0]))
147
+ elif "{ORG}" in tpl and len(entities) > 1:
148
+ outs.append(tpl.replace("{ORG}", entities[1]))
149
+ elif "{KEY}" in tpl and toks:
150
+ outs.append(tpl.replace("{KEY}", toks[0]))
151
+ return uniq(outs)
152
+
153
+ # ---- 新增:读取并展开 generated_questions.json 的 query_templates ----
154
+ def load_query_templates():
155
+ qset_path = CONFIG_DIR / "question_sets" / "generated_questions.json"
156
+ try:
157
+ raw = json.loads(qset_path.read_text(encoding="utf-8"))
158
+ return raw.get("query_templates", {}) or {}
159
+ except Exception:
160
+ return {}
161
+
162
+ def expand_templates_for_dim(dimension: str, templates: list, entities: list, key_terms: list, numbers: list, time_hint: str):
163
+ people = [e for e in entities if e]
164
+ orgs = [e for e in entities if e]
165
+ terms = [t for t in key_terms if t]
166
+ nums = [n for n in numbers if n]
167
+
168
+ p_opts = people[:2] or [""]
169
+ o_opts = orgs[:2] or [""]
170
+ t_opts = terms[:3] or [""]
171
+ n_opts = nums[:2] or [""]
172
+
173
+ out = []
174
+ for tpl in templates or []:
175
+ tpl = str(tpl or "")
176
+ for p in p_opts:
177
+ for o in o_opts:
178
+ for t in t_opts:
179
+ for n in n_opts:
180
+ q = tpl.replace("{PERSON}", p).replace("{ORG}", o).replace("{TERM}", t).replace("{NUM}", n)
181
+ q = q.replace(" ", " ").strip()
182
+ if time_hint and "20" in time_hint and time_hint not in q:
183
+ q = f'{q} {time_hint}'
184
+ out.append(clean_query(q))
185
+ return uniq([q for q in out if q])[:MAX_QUERIES_PER_DIM]
186
+
187
+ def llm_generate_queries(question, context, dimension, hints=None, entities=None, numbers=None, key_terms=None):
188
+ time_hint = "2019..2025" if dimension in ("strategy","objectives","feasibility") else "2021..2025"
189
+ dim_hints = "; ".join(DIM_HINTS.get(dimension.lower(), []))
190
+ hint_text = "; ".join(hints or [])
191
+ ent_text = "; ".join(entities or [])[:240]
192
+ num_text = ", ".join(numbers or [])[:80]
193
+ key_text = "; ".join(key_terms or [])[:240]
194
+
195
+ must_have = uniq((key_terms or []) + (hints or []))[:6]
196
+ mh_text = "; ".join(must_have) if must_have else "core domain terms from the proposal"
197
+
198
+ prompt = f"""
199
+ 你是通用项目检索专家。针对“问题+维度+摘要+提示词+实体+数字”,生成 {FIRST_ROUND_N} 条高质量检索 query。
200
+ 要求:
201
+ - 每条尽量包含:实体/机构/模型名/登记号/关键数字(若有)
202
+ - 使用 site: 与时间窗({time_hint}),中英均可,简洁可直投 Google
203
+ - 可优先结合以下高价值术语(如适用):{mh_text}
204
+ - 输出严格的 JSON 数组(字符串列表),仅内容,无解释
205
+ 问题:{question}
206
+ 维度:{dimension}
207
+ 摘要:{context[:900]}
208
+ hints:{hint_text};{dim_hints}
209
+ 实体:{ent_text}
210
+ 数字:{num_text}
211
+ 关键词:{key_text}
212
+ """
213
+ try:
214
+ rsp = client.chat.completions.create(
215
+ model=model_name, messages=[{"role":"user","content":prompt}], temperature=0.35
216
+ )
217
+ content = rsp.choices[0].message.content.strip()
218
+ queries = safe_json_loads(content)
219
+ queries = [clean_query(q["query"] if isinstance(q, dict) and "query" in q else str(q)) for q in queries]
220
+ zh_variants = [f"{q} 项目 研究 方法 风险 评估 {time_hint}" for q in queries]
221
+ merged = uniq(queries + zh_variants)
222
+ return merged[:MAX_QUERIES_PER_DIM] if merged else [question]
223
+ except Exception as e:
224
+ print(f"⚠️ LLM 生成 Query 失败: {e}")
225
+ return [question]
226
+
227
+ # ===== 新增:基础 must/should 子句拼接 =====
228
+ def build_base_clause(dim_name: str, qcfg: dict, qsets_meta: dict):
229
+ doc_policy = (qsets_meta.get("doc_policy") or {})
230
+ must_terms = list(dict.fromkeys((qcfg.get("search", {}).get("must_terms") or []) + (doc_policy.get("must_terms") or [])))
231
+ should_terms = list(dict.fromkeys((qcfg.get("search", {}).get("should_terms") or []) + (doc_policy.get("should_terms") or [])))
232
+
233
+ def qwrap(t):
234
+ t = str(t).strip()
235
+ if not t: return ""
236
+ return f"({t})" if " " in t and not (t.startswith('"') and t.endswith('"')) else t
237
+
238
+ must_clause = " ".join(qwrap(t) for t in must_terms if t)
239
+ should_clause = ""
240
+ if should_terms:
241
+ should_clause = " (" + " OR ".join(qwrap(t) for t in should_terms if t) + ")"
242
+ base = (must_clause + should_clause).strip()
243
+ return base, must_terms, should_terms
244
+
245
+ # ============ 主流程 ============
246
+
247
+ parser = argparse.ArgumentParser()
248
+ parser.add_argument("--fast", action="store_true", help="仅检索每维前2个问题(调试模式)")
249
+ args = parser.parse_args()
250
+
251
+ # 1) 读取清洗后的维度(固定路径)
252
+ parsed_path = PARSED_DIR / "parsed_dimensions.clean.llm.json"
253
+ if not parsed_path.exists():
254
+ print(f"❌ 未找到清洗后的维度文件:{parsed_path};请先运行 strict_cleanup_llm.py")
255
+ sys.exit(1)
256
+
257
+ try:
258
+ dimensions = json.loads(parsed_path.read_text(encoding="utf-8"))
259
+ except Exception as e:
260
+ print(f"❌ 读取维度文件失败:{e}"); sys.exit(1)
261
+
262
+ # 2) 读取问题集(固定路径)
263
+ qset_path = CONFIG_DIR / "question_sets" / "generated_questions.json"
264
+ if not qset_path.exists():
265
+ print(f"❌ 未找到问题集:{qset_path}"); sys.exit(1)
266
+ try:
267
+ question_sets = json.loads(qset_path.read_text(encoding="utf-8"))
268
+ except Exception as e:
269
+ print(f"❌ 读取问题集失败:{e}"); sys.exit(1)
270
+
271
+ # 3) 从 run_meta.source_path 恢复 proposal_id
272
+ proposal_id = "current_proposal"
273
+ try:
274
+ src_path = (dimensions.get("run_meta") or {}).get("source_path", "")
275
+ if src_path:
276
+ p = Path(src_path)
277
+ if p.name.endswith("_dimensions.json"):
278
+ proposal_id = p.stem.replace("_dimensions", "")
279
+ except Exception:
280
+ pass
281
+ os.environ["CURRENT_PROPOSAL_ID"] = proposal_id
282
+ print(f"📂 当前提案文件: {proposal_id}")
283
+
284
+ EVIDENCE_DIR = EVIDENCE_ROOT / proposal_id
285
+ EVIDENCE_DIR.mkdir(parents=True, exist_ok=True)
286
+
287
+ # 读取顶层 query_templates(问题集里)
288
+ QUERY_TEMPLATES_ALL = question_sets.get("query_templates", {}) or {}
289
+
290
+ global_domain_counter, stats = Counter(), {}
291
+ debug_overview = {}
292
+
293
+ # 供基础子句使用的 meta
294
+ qsets_meta = question_sets.get("meta", {}) or {}
295
+
296
+ for dim, ctx in dimensions.items():
297
+ if dim in IGNORE_DIMS or dim not in question_sets:
298
+ continue
299
+
300
+ print(f"\n🔹 开始维度: {dim}")
301
+ t0 = time.time()
302
+ qcfg = question_sets[dim]
303
+ questions = qcfg.get("questions", []) or []
304
+ if not questions:
305
+ print("ℹ️ 该维度无问题,跳过。"); continue
306
+ if args.fast:
307
+ questions = questions[:2]; print("⚙️ Fast:仅检索前2个问题")
308
+
309
+ # 收集维度上下文信息
310
+ ents, nums, key_terms = collect_entities_numbers_terms(ctx or {})
311
+ dim_evidences = []
312
+ per_q_domain_counter = Counter()
313
+ success, fail = 0, 0
314
+ dim_academic_hits = 0
315
+
316
+ all_queries_fired = []
317
+ needed_hits = EARLY_STOP_ACADEMIC.get(dim, 3)
318
+ dim_soft_cap = DIM_SOFT_EARLY_STOP.get(dim, 8)
319
+
320
+ # 维度模板:来自 generated_questions.json 的 query_templates 对应维度
321
+ dim_templates = (QUERY_TEMPLATES_ALL.get(dim, []) or [])[:10]
322
+ templates_before = len(dim_templates)
323
+
324
+ # ===== 新增:构造基础 must/should 子句 & 合并 search_hints =====
325
+ base_clause, must_terms_used, should_terms_used = build_base_clause(dim, qcfg, qsets_meta)
326
+ merged_hints = list(dict.fromkeys((qcfg.get("search_hints") or []) + (qsets_meta.get("doc_policy", {}) or {}).get("query_hints_merged", [])))[:10]
327
+
328
+ for q in questions:
329
+ if dim_academic_hits >= dim_soft_cap:
330
+ print(f"🛑 维度累计学术命中已达软阈值 {dim_soft_cap},对后续题缩减投放。")
331
+ max_per_question = 2
332
+ else:
333
+ max_per_question = MAX_QUERIES_PER_DIM
334
+
335
+ print(f"\n🧭 问题: {q}")
336
+ # 1) LLM 生成首轮强 query
337
+ llm_queries = llm_generate_queries(
338
+ q, (ctx or {}).get("summary",""), dim,
339
+ hints=merged_hints,
340
+ entities=ents, numbers=nums, key_terms=key_terms
341
+ )
342
+
343
+ # 2) 维度模板占位展开(自动带入时间窗)
344
+ time_hint = "2019..2025" if dim in ("strategy","objectives","feasibility") else "2021..2025"
345
+ expanded_tpl = expand_templates_for_dim(dim, dim_templates, ents, key_terms, nums, time_hint)
346
+
347
+ # 3) 回退模板
348
+ fallbacks = _inject_fallbacks(dim, ents, [q] + DIM_HINTS.get(dim, []))
349
+
350
+ # 4) 基于 hints 的单独查询(每个 hint 一条)
351
+ hint_queries = []
352
+ for h in merged_hints:
353
+ if not h: continue
354
+ hq = h
355
+ if base_clause:
356
+ hq = f"{base_clause} {h}".strip()
357
+ hint_queries.append(clean_query(hq))
358
+
359
+ # 合并顺序:原问题 -> LLM -> 模板 -> 回退 -> hint 单发
360
+ merged_queries = uniq([q] + llm_queries + expanded_tpl + fallbacks + hint_queries)
361
+
362
+ # 在每条 query 前拼接基础 must/should 子句
363
+ if base_clause:
364
+ merged_queries = [clean_query(f"{base_clause} {qq}") for qq in merged_queries]
365
+
366
+ # 控制每题投放上限
367
+ merged_queries = merged_queries[:max_per_question]
368
+
369
+ # ---- 查询执行 ----
370
+ per_question_academic_hits = 0
371
+ empty_hits = 0
372
+ for i, query in enumerate(merged_queries, start=1):
373
+ print(f"🔍 ({i}/{len(merged_queries)}) 搜索: {query}")
374
+ all_queries_fired.append(query)
375
+ try:
376
+ texts, urls = simple_search(
377
+ query, max_results=MAX_RESULTS_PER_QUERY,
378
+ dimension=dim, hints=merged_hints, source="LLM"
379
+ )
380
+ got = 0
381
+ for t, u in zip(texts, urls):
382
+ dim_evidences.append({"query": query, "text": t, "url": u})
383
+ host = urlparse(u).hostname or ""
384
+ if host:
385
+ per_q_domain_counter[host] += 1
386
+ global_domain_counter[host] += 1
387
+ if any(ad in host for ad in ACADEMIC_SITES):
388
+ per_question_academic_hits += 1
389
+ dim_academic_hits += 1
390
+ got += 1
391
+ if got == 0: empty_hits += 1
392
+ success += 1 if got > 0 else 0
393
+ fail += 1 if got == 0 else 0
394
+ except Exception as e:
395
+ print(f"❌ 搜索失败: {e}"); fail += 1; empty_hits += 1
396
+
397
+ if per_question_academic_hits >= needed_hits:
398
+ print("🛑 该题学术来源已足够,提前停止扩张。"); break
399
+ time.sleep(SLEEP_BETWEEN_QUERIES)
400
+
401
+ # 空击回退:若前面全空,尝试“hint/实体 词袋 + 时间窗 + site”的布尔兜底
402
+ if per_question_academic_hits == 0 and empty_hits >= len(merged_queries):
403
+ bag = uniq((merged_hints or []) + ents + key_terms)
404
+ bag = [b for b in bag if len(b) >= 2][:6]
405
+ if bag:
406
+ bag_q = " ".join(f'"{b}"' for b in bag)
407
+ bag_q = f'{bag_q} {time_hint}'
408
+ bag_q = clean_query(bag_q)
409
+ print(f"🧯 兜底搜索:{bag_q}")
410
+ all_queries_fired.append(bag_q)
411
+ try:
412
+ texts, urls = simple_search(
413
+ bag_q, max_results=MAX_RESULTS_PER_QUERY,
414
+ dimension=dim, hints=merged_hints, source="LLM"
415
+ )
416
+ for t, u in zip(texts, urls):
417
+ dim_evidences.append({"query": bag_q, "text": t, "url": u})
418
+ host = urlparse(u).hostname or ""
419
+ if host:
420
+ per_q_domain_counter[host] += 1
421
+ global_domain_counter[host] += 1
422
+ if any(ad in host for ad in ACADEMIC_SITES):
423
+ per_question_academic_hits += 1
424
+ dim_academic_hits += 1
425
+ success += 1 if texts else 0
426
+ fail += 1 if not texts else 0
427
+ except Exception as e:
428
+ print(f"❌ 兜底失败: {e}")
429
+
430
+ # 记录维度 query(诊断信息更丰富)
431
+ queries_record = {
432
+ "dimension": dim,
433
+ "templates_loaded": templates_before,
434
+ "templates_expanded": len(expanded_tpl) if 'expanded_tpl' in locals() else 0,
435
+ "queries_total": len(all_queries_fired),
436
+ "queries": all_queries_fired
437
+ }
438
+ (EVIDENCE_DIR / f"{dim}_queries.json").write_text(
439
+ json.dumps(queries_record, ensure_ascii=False, indent=2),
440
+ encoding="utf-8"
441
+ )
442
+
443
+ if len(dim_evidences) < MIN_SAVE_EVIDENCE:
444
+ print(f"⚠️ {dim} 维度 evidence 过少({len(dim_evidences)}),跳过保存。")
445
+ continue
446
+
447
+ raw_ev = EVIDENCE_DIR / f"{dim}_evidence_raw.json"
448
+ try:
449
+ raw_ev.write_text(json.dumps(dim_evidences, ensure_ascii=False, indent=2), encoding="utf-8")
450
+ except Exception as e:
451
+ print(f"⚠️ 写入原始 evidence 失败:{e}")
452
+
453
+ elapsed = round(time.time() - t0, 1)
454
+ success_rate = round(success / (success + fail + 1e-6), 2)
455
+ print(f"📊 {dim} 完成:成功率 {success_rate}, evidence {len(dim_evidences)}, 耗时 {elapsed}s")
456
+
457
+ stats[dim] = {
458
+ "问题数": len(questions), "成功": success, "失败": fail,
459
+ "成功率": success_rate, "Evidence条数": len(dim_evidences),
460
+ "Top域名": per_q_domain_counter.most_common(6), "耗时(s)": elapsed,
461
+ "维度累计学术命中": dim_academic_hits
462
+ }
463
+ debug_overview[dim] = {
464
+ "queries_total": queries_record["queries_total"],
465
+ "evidence_kept": len(dim_evidences),
466
+ "elapsed_s": elapsed,
467
+ "early_stop_threshold_per_question": EARLY_STOP_ACADEMIC.get(dim, 3),
468
+ "early_stop_soft_dim": DIM_SOFT_EARLY_STOP.get(dim, 8)
469
+ }
470
+
471
+ report = {
472
+ "proposal_id": proposal_id, "provider": provider, "model": model_name,
473
+ "timestamp": time.strftime("%Y-%m-%d %H:%M:%S"),
474
+ "stats": stats, "top_domains_global": global_domain_counter.most_common(12)
475
+ }
476
+ summary_path = EVIDENCE_DIR / "dimension_summary_index.json"
477
+ summary_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")
478
+
479
+ (EVIDENCE_DIR / "dimension_debug_overview.json").write_text(
480
+ json.dumps(debug_overview, ensure_ascii=False, indent=2), encoding="utf-8"
481
+ )
482
+
483
+ print("\n✅ 检索完成。报告保存至:", summary_path)
484
+ print("🌐 Top域名:", report["top_domains_global"])
485
+ print(f"📁 输出目录:{EVIDENCE_DIR}")
486
+ print(f"📈 生成的 combined 文件:{len(list(EVIDENCE_DIR.glob('*_combined.json')))} 个")
src/tools/validate_stage0_outputs.py ADDED
@@ -0,0 +1,47 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # -*- coding: utf-8 -*-
2
+ """Validate multimodal Stage 0 artifacts for a prepared proposal."""
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ from pathlib import Path
8
+
9
+ BASE_DIR = Path(__file__).resolve().parents[2]
10
+ PREPARED_DIR = BASE_DIR / "src" / "data" / "prepared"
11
+
12
+
13
+ def validate(proposal_id: str) -> int:
14
+ out_dir = PREPARED_DIR / proposal_id
15
+ issues = []
16
+ page_sem_path = out_dir / "page_semantics.json"
17
+ if not page_sem_path.exists():
18
+ issues.append("missing page_semantics.json")
19
+ print("[FAIL] missing page_semantics.json")
20
+ return 1
21
+ data = json.loads(page_sem_path.read_text(encoding="utf-8"))
22
+ pages = data.get("pages") or []
23
+ if data.get("num_pages") != len(pages):
24
+ issues.append("num_pages mismatch")
25
+ for page in pages:
26
+ if not page.get("reconstructed_text"):
27
+ issues.append(f"page {page.get('page_index')} missing reconstructed_text")
28
+ if not page.get("blocks"):
29
+ issues.append(f"page {page.get('page_index')} missing blocks")
30
+ if issues:
31
+ print("[WARN] Stage 0 validation issues:")
32
+ for issue in issues:
33
+ print(" -", issue)
34
+ return 1
35
+ print(f"[OK] Stage 0 artifacts look valid for proposal_id={proposal_id}")
36
+ return 0
37
+
38
+
39
+ def main() -> None:
40
+ parser = argparse.ArgumentParser()
41
+ parser.add_argument("proposal_id")
42
+ args = parser.parse_args()
43
+ raise SystemExit(validate(args.proposal_id))
44
+
45
+
46
+ if __name__ == "__main__":
47
+ main()