author | Sylvain Thénault <sylvain.thenault@logilab.fr> |
Fri, 05 Jun 2009 19:04:20 +0200 | |
changeset 2059 | af33833d7571 |
parent 1977 | 606923dff11b |
child 2248 | cbf043a2134a |
permissions | -rw-r--r-- |
0 | 1 |
"""Check integrity of a CubicWeb repository. Hum actually only the system database |
2 |
is checked. |
|
3 |
||
4 |
:organization: Logilab |
|
1977
606923dff11b
big bunch of copyright / docstring update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1802
diff
changeset
|
5 |
:copyright: 2001-2009 LOGILAB S.A. (Paris, FRANCE), license is LGPL v2. |
0 | 6 |
:contact: http://www.logilab.fr/ -- mailto:contact@logilab.fr |
1977
606923dff11b
big bunch of copyright / docstring update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1802
diff
changeset
|
7 |
:license: GNU Lesser General Public License, v2.1 - http://www.gnu.org/licenses |
0 | 8 |
""" |
9 |
__docformat__ = "restructuredtext en" |
|
10 |
||
11 |
import sys |
|
1016
26387b836099
use datetime instead of mx.DateTime
sylvain.thenault@logilab.fr
parents:
713
diff
changeset
|
12 |
from datetime import datetime |
0 | 13 |
|
14 |
from logilab.common.shellutils import ProgressBar |
|
15 |
||
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
16 |
from cubicweb.server.sqlutils import SQL_PREFIX |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
17 |
|
0 | 18 |
def has_eid(sqlcursor, eid, eids): |
19 |
"""return true if the eid is a valid eid""" |
|
20 |
if eids.has_key(eid): |
|
21 |
return eids[eid] |
|
22 |
sqlcursor.execute('SELECT type, source FROM entities WHERE eid=%s' % eid) |
|
23 |
try: |
|
24 |
etype, source = sqlcursor.fetchone() |
|
25 |
except: |
|
26 |
eids[eid] = False |
|
27 |
return False |
|
28 |
if source and source != 'system': |
|
29 |
# XXX what to do... |
|
30 |
eids[eid] = True |
|
31 |
return True |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
32 |
sqlcursor.execute('SELECT * FROM %s%s WHERE %seid=%s' % (SQL_PREFIX, etype, |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
33 |
SQL_PREFIX, eid)) |
0 | 34 |
result = sqlcursor.fetchall() |
35 |
if len(result) == 0: |
|
36 |
eids[eid] = False |
|
37 |
return False |
|
38 |
elif len(result) > 1: |
|
39 |
msg = ' More than one entity with eid %s exists in source !' |
|
40 |
print >> sys.stderr, msg % eid |
|
41 |
print >> sys.stderr, ' WARNING : Unable to fix this, do it yourself !' |
|
42 |
eids[eid] = True |
|
43 |
return True |
|
44 |
||
45 |
# XXX move to yams? |
|
46 |
def etype_fti_containers(eschema, _done=None): |
|
47 |
if _done is None: |
|
48 |
_done = set() |
|
49 |
_done.add(eschema) |
|
50 |
containers = tuple(eschema.fulltext_containers()) |
|
51 |
if containers: |
|
52 |
for rschema, target in containers: |
|
53 |
if target == 'object': |
|
54 |
targets = rschema.objects(eschema) |
|
55 |
else: |
|
56 |
targets = rschema.subjects(eschema) |
|
57 |
for targeteschema in targets: |
|
58 |
if targeteschema in _done: |
|
59 |
continue |
|
60 |
_done.add(targeteschema) |
|
61 |
for container in etype_fti_containers(targeteschema, _done): |
|
62 |
yield container |
|
63 |
else: |
|
64 |
yield eschema |
|
1802
d628defebc17
delete-trailing-whitespace + some copyright update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1398
diff
changeset
|
65 |
|
0 | 66 |
def reindex_entities(schema, session): |
67 |
"""reindex all entities in the repository""" |
|
68 |
# deactivate modification_date hook since we don't want them |
|
69 |
# to be updated due to the reindexation |
|
70 |
from cubicweb.server.hooks import (setmtime_before_update_entity, |
|
71 |
uniquecstrcheck_before_modification) |
|
72 |
from cubicweb.server.repository import FTIndexEntityOp |
|
73 |
repo = session.repo |
|
74 |
repo.hm.unregister_hook(setmtime_before_update_entity, |
|
75 |
'before_update_entity', '') |
|
76 |
repo.hm.unregister_hook(uniquecstrcheck_before_modification, |
|
77 |
'before_update_entity', '') |
|
1161
936c311010fc
ensure do_fti is true in reindex_entities
sylvain.thenault@logilab.fr
parents:
381
diff
changeset
|
78 |
repo.do_fti = True # ensure full-text indexation is activated |
0 | 79 |
etypes = set() |
80 |
for eschema in schema.entities(): |
|
81 |
if eschema.is_final(): |
|
82 |
continue |
|
83 |
indexable_attrs = tuple(eschema.indexable_attributes()) # generator |
|
84 |
if not indexable_attrs: |
|
85 |
continue |
|
86 |
for container in etype_fti_containers(eschema): |
|
87 |
etypes.add(container) |
|
88 |
print 'Reindexing entities of type %s' % \ |
|
89 |
', '.join(sorted(str(e) for e in etypes)) |
|
90 |
pb = ProgressBar(len(etypes) + 1) |
|
91 |
# first monkey patch Entity.check to disable validation |
|
713
5adb6d8e5fa7
update imports of "cubicweb.common.entity" and use the new module path "cubicweb.entity"
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
381
diff
changeset
|
92 |
from cubicweb.entity import Entity |
0 | 93 |
_check = Entity.check |
94 |
Entity.check = lambda self, creation=False: True |
|
95 |
# clear fti table first |
|
96 |
session.system_sql('DELETE FROM %s' % session.repo.system_source.dbhelper.fti_table) |
|
97 |
pb.update() |
|
98 |
# reindex entities by generating rql queries which set all indexable |
|
99 |
# attribute to their current value |
|
100 |
for eschema in etypes: |
|
101 |
for entity in session.execute('Any X WHERE X is %s' % eschema).entities(): |
|
102 |
FTIndexEntityOp(session, entity=entity) |
|
103 |
pb.update() |
|
104 |
# restore Entity.check |
|
105 |
Entity.check = _check |
|
106 |
||
1802
d628defebc17
delete-trailing-whitespace + some copyright update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1398
diff
changeset
|
107 |
|
380 | 108 |
def check_schema(schema, session, eids, fix=1): |
0 | 109 |
"""check serialized schema""" |
110 |
print 'Checking serialized schema' |
|
111 |
unique_constraints = ('SizeConstraint', 'FormatConstraint', |
|
112 |
'VocabularyConstraint', 'RQLConstraint', |
|
113 |
'RQLVocabularyConstraint') |
|
114 |
rql = ('Any COUNT(X),RN,EN,ECTN GROUPBY RN,EN,ECTN ORDERBY 1 ' |
|
1398
5fe84a5f7035
rename internal entity types to have CW prefix instead of E
sylvain.thenault@logilab.fr
parents:
1263
diff
changeset
|
115 |
'WHERE X is CWConstraint, R constrained_by X, ' |
0 | 116 |
'R relation_type RT, R from_entity ET, RT name RN, ' |
117 |
'ET name EN, X cstrtype ECT, ECT name ECTN') |
|
118 |
for count, rn, en, cstrname in session.execute(rql): |
|
119 |
if count == 1: |
|
120 |
continue |
|
121 |
if cstrname in unique_constraints: |
|
122 |
print "ERROR: got %s %r constraints on relation %s.%s" % ( |
|
123 |
count, cstrname, en, rn) |
|
124 |
||
125 |
||
1802
d628defebc17
delete-trailing-whitespace + some copyright update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1398
diff
changeset
|
126 |
|
0 | 127 |
def check_text_index(schema, session, eids, fix=1): |
128 |
"""check all entities registered in the text index""" |
|
129 |
print 'Checking text index' |
|
130 |
cursor = session.system_sql('SELECT uid FROM appears;') |
|
131 |
for row in cursor.fetchall(): |
|
132 |
eid = row[0] |
|
133 |
if not has_eid(cursor, eid, eids): |
|
134 |
msg = ' Entity with eid %s exists in the text index but in no source' |
|
135 |
print >> sys.stderr, msg % eid, |
|
136 |
if fix: |
|
137 |
session.system_sql('DELETE FROM appears WHERE uid=%s;' % eid) |
|
138 |
print >> sys.stderr, ' [FIXED]' |
|
139 |
else: |
|
140 |
print >> sys.stderr |
|
141 |
||
142 |
||
143 |
def check_entities(schema, session, eids, fix=1): |
|
144 |
"""check all entities registered in the repo system table""" |
|
145 |
print 'Checking entities system table' |
|
146 |
cursor = session.system_sql('SELECT eid FROM entities;') |
|
147 |
for row in cursor.fetchall(): |
|
148 |
eid = row[0] |
|
149 |
if not has_eid(cursor, eid, eids): |
|
150 |
msg = ' Entity with eid %s exists in the system table but in no source' |
|
151 |
print >> sys.stderr, msg % eid, |
|
152 |
if fix: |
|
153 |
session.system_sql('DELETE FROM entities WHERE eid=%s;' % eid) |
|
154 |
print >> sys.stderr, ' [FIXED]' |
|
155 |
else: |
|
156 |
print >> sys.stderr |
|
157 |
print 'Checking entities tables' |
|
158 |
for eschema in schema.entities(): |
|
159 |
if eschema.is_final(): |
|
160 |
continue |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
161 |
table = SQL_PREFIX + eschema.type |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
162 |
column = SQL_PREFIX + 'eid' |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
163 |
cursor = session.system_sql('SELECT %s FROM %s;' % (column, table)) |
0 | 164 |
for row in cursor.fetchall(): |
165 |
eid = row[0] |
|
166 |
# eids is full since we have fetched everyting from the entities table, |
|
167 |
# no need to call has_eid |
|
168 |
if not eid in eids or not eids[eid]: |
|
169 |
msg = ' Entity with eid %s exists in the %s table but not in the system table' |
|
170 |
print >> sys.stderr, msg % (eid, eschema.type), |
|
171 |
if fix: |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
172 |
session.system_sql('DELETE FROM %s WHERE %s=%s;' % (table, column, eid)) |
0 | 173 |
print >> sys.stderr, ' [FIXED]' |
174 |
else: |
|
175 |
print >> sys.stderr |
|
1802
d628defebc17
delete-trailing-whitespace + some copyright update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1398
diff
changeset
|
176 |
|
d628defebc17
delete-trailing-whitespace + some copyright update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1398
diff
changeset
|
177 |
|
0 | 178 |
def bad_related_msg(rtype, target, eid, fix): |
179 |
msg = ' A relation %s with %s eid %s exists but no such entity in sources' |
|
180 |
print >> sys.stderr, msg % (rtype, target, eid), |
|
181 |
if fix: |
|
182 |
print >> sys.stderr, ' [FIXED]' |
|
183 |
else: |
|
184 |
print >> sys.stderr |
|
1802
d628defebc17
delete-trailing-whitespace + some copyright update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1398
diff
changeset
|
185 |
|
d628defebc17
delete-trailing-whitespace + some copyright update
Adrien Di Mascio <Adrien.DiMascio@logilab.fr>
parents:
1398
diff
changeset
|
186 |
|
0 | 187 |
def check_relations(schema, session, eids, fix=1): |
188 |
"""check all relations registered in the repo system table""" |
|
189 |
print 'Checking relations' |
|
190 |
for rschema in schema.relations(): |
|
191 |
if rschema.is_final(): |
|
192 |
continue |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
193 |
if rschema == 'identity': |
0 | 194 |
continue |
195 |
if rschema.inlined: |
|
196 |
for subjtype in rschema.subjects(): |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
197 |
table = SQL_PREFIX + str(subjtype) |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
198 |
column = SQL_PREFIX + str(rschema) |
380 | 199 |
sql = 'SELECT %s FROM %s WHERE %s IS NOT NULL;' % ( |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
200 |
column, table, column) |
380 | 201 |
cursor = session.system_sql(sql) |
0 | 202 |
for row in cursor.fetchall(): |
203 |
eid = row[0] |
|
204 |
if not has_eid(cursor, eid, eids): |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
205 |
bad_related_msg(rschema, 'object', eid, fix) |
0 | 206 |
if fix: |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
207 |
sql = 'UPDATE %s SET %s = NULL WHERE %seid=%s;' % ( |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
208 |
table, column, SQL_PREFIX, eid) |
381 | 209 |
session.system_sql(sql) |
0 | 210 |
continue |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
211 |
cursor = session.system_sql('SELECT eid_from FROM %s_relation;' % rschema) |
0 | 212 |
for row in cursor.fetchall(): |
213 |
eid = row[0] |
|
214 |
if not has_eid(cursor, eid, eids): |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
215 |
bad_related_msg(rschema, 'subject', eid, fix) |
0 | 216 |
if fix: |
380 | 217 |
sql = 'DELETE FROM %s_relation WHERE eid_from=%s;' % ( |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
218 |
rschema, eid) |
380 | 219 |
session.system_sql(sql) |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
220 |
cursor = session.system_sql('SELECT eid_to FROM %s_relation;' % rschema) |
0 | 221 |
for row in cursor.fetchall(): |
222 |
eid = row[0] |
|
223 |
if not has_eid(cursor, eid, eids): |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
224 |
bad_related_msg(rschema, 'object', eid, fix) |
0 | 225 |
if fix: |
380 | 226 |
sql = 'DELETE FROM %s_relation WHERE eid_to=%s;' % ( |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
227 |
rschema, eid) |
380 | 228 |
session.system_sql(sql) |
0 | 229 |
|
230 |
||
231 |
def check_metadata(schema, session, eids, fix=1): |
|
232 |
"""check entities has required metadata |
|
233 |
||
234 |
FIXME: rewrite using RQL queries ? |
|
235 |
""" |
|
236 |
print 'Checking metadata' |
|
237 |
cursor = session.system_sql("SELECT DISTINCT type FROM entities;") |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
238 |
eidcolumn = SQL_PREFIX + 'eid' |
0 | 239 |
for etype, in cursor.fetchall(): |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
240 |
table = SQL_PREFIX + etype |
1016
26387b836099
use datetime instead of mx.DateTime
sylvain.thenault@logilab.fr
parents:
713
diff
changeset
|
241 |
for rel, default in ( ('creation_date', datetime.now()), |
26387b836099
use datetime instead of mx.DateTime
sylvain.thenault@logilab.fr
parents:
713
diff
changeset
|
242 |
('modification_date', datetime.now()), ): |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
243 |
column = SQL_PREFIX + rel |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
244 |
cursor = session.system_sql("SELECT %s FROM %s WHERE %s is NULL" |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
245 |
% (eidcolumn, table, column)) |
0 | 246 |
for eid, in cursor.fetchall(): |
247 |
msg = ' %s with eid %s has no %s' |
|
248 |
print >> sys.stderr, msg % (etype, eid, rel), |
|
249 |
if fix: |
|
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
250 |
session.system_sql("UPDATE %s SET %s=%%(v)s WHERE %s=%s ;" |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
251 |
% (table, column, eidcolumn, eid), |
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
252 |
{'v': default}) |
0 | 253 |
print >> sys.stderr, ' [FIXED]' |
254 |
else: |
|
255 |
print >> sys.stderr |
|
1398
5fe84a5f7035
rename internal entity types to have CW prefix instead of E
sylvain.thenault@logilab.fr
parents:
1263
diff
changeset
|
256 |
cursor = session.system_sql('SELECT MIN(%s) FROM %sCWUser;' % (eidcolumn, |
1251
af40e615dc89
introduce a 'cw_' prefix on entity table and column names so we don't conflict with sql or DBMS specific keywords
sylvain.thenault@logilab.fr
parents:
1161
diff
changeset
|
257 |
SQL_PREFIX)) |
0 | 258 |
default_user_eid = cursor.fetchone()[0] |
259 |
assert default_user_eid is not None, 'no user defined !' |
|
260 |
for rel, default in ( ('owned_by', default_user_eid), ): |
|
261 |
cursor = session.system_sql("SELECT eid, type FROM entities " |
|
262 |
"WHERE NOT EXISTS " |
|
263 |
"(SELECT 1 FROM %s_relation WHERE eid_from=eid);" |
|
264 |
% rel) |
|
265 |
for eid, etype in cursor.fetchall(): |
|
266 |
msg = ' %s with eid %s has no %s relation' |
|
267 |
print >> sys.stderr, msg % (etype, eid, rel), |
|
268 |
if fix: |
|
269 |
session.system_sql('INSERT INTO %s_relation VALUES (%s, %s) ;' |
|
270 |
% (rel, eid, default)) |
|
271 |
print >> sys.stderr, ' [FIXED]' |
|
272 |
else: |
|
273 |
print >> sys.stderr |
|
274 |
||
275 |
||
276 |
def check(repo, cnx, checks, reindex, fix): |
|
277 |
"""check integrity of application's repository, |
|
278 |
using given user and password to locally connect to the repository |
|
279 |
(no running cubicweb server needed) |
|
280 |
""" |
|
281 |
session = repo._get_session(cnx.sessionid, setpool=True) |
|
282 |
# yo, launch checks |
|
283 |
if checks: |
|
284 |
eids_cache = {} |
|
285 |
for check in checks: |
|
286 |
check_func = globals()['check_%s' % check] |
|
287 |
check_func(repo.schema, session, eids_cache, fix=fix) |
|
288 |
if fix: |
|
289 |
cnx.commit() |
|
290 |
else: |
|
291 |
print |
|
292 |
if not fix: |
|
293 |
print 'WARNING: Diagnostic run, nothing has been corrected' |
|
294 |
if reindex: |
|
295 |
cnx.rollback() |
|
296 |
session.set_pool() |
|
297 |
reindex_entities(repo.schema, session) |
|
298 |
cnx.commit() |