kallithea Changeset - 8df1e9edd68f

Changeset - 8df1e9edd68f

Parent rev.

Child rev.

[Not reviewed]

default

0 3 0

FUJIWARA Katsunori - 9 years ago 2017-01-30 11:09:45
foozy@lares.dti.ne.jp

indexers: use correct full repository name, which contains group name, at indexing

Before this revision, searching under the specific repository could
cause unexpected result, because repository names used for indexing didn't
contain the group name.

This issue was introduced by 8b7c0ef62427, which uses
repo.name_unicode as repository name instead of
safe_unicode(repo_name) to reduce unicode conversion cost while
repetition at indexing.

To use correct repository name at indexing, this revision replaces
repo.name_unicode by safe_unicode(repo_name). Reducing cost of repeated
unicode conversion cost while will (and should) be addressed in the
future.

This revision also adds a comment to BaseRepository.name property, to
avoid similar misunderstandings in the future.

3 files changed with 24 insertions and 2 deletions:

kallithea/lib/indexers/daemon.py

kallithea/lib/vcs/backends/base.py

kallithea/tests/functional/test_search_indexing.py

0 comments (0 inline, 0 general)

kallithea/lib/indexers/daemon.py

➞

Show inline comments

@@ @@ -182,88 +182,88 @@ class WhooshIndexingDaemon(object): @@
             log.debug("couldn't add doc - %s did not have %r at %s", repo, path, index_rev)
             return 0, 0
         indexed = indexed_w_content = 0
         if self.is_indexable_node(node):
             u_content = node.content
             if not isinstance(u_content, unicode):
                 log.warning('  >> %s Could not get this content as unicode '
                             'replacing with empty content' % path)
                 u_content = u''
             else:
                 log.debug('    >> %s [WITH CONTENT]', path)
                 indexed_w_content += 1
         else:
             log.debug('    >> %s', path)
             # just index file name without it's content
             u_content = u''
             indexed += 1
         p = safe_unicode(path)
         writer.add_document(
             fileid=p,
             owner=unicode(repo.contact),
-            repository_rawname=repo.name_unicode,
+            repository_rawname=safe_unicode(repo_name),
             repository=safe_unicode(repo_name),
             path=p,
             content=u_content,
             modtime=self.get_node_mtime(node),
             extension=node.extension
+        )
         return indexed, indexed_w_content
     def index_changesets(self, writer, repo_name, repo, start_rev=None):
         """
         Add all changeset in the vcs repo starting at start_rev
         to the index writer
         :param writer: the whoosh index writer to add to
         :param repo_name: name of the repository from whence the
           changeset originates including the repository group
         :param repo: the vcs repository instance to index changesets for,
           the presumption is the repo has changesets to index
         :param start_rev=None: the full sha id to start indexing from
           if start_rev is None then index from the first changeset in
           the repo
         """
         if start_rev is None:
             start_rev = repo[0].raw_id
         log.debug('indexing changesets in %s starting at rev: %s',
                   repo_name, start_rev)
         indexed = 0
         cs_iter = repo.get_changesets(start=start_rev)
         total = len(cs_iter)
         for cs in cs_iter:
             log.debug('    >> %s/%s', cs, total)
             writer.add_document(
                 raw_id=unicode(cs.raw_id),
                 owner=unicode(repo.contact),
                 date=cs._timestamp,
-                repository_rawname=repo.name_unicode,
+                repository_rawname=safe_unicode(repo_name),
                 repository=safe_unicode(repo_name),
                 author=cs.author,
                 message=cs.message,
                 last=cs.last,
                 added=u' '.join([safe_unicode(node.path) for node in cs.added]).lower(),
                 removed=u' '.join([safe_unicode(node.path) for node in cs.removed]).lower(),
                 changed=u' '.join([safe_unicode(node.path) for node in cs.changed]).lower(),
                 parents=u' '.join([cs.raw_id for cs in cs.parents]),
+            )
             indexed += 1
         log.debug('indexed %d changesets for repo %s', indexed, repo_name)
         return indexed
     def index_files(self, file_idx_writer, repo_name, repo):
         """
         Index files for given repo_name
         :param file_idx_writer: the whoosh index writer to add to
         :param repo_name: name of the repository we're indexing
         :param repo: instance of vcs repo
         """
         i_cnt = iwc_cnt = 0
         log.debug('building index for %s @revision:%s', repo.path,

kallithea/lib/vcs/backends/base.py

➞

Show inline comments

@@ @@ -76,48 +76,51 @@ class BaseRepository(object): @@
     def __str__(self):
         return '<%s at %s>' % (self.__class__.__name__, self.path)
     def __repr__(self):
         return self.__str__()
     def __len__(self):
         return self.count()
     def __eq__(self, other):
         same_instance = isinstance(other, self.__class__)
         return same_instance and getattr(other, 'path', None) == self.path
     def __ne__(self, other):
         return not self.__eq__(other)
     @LazyProperty
     def alias(self):
         for k, v in settings.BACKENDS.items():
             if v.split('.')[-1] == str(self.__class__.__name__):
                 return k
     @LazyProperty
     def name(self):
         """
         Return repository name (without group name)
         """
         raise NotImplementedError
     @property
     def name_unicode(self):
         return safe_unicode(self.name)
     @LazyProperty
     def owner(self):
         raise NotImplementedError
     @LazyProperty
     def description(self):
         raise NotImplementedError
     @LazyProperty
     def size(self):
         """
         Returns combined size in bytes for all repository files
         """
         size = 0
         try:
             tip = self.get_changeset()
             for topnode, dirs, files in tip.walk('/'):

kallithea/tests/functional/test_search_indexing.py

➞

Show inline comments

@@ @@ -104,48 +104,67 @@ class TestSearchControllerIndexing(TestC @@
         rebuild_index(full_index=True) # rebuild fully for subsequent tests
     @parametrize('reponame', [
         (u'indexing_test'),
         (u'indexing_test-fork'),
         (u'group/indexing_test'),
         (u'this-is-it'),
         (u'*-fork'),
         (u'group/*'),
     ])
     @parametrize('searchtype,query,hit', [
         ('content', 'this_should_be_unique_content', 1),
         ('commit', 'this_should_be_unique_commit_log', 1),
         ('path', 'this_should_be_unique_filename.txt', 1),
     ])
     def test_repository_tokenization(self, reponame, searchtype, query, hit):
         self.log_user()
         q = 'repository:%s %s' % (reponame, query)
         response = self.app.get(url(controller='search', action='index'),
                                 {'q': q, 'type': searchtype})
         response.mustcontain('>%d results' % hit)
     @parametrize('reponame', [
         (u'indexing_test'),
         (u'indexing_test-fork'),
         (u'group/indexing_test'),
         (u'this-is-it'),
     ])
     @parametrize('searchtype,query,hit', [
         ('content', 'this_should_be_unique_content', 1),
         ('commit', 'this_should_be_unique_commit_log', 1),
         ('path', 'this_should_be_unique_filename.txt', 1),
     ])
     def test_searching_under_repository(self, reponame, searchtype, query, hit):
         self.log_user()
         response = self.app.get(url(controller='search', action='index',
                                     repo_name=reponame),
                                 {'q': query, 'type': searchtype})
         response.mustcontain('>%d results' % hit)
     @parametrize('searchtype,query,hit', [
         ('content', 'this_should_be_unique_content', 1),
         ('commit', 'this_should_be_unique_commit_log', 1),
         ('path', 'this_should_be_unique_filename.txt', 1),
     ])
     def test_repository_case_sensitivity(self, searchtype, query, hit):
         self.log_user()
         lname = u'indexing_test-foo'
         uname = u'indexing_test-FOO'
         # (1) "repository:REPONAME" condition should match against
         # repositories case-insensitively
         q = 'repository:%s %s' % (lname, query)
         response = self.app.get(url(controller='search', action='index'),
                                 {'q': q, 'type': searchtype})
         response.mustcontain('>%d results' % (hit * 2))
         # (2) on the other hand, searching under the specific
         # repository should return results only for that repository,
         # even if specified name matches against another repository
         # case-insensitively.
         response = self.app.get(url(controller='search', action='index',

0 comments (0 inline, 0 general)