iwla

iwla Git Source Tree

Root/plugins/post_analysis/top_pages.py

Source at commit 4e02325733e5e8e4f5de2f0046e721f8da7abfff created 6 years 10 months ago.
By Gregory Soutade, Initial commit
1# -*- coding: utf-8 -*-
2#
3# Copyright Grégory Soutadé 2015
4
5# This file is part of iwla
6
7# iwla is free software: you can redistribute it and/or modify
8# it under the terms of the GNU General Public License as published by
9# the Free Software Foundation, either version 3 of the License, or
10# (at your option) any later version.
11#
12# iwla is distributed in the hope that it will be useful,
13# but WITHOUT ANY WARRANTY; without even the implied warranty of
14# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
15# GNU General Public License for more details.
16#
17# You should have received a copy of the GNU General Public License
18# along with iwla. If not, see <http://www.gnu.org/licenses/>.
19#
20
21import re
22
23from iwla import IWLA
24from iplugin import IPlugin
25
26"""
27Post analysis hook
28
29Count TOP pages
30
31Plugin requirements :
32 None
33
34Conf values needed :
35 None
36
37Output files :
38 None
39
40Statistics creation :
41 None
42
43Statistics update :
44month_stats:
45 top_pages =>
46 uri
47
48Statistics deletion :
49 None
50"""
51
52class IWLAPostAnalysisTopPages(IPlugin):
53 def __init__(self, iwla):
54 super(IWLAPostAnalysisTopPages, self).__init__(iwla)
55 self.API_VERSION = 1
56
57 def load(self):
58 self.index_re = re.compile(r'/index.*')
59 return True
60
61 def hook(self):
62 stats = self.iwla.getCurrentVisists()
63 month_stats = self.iwla.getMonthStats()
64
65 top_pages = month_stats.get('top_pages', {})
66
67 for (k, super_hit) in stats.items():
68 if super_hit['robot']: continue
69 for r in super_hit['requests'][::-1]:
70 if not self.iwla.isValidForCurrentAnalysis(r):
71 break
72 if not self.iwla.hasBeenViewed(r) or\
73 not r['is_page']:
74 continue
75
76 uri = r['extract_request']['extract_uri']
77 if self.index_re.match(uri):
78 uri = '/'
79
80 uri = "%s%s" % (r.get('server_name', ''), uri)
81
82 if not uri in top_pages.keys():
83 top_pages[uri] = 1
84 else:
85 top_pages[uri] += 1
86
87 month_stats['top_pages'] = top_pages

Archive Download this file

Branches

Tags