iwla

iwla Git Source Tree

Root/plugins/post_analysis/top_downloads.py

Source at commit 4e02325733e5e8e4f5de2f0046e721f8da7abfff created 6 years 10 months ago.
By Gregory Soutade, Initial commit
1# -*- coding: utf-8 -*-
2#
3# Copyright Grégory Soutadé 2015
4
5# This file is part of iwla
6
7# iwla is free software: you can redistribute it and/or modify
8# it under the terms of the GNU General Public License as published by
9# the Free Software Foundation, either version 3 of the License, or
10# (at your option) any later version.
11#
12# iwla is distributed in the hope that it will be useful,
13# but WITHOUT ANY WARRANTY; without even the implied warranty of
14# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
15# GNU General Public License for more details.
16#
17# You should have received a copy of the GNU General Public License
18# along with iwla. If not, see <http://www.gnu.org/licenses/>.
19#
20
21import re
22
23from iwla import IWLA
24from iplugin import IPlugin
25
26"""
27Post analysis hook
28
29Count TOP downloads
30
31Plugin requirements :
32 None
33
34Conf values needed :
35 None
36
37Output files :
38 None
39
40Statistics creation :
41 None
42
43Statistics update :
44month_stats:
45 top_downloads =>
46 uri
47
48Statistics deletion :
49 None
50"""
51
52class IWLAPostAnalysisTopDownloads(IPlugin):
53 def __init__(self, iwla):
54 super(IWLAPostAnalysisTopDownloads, self).__init__(iwla)
55 self.API_VERSION = 1
56 self.conf_requires = ['multimedia_files', 'viewed_http_codes']
57
58 def hook(self):
59 stats = self.iwla.getCurrentVisists()
60 month_stats = self.iwla.getMonthStats()
61
62 multimedia_files = self.iwla.getConfValue('multimedia_files')
63 viewed_http_codes = self.iwla.getConfValue('viewed_http_codes')
64
65 top_downloads = month_stats.get('top_downloads', {})
66
67 for (k, super_hit) in stats.items():
68 if super_hit['robot']: continue
69 for r in super_hit['requests'][::-1]:
70 if not self.iwla.isValidForCurrentAnalysis(r):
71 break
72 if not self.iwla.hasBeenViewed(r) or\
73 r['is_page']:
74 continue
75
76 uri = r['extract_request']['extract_uri'].lower()
77
78 isMultimedia = False
79 for ext in multimedia_files:
80 if uri.endswith(ext):
81 isMultimedia = True
82 break
83
84 if isMultimedia: continue
85
86 uri = "%s%s" % (r.get('server_name', ''),
87 r['extract_request']['extract_uri'])
88
89 if not uri in top_downloads.keys():
90 top_downloads[uri] = 1
91 else:
92 top_downloads[uri] += 1
93
94 month_stats['top_downloads'] = top_downloads

Archive Download this file

Branches

Tags