Report example
Report of where your hits come from
To run this program, simply type in the script's URL - or set it as a bookmark. This script reads the Sources.Log file, sorts the entries, and gives you percentages for different types of hits.
# specify the perl path
#!/usr/bin/perl
# SCRIPT TO REPORT ON YOUR SITE AUDIENCE
# Specify the log file that stores the referring URLs. The record format of
# each line is XX www.somewhere.com where XX is the number of people who have
# come from that page. There is a line for each site, and it is case sensitive.
# Modify the relative path of the log filename if appropriate.
$source_file = "../logs/Sources.Log";
# Set your domain name.
$my_url = "myurl.com";
# Names of search engines you may get hits from (add to it as you see fit)
@search_engine_ids = ("altavista", "dogpile", "excite", "hotbot", "infoseek", "lycos", "mckinley", "metacrawler", "webcrawler", "yahoo");
# Directs the script's output (the "print" statements) to your browser.
print "Content-Type: text/html\n\n";
# Open and read the log file that stores the referring URLs.
open(source_file,"$source_file");
flock(source_file,2);
# Put the data into an array named @sources.
@sources = <source_file>;
close(source_file);
# what does this do?
chop(@sources);
# Initialise total sources count.
$total_sources = 0;
# Look at each line in the array, one by one
foreach $refer (@sources)
# Separate separates the number from the URL. The blank space between the two
# slashes tells Perl to assign anything before the space to the variable
# $number, and anything after the space to the variable $url.
{ ($number,$url) = split(/ /,$refer,2);
# Keep a running count by adding the number for each URL to the
# $total_sources variable.
$total_sources = ($total_sources + $number);
# CATEGORISE THE URLs.
# We are going to search whether the current URL is for a search engine, so we
# start by setting the 'found' variable to false - we will set it to true if we
# find a match.
$found = "false";
# Search the array of search engine ids, comparing each against the current URL
# from the log file. If a match is found, set found = true. Also set
# found = true if there is an "=" in the URL, as search engines tend to use
# the '=' character when appending the user's keywords to the URL.
foreach $engine(@search_engine_ids)
{ if ((/$engine/) || (/=/))
{ $found = "true"; }
}
# If it is not a search engine, check for a direct hit and if so, set the
# number of direct hits.
if ($url eq "Direct Hit")
{ $direct_total = $number; }
# If it is not a direct hit either, see if this line is for hits from other
# pages on your site, and store the URL in the 'my' array.
elsif (/$my_url/)
{ $my_total = ($my_total + $number);
push(@my_url,$_);
}
# If we found a search engine, add the number to the search engine total and
# store the URL in the 'search engines' array.
elsif ($found eq "true")
{ $search_total = ($search_total + $number);
push(@search_engines,$_);
}
# Otherwise add the number to the total of other URLs, and store the URL in the
# 'other' array.
else
{ $other_total = ($other_total + $number);
push(@other_urls,$_);
}
}
# Calculate the percentage of total hits from each category.
# First check that there is at least one, so we don't get divide by zero
# errors. If there are none, set all the percentages to "n/a." instead.
if ($total_sources > 0)
# Calculate the percentage of total hits from each category.
{ $direct_percent = substr((($direct_total/$total_sources)*100),0,5);
$my_percent = substr((($my_total/$total_sources)*100),0,5);
$other_percent = substr((($other_total/$total_sources)*100),0,5);
$search_percent = substr((($search_total/$total_sources)*100),0,5);
}
# If there are no hits, set all the percentages to "n/a".
else
{ $direct_percent = "n/a";
$my_percent = "n/a";
$other_percent = "n/a";
$search_percent = "n/a";
}
# PRINT ALL THE INFORMATION TO BROWSER
# One table per category, with a header including the number of hits for the
# category and the percentage of the total number of hits the category
# accounts for, and one line per URL in the category (except Direct Hits).
# HTML for start of table tag.
print "<table>\n";
# where do report_size and count_table come from?
for ($x=0; $x < $report_size; $x++)
{ print "@count_table[$x]"; }
# HTML for end of table tag.
print "</table>\n\n";
# Print the 'Direct Hits' total and percentage.
print "<h1>Referring URL's</h1>";
print "Total number of hits: $total_sources<p>\n";
print "<h2>Direct Hits</h2>\n";
print "Total: $direct_total - Percentage of all hits:
$direct_percent<p>\n";
# Print the 'Hits from other URL's' total and percentage.
print "<h2>Hits from other URL's</h2>\n";
print "Total: $other_total - Percentage of all hits:
$other_percent<p>\n";
# Call the print table subroutine to print the 'other' URLs.
@array=@other_urls;
&print_table;
# Print the 'Hits from search engines' total and percentage.
print "<a name=\"search\"></a>\n";
print "<h2>Hits from search engines</h2>\n";
print "Total: $search_total - Percentage of all hits:
$search_percent<p>\n";
# Call the print table subroutine to print the 'search engine' URLs.
@array=@search_engines;
&print_table;
# Print the 'Hits from my site' total and percentage.
print "<h2>Hits from my site</h2>\n";
print "Total: $my_total - Percentage of all hits: $my_percent<p>\n";
# Call the print table subroutine to print the 'my URL' URLs.
@array=@my_url;
&print_table;
# Display the form button to submit.
# where does clear.pl come from?
print "<form method=\"post\" action=\"../cgi/clear.pl\">";
print "<input type=\"text\" name=\"l\">";
print "<input type=\"submit\" value=\"submit\">";
print "</form>";
# Subroutine called once for each table, with a different array each time.
sub print_table
{
# Sort the array in descending order using the number, so that the most popular
# URLs end up at the top.
@array = sort descending @array;
sub descending
{ if ($a < $b) { 1; }
elsif ($a == $b) { 0; }
elsif ($a > $b) { -1; }
}
# PRINT EACH URL.
# HTML for start of table tag.
print "<table cellpadding=5 border=1>\n";
# Print the number and percentage for each URL.
foreach $line(@array)
{ ($number,$url) = split(/ /,$line,2);
$percentage = substr((($number/$total_sources)*100),0,5);
$percentage = substr($percentage,1,5);
# Each URL is formatted as a link so you can go and look.
print "<tr><td>$number<td>$percentage%<td><a href=\"$url\">$url\n";
}
# HTML for end of table tag.
print "</table><p>\n\n";
}
If the tables become large and slow to load, or information hard to find, you can simply replace the full Sources.Log with an empty one.
