Software Sue

Link to Simply Sue
Link to Sustainability Sue
location: software/scripting/perl report script Skip navigation : Home  

Report example

Report of where your hits come from


To run this program, simply type in the script's URL - or set it as a bookmark. This script reads the Sources.Log file, sorts the entries, and gives you percentages for different types of hits.

# specify the perl path

#!/usr/bin/perl

# SCRIPT TO REPORT ON YOUR SITE AUDIENCE

# Specify the log file that stores the referring URLs. The record format of
# each line is XX www.somewhere.com where XX is the number of people who have
# come from that page. There is a line for each site, and it is case sensitive.
# Modify the relative path of the log filename if appropriate.

$source_file = "../logs/Sources.Log";

# Set your domain name.

$my_url = "myurl.com";

# Names of search engines you may get hits from (add to it as you see fit)

@search_engine_ids = ("altavista", "dogpile", "excite", "hotbot", "infoseek", "lycos", "mckinley", "metacrawler", "webcrawler", "yahoo");

# Directs the script's output (the "print" statements) to your browser.

print "Content-Type: text/html\n\n";

# Open and read the log file that stores the referring URLs.

open(source_file,"$source_file");
flock(source_file,2);

# Put the data into an array named @sources.

@sources = <source_file>;
close(source_file);

# what does this do?

chop(@sources);

# Initialise total sources count.

$total_sources = 0;

# Look at each line in the array, one by one

foreach $refer (@sources)

# Separate separates the number from the URL. The blank space between the two
# slashes tells Perl to assign anything before the space to the variable
# $number, and anything after the space to the variable $url.

{ ($number,$url) = split(/ /,$refer,2);

# Keep a running count by adding the number for each URL to the
# $total_sources variable.

    $total_sources = ($total_sources + $number);

# CATEGORISE THE URLs.

# We are going to search whether the current URL is for a search engine, so we
# start by setting the 'found' variable to false - we will set it to true if we
# find a match.

    $found = "false";

# Search the array of search engine ids, comparing each against the current URL
# from the log file. If a match is found, set found = true. Also set
# found = true if there is an "=" in the URL, as search engines tend to use
# the '=' character when appending the user's keywords to the URL.

    foreach $engine(@search_engine_ids)
      { if ((/$engine/) || (/=/))
        { $found = "true"; }
      }

# If it is not a search engine, check for a direct hit and if so, set the
# number of direct hits.

   if ($url eq "Direct Hit")
     { $direct_total = $number; }

# If it is not a direct hit either, see if this line is for hits from other
# pages on your site, and store the URL in the 'my' array.

   elsif (/$my_url/)
     { $my_total = ($my_total + $number);
       push(@my_url,$_);
     }

# If we found a search engine, add the number to the search engine total and
# store the URL in the 'search engines' array.

   elsif ($found eq "true")
     { $search_total = ($search_total + $number);
       push(@search_engines,$_);
     }

# Otherwise add the number to the total of other URLs, and store the URL in the
# 'other' array.

   else
     { $other_total = ($other_total + $number);
       push(@other_urls,$_);
     }
 }

# Calculate the percentage of total hits from each category.

# First check that there is at least one, so we don't get divide by zero
# errors. If there are none, set all the percentages to "n/a." instead.

if ($total_sources > 0)

# Calculate the percentage of total hits from each category.

  { $direct_percent = substr((($direct_total/$total_sources)*100),0,5);
    $my_percent = substr((($my_total/$total_sources)*100),0,5);
    $other_percent = substr((($other_total/$total_sources)*100),0,5);
    $search_percent = substr((($search_total/$total_sources)*100),0,5);
  }

# If there are no hits, set all the percentages to "n/a".

 else
 { $direct_percent = "n/a";
   $my_percent = "n/a";
   $other_percent = "n/a";
   $search_percent = "n/a";
 }

# PRINT ALL THE INFORMATION TO BROWSER
# One table per category, with a header including the number of hits for the
# category and the percentage of the total number of hits the category
# accounts for, and one line per URL in the category (except Direct Hits).

# HTML for start of table tag.

print "<table>\n";

# where do report_size and count_table come from?

for ($x=0; $x < $report_size; $x++)
 { print "@count_table[$x]"; }

# HTML for end of table tag.

print "</table>\n\n";

# Print the 'Direct Hits' total and percentage.

print "<h1>Referring URL's</h1>";
print "Total number of hits: $total_sources<p>\n";
print "<h2>Direct Hits</h2>\n";
print "Total: $direct_total - Percentage of all hits:
$direct_percent<p>\n";

# Print the 'Hits from other URL's' total and percentage.

print "<h2>Hits from other URL's</h2>\n";
print "Total: $other_total - Percentage of all hits:
$other_percent<p>\n";

# Call the print table subroutine to print the 'other' URLs.

@array=@other_urls;
&print_table;

# Print the 'Hits from search engines' total and percentage.

print "<a name=\"search\"></a>\n";
print "<h2>Hits from search engines</h2>\n";
print "Total: $search_total - Percentage of all hits:
$search_percent<p>\n";

# Call the print table subroutine to print the 'search engine' URLs.

@array=@search_engines;
&print_table;

# Print the 'Hits from my site' total and percentage.

print "<h2>Hits from my site</h2>\n";
print "Total: $my_total - Percentage of all hits: $my_percent<p>\n";

# Call the print table subroutine to print the 'my URL' URLs.

@array=@my_url;
&print_table;

# Display the form button to submit.
# where does clear.pl come from?

print "<form method=\"post\" action=\"../cgi/clear.pl\">";
print "<input type=\"text\" name=\"l\">";
print "<input type=\"submit\" value=\"submit\">";
print "</form>";

# Subroutine called once for each table, with a different array each time.

sub print_table
{

# Sort the array in descending order using the number, so that the most popular
# URLs end up at the top.

@array = sort descending @array;
sub descending
  { if ($a < $b) { 1; }
    elsif ($a == $b) { 0; }
    elsif ($a > $b) { -1; }
  }

# PRINT EACH URL.

# HTML for start of table tag.

print "<table cellpadding=5 border=1>\n";

# Print the number and percentage for each URL.

foreach $line(@array)
  { ($number,$url) = split(/ /,$line,2);
    $percentage = substr((($number/$total_sources)*100),0,5);
    $percentage = substr($percentage,1,5);

# Each URL is formatted as a link so you can go and look.

    print "<tr><td>$number<td>$percentage%<td><a href=\"$url\">$url\n";
  }

# HTML for end of table tag.

print "</table><p>\n\n"; }

If the tables become large and slow to load, or information hard to find, you can simply replace the full Sources.Log with an empty one.