All pastes #1770358 Raw Edit

Tumblr Duplicate Detector

public text v1 · immutable
#1770358 ·published 2010-01-29 05:01 UTC
rendered paste body
#!/usr/bin/perl -s

use strict;
use warnings;
use WWW::Tumblr;
use File::Slurp;

my ($tumblr, $blog, $nposts, $post, $source, $ts, %tab, $url);

$| = 1;
$tumblr = WWW::Tumblr->new;

my $user = "bondagescout"; # change to the name of your blog or use -u=blah
$user = $main::u if (defined($main::u) && length($main::u));
$tumblr->user($user);

my $refetch;
if (defined($main::r) && defined($main::r)){
	$refetch = "1";
} else {
	$refetch = "0";
}

my $cache = "cache-${user}.xml";
$refetch = 1 unless ( -f $cache );
my $n = 50;
my $st = 0;
if ( $refetch ){
	unlink($cache);
	open(F, ">>$cache"); # should really use mkstemp(3)
	do {
		do {
			$blog = $tumblr->read(start => $st, num => $n);
			sleep(1) unless ($nposts = extract_postcount($blog));
		} until ($nposts);
		print F $blog . "\n";
		print "$st/$nposts   ";
	} while (($st += $n) <= $nposts);
	close(F);
	print "$nposts/$nposts\n";
}

$blog = read_file($cache);
$nposts = 0;
while ($blog =~ /(<post.+?<\/post>)/gsm){
	$nposts++;
	$post = $1;

	$ts = extract_time_from_xml($post);
	$source = extract_source_from_xml($post);
	$post =~ /url="(.*?)"/gism;
	$url = $1;

	if ($source =~ /http:\/\//) {
	#	print "$nposts $ts $source\n";
		if (defined($tab{$source})){
			printf("dup: %s %s\n", $url, $tab{$source});
		}
		$tab{$source} = $url;
	} else {
		print "$nposts $ts missing source in $url\n";
		exit;
	}
}

sub extract_postcount {
	my $blog = shift;
	return $1 if ($blog =~ /<posts start="\d+" total="(\d+)">/);
	return 0;
}

sub extract_time_from_xml {
	my $post = shift;
	if ($post =~ /unix-timestamp="(\d+)"/gsm) {
		return $1;
	} else {
		return 0;
	}
}
sub extract_source_from_xml {
	my $post = shift;
	if ($post =~ /<photo-caption>.*?href="(.*?\/post\/\d+)(\/.*?)?"/gism) {
		return $1;
	} elsif ($post =~ /via.*?a href="(.*?\/post\/\d+)(\/.*?)?"/gism) {
		return $1;
	} elsif ($post =~ /<regular-body>.*?href="(.*?\/post\/\d+)(\/.*?)?"/gism) {
		return $1;
	} elsif ($post =~ /<photo-link-url>(.*?)<\/photo-link-url>/gism) {
		return $1;
	} elsif ($post =~ /url="(.*?)"/gism) {
		my $x = $1;
		return $x if ($post =~ /format="html"/);
	}
	return '';
}