39 Commits
Author SHA1 Message Date
Brian Ó Donnell 529b0336da Merge from master 2016-08-11 11:17:10 -04:00
Brian Ó Donnell a2d5be606c Merge from master 2016-08-11 11:09:52 -04:00
Brian Ó Donnell ddf4746b7a merge from master 2016-04-19 22:22:32 -04:00
Brian Ó Donnell 0fb637dcb4 merge from master 2016-04-19 16:51:00 -04:00
Brian Ó Donnell 0e4220b8f2 merge from master 2016-04-19 16:19:17 -04:00
Brian Ó Donnell a11324bec9 merge from master 2016-04-18 19:46:56 -04:00
Brian Ó Donnell 0e214cccf9 merge from master 2016-04-15 08:07:24 -04:00
Brian O'Donnell 164a753881 merge from master 2016-04-14 13:26:16 -04:00
Brian O'Donnell cd3ef8c875 merge from master 2016-04-14 10:18:12 -04:00
Brian O'Donnell 230bbda35d merge from master 2016-04-13 21:56:04 -04:00
Brian Ó Donnell dc457c1907 Merge from master 2016-02-22 22:59:03 -05:00
Brian Ó Donnell 22c263f470 Merge from master 2015-12-30 12:49:39 -05:00
Brian Ó Donnell 51ee245558 Merge from master 2015-12-30 11:04:38 -05:00
Brian O'Donnell f98620d806 Merge from master 2015-11-18 20:18:26 -05:00
Brian O'Donnell 89ac6d1ada Merge from master 2015-11-18 14:16:09 -05:00
Brian O'Donnell d5c29b0b96 Strip leading/trailing spaces and newlines from feed elements 2015-08-09 22:19:45 -04:00
Brian O'Donnell ea5c0d0c2c Recovered pod_tweeter.pl from previous commit 2015-07-27 23:44:30 -04:00
Brian O'Donnell eb791d595e Merge changes from master 2015-07-27 23:40:35 -04:00
Brian O'Donnell ef751f7598 Make sure the image link is in an 'img' tag 2015-07-25 22:20:52 -04:00
Brian O'Donnell 2c1055128b Merged changes from master 2015-07-25 15:45:00 -04:00
Brian O'Donnell 518ce2eecf Merge from master 2015-07-16 13:17:18 -04:00
Brian O'Donnell 39e36d8112 Merge from master 2015-07-16 08:12:24 -04:00
Brian O'Donnell 06d031cd98 Reverted to an accidentally stomped over version 2015-07-07 11:48:23 -04:00
Brian O'Donnell b5e663834f Merged changes from master 2015-07-06 22:04:03 -04:00
Brian O'Donnell 19228c2d73 Embedded images now link back to the feed item 2015-07-06 22:03:39 -04:00
Brian O'Donnell 085ba1853f Merged changes from master 2015-07-06 21:16:12 -04:00
Brian O'Donnell c86e165726 Merged changes from master 2015-07-06 20:32:19 -04:00
Brian O'Donnell 34475890a9 Merged changes from master 2015-07-05 16:39:32 -04:00
Brian O'Donnell d6c719ff76 Merged changes from master 2015-07-05 16:20:33 -04:00
Brian O'Donnell 0a64ffd738 Merged changes from master 2015-07-05 11:56:23 -04:00
Brian O'Donnell f91081485d Merge from master 2015-05-28 22:41:48 -04:00
Brian O'Donnell d374ffd0d0 Merge changes from master 2015-05-27 10:46:02 -04:00
Brian O'Donnell 1619049c86 Substitue link for guid if guid field is missing 2015-05-24 21:43:49 -04:00
Brian O'Donnell 962f62e210 Merge branch 'pod_tweeter' of github.com:rev138/pod_feeder into pod_tweeter 2015-05-13 22:15:58 -04:00
Brian O'Donnell c600a039ea Fixed table name in query 2015-05-13 22:15:47 -04:00
rev138 c13df3d968 Update README.md 2015-05-13 21:44:57 -04:00
Brian O'Donnell b829ee2453 Fixed options and help output 2015-05-13 21:33:28 -04:00
Brian O'Donnell 6fe3aa56e8 Changed provider name to 'pod_tweeter' 2015-05-13 19:42:12 -04:00
Brian O'Donnell d453b7a59a Adapted pod_feeder.pl to slurp twitter feeds 2015-05-13 18:56:47 -04:00
4 changed files with 754 additions and 150 deletions
+1
View File
@@ -0,0 +1 @@
*.db
+45
View File
@@ -52,3 +52,48 @@ Modify it thusly, then feed it to the script:
[https://www.youtube.com/**rss**/channel/UCQzdMyuz0Lf4zo4uGcEujFw/**feed.rss**](https://www.youtube.com/rss/channel/UCQzdMyuz0Lf4zo4uGcEujFw/feed.rss)
If you'd want Diaspora to automatically embed the video, you must also pass the `--post-raw-link` argument
# pod_tweeter
Publishes twitter feeds to Diaspora*
This is a lightweight, customizable "bot" script to harvest twitter feeds and re-publish them to the Diaspora social network. It is posted here without warranty, for public use.
## Installation
You must have the following perl modules installed:
- LWP::UserAgent
- URI::Escape
- HTML::Entities
- JSON
- DBD::SQLite
- Unicode::Normalize
- Getopt::Long
- Net::Twitter::Lite::WithAPIv1_1
- DateTime
This script is intended to be run as a cron job, which might look something like this:
`@hourly ~/pod_tweeter.pl --timeline-id mytimeline --screen-name\@AesopRockWins --access-token beefbeefbeefbeef --access-token-secret feedfeedfeedfeed --consumer-token effeffeffeff --consumer-secret 0e0e0e0efff --pod-url https://diaspora.hzsogood.net --username user --password supersecretpassword > /dev/null 2>&1`
## Usage
-a --aspect-id <id> Aspects to share with. May specify multiple times (default: 'public')
-c --consumer-key <string> The twitter API consumer key
-d --database <sqlite file> The SQLite file to store feed data (default: 'feed.db')
-e --access-token-secret <string> The twitter API access token secret
-i --timeline-id <string> An arbitrary identifier to associate database entries with this feed
-k --access-token <string> The twitter API access token
-l --pod-url <https://...> The pod URL
-m --timeout <hours> How long (in hours) to keep attempting failed posts (default: 72)
-o --fetch-only Don't publish to Diaspora, just queue the new feed items for later
-p --password <********> The D* user password
-r --consumer-secret <string> The twitter API consumer secret
-s --screen-name <@screenname> The twitter feed to scrape (default: the user associated with the API keys)
-t --auto-tag <#hashtag> Hashtags to add to all posts. May be specified multiple times (default: none)
-u --username <user> The D* login username
-x --limit <n> Only post n items per script run, to prevent post-spamming (default: no limit)
## Note
In order to use this script, you must have a twitter developer account, create an "app" and generate the necessary tokens and secret keys. See https://apps.twitter.com
+274 -150
View File
@@ -25,11 +25,13 @@ use Unicode::Normalize 'normalize';
use Getopt::Long;
my $opts = {
'database' => './pod_feeder.db',
'limit' => 0,
'timeout' => 72, # hours
'database' => './pod_feeder.db',
'limit' => 0,
'timeout' => 72, # hours
'via' => 'pod_feeder',
};
my @auto_tags = ();
my @ignored_tags = ();
my @aspect_ids = ();
GetOptions(
@@ -38,11 +40,15 @@ GetOptions(
'auto-tag|t=s' => \@auto_tags,
'category-tags|c',
'database|d=s',
'embed-image|b',
'feed-id|i=s',
'feed-url|f=s',
'fetch-only|o',
'help|h', => \&usage,
'help|h', => \&usage,
'ignore-tag|n=s', => \@ignored_tags,
'insecure|s=s',
'limit|x=i',
'no-branding',
'password|p=s',
'pod-url|l=s',
'post-raw-links|w',
@@ -51,6 +57,7 @@ GetOptions(
'url-tags|r',
'user-agent|g=s',
'username|u=s',
'via|v=s',
);
# defaults to 'public' if no aspect ids are specified
@@ -70,13 +77,14 @@ if( $fetched ){
eval {
# update the database
update_feed(
$feed,
db_file => $opts->{'database'},
feed_id => $opts->{'feed-id'},
auto_tags => hashtagify( \@auto_tags ),
extract_tags_from_url => $opts->{'url-tags'},
extract_tags_from_title => $opts->{'title-tags'},
tag_categories => $opts->{'category-tags'},
$feed,
db_file => $opts->{'database'},
feed_id => $opts->{'feed-id'},
auto_tags => hashtagify( \@auto_tags ),
ignored_tags => hashtagify( \@ignored_tags ),
extract_tags_from_url => $opts->{'url-tags'},
extract_tags_from_title => $opts->{'title-tags'},
tag_categories => $opts->{'category-tags'},
);
};
warn "$@" if $@;
@@ -84,15 +92,19 @@ if( $fetched ){
eval {
# publish new feed items to the pod, unless the user specified --fetch-only
publish_feed_items(
db_file => $opts->{'database'},
feed_id => $opts->{'feed-id'},
timeout => $opts->{'timeout'},
pod_url => $opts->{'pod-url'},
username => $opts->{'username'},
password => $opts->{'password'},
aspect_ids => \@aspect_ids,
raw_link => $opts->{'post-raw-links'},
limit => $opts->{'limit'},
db_file => $opts->{'database'},
embed_image => $opts->{'embed-image'},
feed_id => $opts->{'feed-id'},
timeout => $opts->{'timeout'},
pod_url => $opts->{'pod-url'},
username => $opts->{'username'},
password => $opts->{'password'},
aspect_ids => \@aspect_ids,
raw_link => $opts->{'post-raw-links'},
limit => $opts->{'limit'},
no_branding => $opts->{'no-branding'},
via => $opts->{'via'},
insecure => $opts->{'insecure'},
) unless $opts->{'fetch-only'};
};
warn "$@" if $@;
@@ -105,12 +117,12 @@ else {
sub publish_feed_items {
my ( %params ) = @_;
my @updates = ();
my $query_string = "SELECT guid, title, link, hashtags FROM feeds WHERE feed_id == ? AND posted == 0 AND timestamp > ? ORDER BY timestamp";
my $query_string = "SELECT guid, title, link, image, image_title, hashtags FROM feeds WHERE feed_id == ? AND posted == 0 AND timestamp > ? ORDER BY timestamp";
my $dbh = connect_to_db( $params{'db_file'} );
# limit the number of items published if limit is specified
$query_string .= " LIMIT $params{'limit'}" if $params{'limit'} > 0;
my $sth = $dbh->prepare( $query_string ) or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $params{'feed_id'}, time - ( $params{'timeout'} * 3600 ) ) or die "Can't execute statement: $DBI::errstr";
@@ -122,74 +134,91 @@ sub publish_feed_items {
foreach my $update ( @updates ){
my $content = $update->{'hashtags'};
# to hyperlink the title or not to hyperlink the title...
if( $params{'raw_link'} ){
$content = '### ' . $update->{'title'} . "\n\n" . $update->{'link'} . "\n" . $content;
}
else {
$content = '### [' . $update->{'title'} . '](' . $update->{'link'} . ")\n\n" . $content;
}
if( $params{'embed_image'} and length $update->{'image'} ){
my $image_link = '[![](' . $update->{'image'};
$image_link .= ' "' . $update->{'image_title'} . '"' if length $update->{'image_title'};
$image_link .= ')](' . $update->{'link'} . ')';
$content = "$image_link\n$content";
}
print "Publishing $params{'feed_id'}\t$update->{'guid'}\n";
my $post = publish_post( $content, %params );
# mark the item as successfully posted
if( $post->is_success ){
$sth = $dbh->prepare( "UPDATE feeds SET posted = 1 WHERE guid = ?" ) or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $update->{'guid'} ) or die "Can't execute statement: $DBI::errstr";
}
else {
warn $post->code . ' ' . $post->message;
}
# Now, don't be hasty, master Meriadoc
sleep 1;
# to hyperlink the title or not to hyperlink the title...
if( $params{'raw_link'} ){
$content = '### ' . $update->{'title'} . "\n\n" . $update->{'link'} . "\n" . $content;
}
else {
$content = '### [' . $update->{'title'} . '](' . $update->{'link'} . ")\n\n" . $content;
}
$dbh->disconnect();
print "Publishing $params{'feed_id'}\t$update->{'guid'}\n";
my $post = publish_post( $content, %params );
# mark the item as successfully posted
if( $post->is_success ){
$sth = $dbh->prepare( "UPDATE feeds SET posted = 1 WHERE guid = ?" ) or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $update->{'guid'} ) or die "Can't execute statement: $DBI::errstr";
}
else {
warn $post->code . ' ' . $post->message;
}
# Now, don't be hasty, master Meriadoc
sleep 1;
}
$dbh->disconnect();
}
# adds new feed items to the database
sub update_feed {
my ( $feed, %params ) = @_;
$params{'auto_tags'} = 0 unless defined $params{'auto_tags'};
$params{'auto_tags'} = [] unless defined $params{'auto_tags'};
$params{'extract_tags_from_url'} = 0 unless defined $params{'extract_tags_from_url'};
$params{'tag_categories'} = 0 unless defined $params{'tag_categories'};
$params{'ignored_tags'} = [] unless defined $params{'ignored_tags'};
my $items = get_feed_items( $feed, %params );
my $dbh = connect_to_db( $params{'db_file'} );
foreach my $item ( @$items ){
# check to see if it exists already
my $sth = $dbh->prepare("SELECT guid FROM feeds WHERE guid == ? LIMIT 1") or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $item->{'guid'} ) or die "Can't execute statement: $DBI::errstr";
my $row = $sth->fetch();
# strip junk
map { $item->{$_} =~ s/^\s+|\s+$//g } keys %$item;
map { $item->{$_} =~ s/^\n+|\n+$//g } keys %$item;
# and if not, insert it
unless( defined $row ){
$sth = $dbh->prepare(
"INSERT INTO feeds( guid, feed_id, title, link, hashtags, posted, timestamp ) VALUES( ?, ?, ?, ?, ?, ?, ?)"
) or die "Can't prepare statement: $DBI::errstr";
$sth->execute(
$item->{'guid'},
$params{'feed_id'},
$item->{'title'},
$item->{'link'},
join( ' ', @{$item->{'hashtags'}} ),
0,
time,
) or die "Can't execute statement: $DBI::errstr";
}
# decode uft8 strings before storing in the db
map { utf8::decode($item->{'title'}) } keys %$item;
# check to see if it exists already
my $sth = $dbh->prepare("SELECT guid FROM feeds WHERE guid == ? LIMIT 1") or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $item->{'guid'} ) or die "Can't execute statement: $DBI::errstr";
my $row = $sth->fetch();
# and if not, insert it
unless( defined $row ){
$sth = $dbh->prepare(
"INSERT INTO feeds( guid, feed_id, title, link, image, image_title, hashtags, posted, timestamp ) VALUES( ?, ?, ?, ?, ?, ?, ?, ?, ? )"
) or die "Can't prepare statement: $DBI::errstr";
$sth->execute(
$item->{'guid'},
$params{'feed_id'},
$item->{'title'},
$item->{'link'},
$item->{'image'},
$item->{'image_title'},
join( ' ', @{$item->{'hashtags'}} ),
0,
time,
) or die "Can't execute statement: $DBI::errstr";
}
}
$dbh->disconnect();
}
sub connect_to_db {
my ( $db_file ) = @_;
my $dbh = DBI->connect("dbi:SQLite:dbname=$db_file", '', '', { RaiseError => 1 } ) or die $DBI::errstr;
my $dbh = DBI->connect("dbi:SQLite:dbname=$db_file", '', '', { RaiseError => 1, sqlite_unicode => 1 } ) or die $DBI::errstr;
return $dbh;
}
@@ -198,20 +227,26 @@ sub connect_to_db {
sub get_feed_items {
my ( $feed, %params ) = @_;
my @items = ();
my $list = decode_feed( $feed );
my $list = decode_feed( $feed );
$params{'auto_tags'} = 0 unless defined $params{'auto_tags'};
$params{'auto_tags'} = [] unless defined $params{'auto_tags'};
$params{'extract_tags_from_url'} = 0 unless defined $params{'extract_tags_from_url'};
$params{'tag_categories'} = 0 unless defined $params{'tag_categories'};
$params{'ignored_tags'} = [] unless defined $params{'ignored_tags'};
foreach my $item ( @$list ){
my $link = $item->{'link'};
my $title =
# no link, no go
next unless defined $link and ref $link ne 'HASH' and ref $link ne 'ARRAY';
my @hashtags = ();
my $guid = '';
my $guid = undef;
my $image = '';
my $image_title = '';
# strip trailing /
$link =~ s/\/+$//;
$link =~ s/\/+$// if defined $link;
# add user-specified tags
push( @hashtags, @{$params{'auto_tags'}} ) if defined $params{'auto_tags'};
@@ -233,18 +268,18 @@ sub get_feed_items {
# try to guess tags from the title
if( $params{'extract_tags_from_title'} ){
my $title = $item->{'title'};
my $title = $item->{'title'};
# strip apostrophes
$title =~ s/'//g;
# strip apostrophes
$title =~ s/('|`)//g;
# split up string on non-alphanumerics
my @parts = split( /([^(\p{Letter}|\p{Number})]|\p{Punctuation})/, $title );
my @tags = ();
my @tags = ();
foreach my $part ( @parts ){
push( @tags, $part ) unless $part =~ m/^(\s+)?$/;
}
foreach my $part ( @parts ){
push( @tags, $part ) unless $part =~ m/^(\s+)?$/;
}
push( @hashtags, @tags );
}
@@ -263,76 +298,151 @@ sub get_feed_items {
push ( @hashtags, @categories );
}
@hashtags = sort @hashtags;
# extract image link and hover text from content:encoded if it exists
if( defined $item->{'content:encoded'} ){
$item->{'content:encoded'} =~ /img .* ?src=\\?'(https?:\/\/[^']+)/ unless $item->{'content:encoded'} =~ /img .* ?src=\\?"(https?:\/\/[^"]+)/;
if( defined $item->{'guid'} ){
if( ref $item->{'guid'} eq 'HASH' and defined $item->{'guid'}->{'content'} ){
$guid = $item->{'guid'}->{'content'};
}
else {
$guid = $item->{'guid'};
}
if( defined $1 ){
$image = $1;
$item->{'content:encoded'} =~ / title='([^']+)/ unless $item->{'content:encoded'} =~ / title="([^"]+)/;
$image_title = $1 if defined $1;
}
}
# extract image link and hover text from description if it exists
if( not length $image and defined $item->{'description'} ){
$item->{'description'} =~ /img .* ?src='(https?:\/\/[^']+)/ unless $item->{'description'} =~ /img .* ?src="(https?:\/\/[^"]+)/;
if( defined $1 ){
$image = $1;
$item->{'description'} =~ / title='([^']+)/ unless $item->{'description'} =~ / title="([^"]+)/;
$image_title = $1 if defined $1;
}
}
# extract the image link from the enclosure tag if it exists
if( not length $image and defined $item->{'enclosure'} and defined $item->{'enclosure'}->{'type'} and $item->{'enclosure'}->{'type'} =~ /^image\// ){
$image = $item->{'enclosure'}->{'url'} if defined $item->{'enclosure'}->{'url'};
}
# remove any query params from image link
$image =~ s/(\?.*)$//;
@hashtags = sort @hashtags;
if( defined $item->{'guid'} ){
if( ref $item->{'guid'} eq 'HASH' and defined $item->{'guid'}->{'content'} ){
$guid = $item->{'guid'}->{'content'};
}
elsif( defined $item->{'id'} ){
$guid = $item->{'id'};
elsif( ref $item->{'guid'} ne 'HASH' ) {
$guid = $item->{'guid'};
}
my $obj = {
guid => $guid,
title => $item->{'title'},
link => $link,
hashtags => hashtagify( \@hashtags ),
};
$items[@items] = $obj;
}
elsif( defined $item->{'id'} ){
$guid = $item->{'id'};
}
else { $guid = $link }
return \@items;
@hashtags = @{ hashtagify( \@hashtags ) };
# filter out ignored tags
for( my $t = 0; $t < @hashtags; $t++ ){
foreach my $ignored ( @{$params{'ignored_tags'}} ){
splice( @hashtags, $t, 1 ) if $hashtags[$t] eq $ignored;
}
}
my $obj = {
guid => $guid,
title => $item->{'title'},
link => $link,
image => $image,
image_title => $image_title,
hashtags => \@hashtags,
};
$items[@items] = $obj;
}
# the last shall be first and the first shall be last
my @reversed = ();
for( my $i = $#items; $i >= 0; $i-- ){
$reversed[@reversed] = $items[$i];
}
return \@reversed;
}
# extract the data we need based on feed type (RSS v. Atom)
sub decode_feed{
my ( $feed ) = @_;
my @list = ();
my ( $feed ) = @_;
my @list = ();
# RSS
if( defined $feed->{'channel'} and defined $feed->{'channel'}->{'item'} and ref $feed->{'channel'}->{'item'} eq 'ARRAY' ){
@list = @{$feed->{'channel'}->{'item'}};
}
# Atom
elsif( defined $feed->{'entry'} and ref $feed->{'entry'} eq 'HASH' ){
my $entries = $feed->{'entry'};
# RSS
if( defined $feed->{'channel'} and defined $feed->{'channel'}->{'item'} ){
if( ref $feed->{'channel'}->{'item'} eq 'ARRAY' ){
@list = @{$feed->{'channel'}->{'item'}};
}
elsif( ref $feed->{'channel'}->{'item'} eq 'HASH' ){
if( length( keys %{$feed->{'channel'}->{'item'}} ) == 1 ){
$list[@list] = $feed->{'channel'}->{'item'}
}
else{
@list = values %{$feed->{'channel'}->{'item'}};
}
}
}
elsif( defined $feed->{'item'} ){
if( ref $feed->{'item'} eq 'ARRAY' ){
@list = @{$feed->{'item'}};
}
elsif( ref $feed->{'item'} eq 'HASH' ){
@list = values %{$feed->{'item'}};
}
}
# Atom
elsif( defined $feed->{'entry'} and ref $feed->{'entry'} eq 'HASH' ){
my $entries = $feed->{'entry'};
foreach my $guid ( keys %$entries ){
my $item = {
guid => $guid,
};
foreach my $guid ( keys %$entries ){
my $item = {
guid => $guid,
};
if( defined $entries->{$guid}->{'title'} ){
if( ref $entries->{$guid}->{'title'} eq 'HASH' and defined $entries->{$guid}->{'title'}->{'content'} ){
$item->{'title'} = $entries->{$guid}->{'title'}->{'content'};
}
elsif( ref $entries->{$guid}->{'title'} eq '' ){
$item->{'title'} = $entries->{$guid}->{'title'};
}
}
if( defined $entries->{$guid}->{'title'} ){
if( ref $entries->{$guid}->{'title'} eq 'HASH' and defined $entries->{$guid}->{'title'}->{'content'} ){
$item->{'title'} = $entries->{$guid}->{'title'}->{'content'};
}
elsif( ref $entries->{$guid}->{'title'} eq '' ){
$item->{'title'} = $entries->{$guid}->{'title'};
}
}
if( defined $entries->{$guid}->{'link'} ){
if( ref $entries->{$guid}->{'link'} eq 'HASH' and defined $entries->{$guid}->{'link'}->{'href'} ){
$item->{'link'} = $entries->{$guid}->{'link'}->{'href'};
}
elsif( ref $entries->{$guid}->{'link'} eq '' ){
$item->{'link'} = $entries->{$guid}->{'link'};
}
}
if( defined $entries->{$guid}->{'link'} ){
if( ref $entries->{$guid}->{'link'} eq 'HASH' and defined $entries->{$guid}->{'link'}->{'href'} ){
$item->{'link'} = $entries->{$guid}->{'link'}->{'href'};
}
elsif( ref $entries->{$guid}->{'link'} eq '' ){
$item->{'link'} = $entries->{$guid}->{'link'};
}
}
$item->{'category'} = $entries->{'category'} if defined $entries->{'category'};
$item->{'category'} = $entries->{'category'} if defined $entries->{'category'};
push( @list, $item ) if defined $item->{'link'} and defined $item->{'title'};
}
}
if( defined $entries->{$guid}->{'summary'} ){
if( ref($entries->{$guid}->{'summary'}) eq 'HASH' and defined $entries->{$guid}->{'summary'}->{'content'} ){
$item->{'description'} = $entries->{$guid}->{'summary'}->{'content'};
}
else {
$item->{'description'} = $entries->{$guid}->{'summary'};
}
}
return \@list;
push( @list, $item ) if defined $item->{'link'} and defined $item->{'title'};
}
}
return \@list;
}
# fetch the feed and convert the XML to an data object
@@ -363,8 +473,9 @@ sub hashtagify {
$item =~ s/[^(\p{Letter}|\p{Number})]//g;
# drop stop words
# TODO : make these overridable
next if length( $item ) < 3;
next if lc( $item ) =~ m/^(and|are|but|for|from|how|its|the|this)$/;
next if lc( $item ) =~ m/^(a(lso|nd|ny|re)|been|but|can(not|t)?|e(ach|tc|very)|for|from|g(e|o)t|ha(d|ve)|has(nt)?|hers?|hi(m|s)|how|its|no(r|t)|ours?|she|some|th(an|at|em?|eirs?|(e|o)se|ey|eyre|is)|too|very|was|wh(at|en|o)|with|you(r|rs)?)$/;
# hashtagify it
$item = '#' . $item;
# use a hash here instead of an ordered list for auto-dedupe
@@ -388,12 +499,20 @@ sub publish_post {
# initialize an empty cookie jar
$ua->cookie_jar( {} );
# allow option for insecure certs
if ($params{'insecure'}){
$ua->ssl_opts( verify_hostname => 0);
}
# log in
my $login_response = login( $ua, $params{'pod_url'}, $params{'username'}, $params{'password'} ) ;
# if we've logged in successfully, post the message
if( $login_response->is_success ){
my $post = post_message( $ua, $params{'pod_url'}, $content, $params{'aspect_ids'} );
# encode utf-8 characters
utf8::encode($content);
my $post = post_message( $ua, $params{'pod_url'}, $content, $params{'aspect_ids'}, %params );
#logout( $ua, $pod_url );
return $post;
}
@@ -465,23 +584,23 @@ sub extract_token {
# make any necessary string manipulations to play nice with markdown
sub format_content {
my ( $content ) = @_;
my ( $content, %params ) = @_;
$content =~ s/\n/\n\n/g;
$content .= "\nposted by [pod_feeder](https://github.com/rev138/pod_feeder)";
$content .= "\nposted by [pod_feeder](https://github.com/rev138/pod_feeder)" unless( $params{'no_branding'} );
return $content;
}
# post a message
sub post_message {
my ( $ua, $base_url, $content, $aspect_ids ) = @_;
my ( $ua, $base_url, $content, $aspect_ids, %params ) = @_;
my ( $get_stream, $result ) = get_page( $ua, "$base_url/stream" );
if( $get_stream ){
my $csrf = extract_token( $result );
my $post_url = "$base_url/status_messages";
my $message = { status_message => { text => format_content( $content ), provider_display_name => 'pod_feeder' }, aspect_ids => $aspect_ids };
my $message = { status_message => { text => format_content( $content, %params ), provider_display_name => $params{'via'} }, aspect_ids => $aspect_ids };
my $json = JSON->new->allow_nonref;
$json = $json->utf8(0) unless utf8::is_utf8( $message );
@@ -496,38 +615,43 @@ sub post_message {
# create a new sqlite db file with a 'feeds' table if it does not exist already
sub init_database {
my ( $db_file ) = @_;
my ( $db_file ) = @_;
unless( -e $db_file ){
my $dbh = connect_to_db( $db_file );
my $sth = $dbh->prepare(
'CREATE TABLE feeds(guid VARCHAR(255) PRIMARY KEY,feed_id VARCHAR(127),title VARCHAR(255),link VARCHAR(255),hashtags VARCHAR(255),timestamp INTEGER(10),posted INTEGER(1))'
) or die "Can't prepare statement: $DBI::errstr";
unless( -e $db_file ){
my $dbh = connect_to_db( $db_file );
my $sth = $dbh->prepare(
'CREATE TABLE feeds(guid VARCHAR(255) PRIMARY KEY,feed_id VARCHAR(127),title VARCHAR(255),link VARCHAR(255),image VARCHAR(255),image_title VARCHAR(255),hashtags VARCHAR(255),timestamp INTEGER(10),posted INTEGER(1))'
) or die "Can't prepare statement: $DBI::errstr";
$sth->execute() or die "Can't execute statement: $DBI::errstr";
$dbh->disconnect();
}
$sth->execute() or die "Can't execute statement: $DBI::errstr";
$dbh->disconnect();
}
}
sub usage {
print "$0\n";
print "usage:\n";
print " -a --aspect-id <id> Aspects to share with. May specify multiple times (default: 'public')\n";
print " -b --embed-image Embed an image in the post if a link exists (default: off)\n";
print " -c --category-tags Attempt to automatically hashtagify RSS item 'categories' (default: off)\n";
print " -d --database <sqlite file> The SQLite file to store feed data (default: 'feed.db')\n";
print " -e --title-tags Automatically hashtagify RSS item title\n";
print " -e --title-tags Automatically hashtagify RSS item title\n";
print " -f --feed-url <http://...> The feed URL\n";
print " -g --user-agent <string> Use this to spoof the user-agent if the feed blocks bots (ex: 'Mozilla/5.0')\n";
print " -i --feed-id <string> An arbitrary identifier to associate database entries with this feed\n";
print " -j --no-branding Do not include 'posted via pod_feeder' footer to posts\n";
print " -l --pod-url <https://...> The pod URL\n";
print " -m --timeout <hours> How long (in hours) to keep attempting failed posts (default 72)\n";
print " -n --ignore-tag <#hashtag> Hashtags to filter out. May be specified multiple times (default: none)\n";
print " -o --fetch-only Don't publish to Diaspora, just queue the new feed items for later\n";
print " -p --password <********> The D* user password\n";
print " -r --url-tags Attempt to automatically hashtagify the RSS link URL (default: off)\n";
print " -t --auto-tag <#hashtag> Hashtags to add to all posts. May be specified multiple times (default: none)\n";
print " -s --insecure Allows the option to bypass any errors caused from self-signed certificates(default: off)\n";
print " -u --username <user> The D* login username\n";
print " -v --via <string> Sets the 'posted via' text (default: 'pod_feeder')\n";
print " -w --post-raw-link Post the raw link instead of hyperlinking the article title (default: off)\n";
print " -x --limit <n> Only post n items per script run, to prevent post-spamming (default: no limit)\n";
print " -x --limit <n> Only post n items per script run, to prevent post-spamming (default: no limit)\n";
print "\n";
exit;
Executable
+434
View File
@@ -0,0 +1,434 @@
#!/usr/bin/perl
##
## pod_tweeter.pl
##
## A script to auto-post Twitter feeds to a Diaspora account
##
## created 20150416 by Brian Ó <brian@hzsogood.net>
## (on diaspora: brian@diaspora.hzsogood.net)
## https://github.com/rev138/pod_feeder
##
## I owe a great debt to the code of diaspora-rss-bot (https://github.com/spkdev/diaspora-rss-bot)
## for helping me understand how play nice with CSRF tokens et al
##
use strict;
use warnings;
use utf8;
use LWP::UserAgent;
use URI::Escape;
use HTML::Entities;
use JSON;
use DBI;
use Unicode::Normalize 'normalize';
use Getopt::Long;
use Net::Twitter::Lite::WithAPIv1_1;
use DateTime;
my $opts = {
'database' => './pod_tweeter.db',
'limit' => 0,
'timeout' => 72, # hours
};
my @auto_tags = ();
my @aspect_ids = ();
GetOptions(
$opts,
'access-token|k=s',
'access-token-secret|e=s',
'aspect-id|a=s' => \@aspect_ids,
'auto-tag|t=s' => \@auto_tags,
'consumer-key|c=s',
'consumer-secret|r=s',
'database|d=s',
'timeline-id|i=s',
'fetch-only|o',
'help|h', => \&usage,
'limit|x=i',
'password|p=s',
'pod-url|l=s',
'screen-name|s=s',
'timeout|m=i',
'username|u=s',
);
# defaults to 'public' if no aspect ids are specified
$aspect_ids[@aspect_ids] = 'public' unless @aspect_ids;
# initialize the database if it does not exist
eval { init_database( $opts->{'database'} ) };
die "ERROR: Coult not initialize the database: $@" if $@;
eval {
my $last_id = get_last_id( $opts->{'database'} );
my %params = (
access_token => $opts->{'access-token'},
access_token_secret => $opts->{'access-token-secret'},
consumer_key => $opts->{'consumer-key'},
consumer_secret => $opts->{'consumer-secret'},
);
# limit the search to tweets since the last fetched, if we've fetched any
$params{'since_id'} = $last_id if defined $last_id;
# limit the results to the 10 most recent if we haven't fetched any yet
$params{'count'} = 10 unless defined $last_id;
# get the specified user's tweets
$params{'screen_name'} = $opts->{'screen-name'} if defined $opts->{'screen-name'};
# get the tweets
my $tweets = get_tweets( %params );
# update the database
update_tweets(
$tweets,
db_file => $opts->{'database'},
timeline_id => $opts->{'timeline-id'},
auto_tags => hashtagify( \@auto_tags ),
);
};
warn "$@" if $@;
eval {
# publish new feed items to the pod, unless the user specified --fetch-only
publish_feed_items(
db_file => $opts->{'database'},
timeline_id => $opts->{'timeline-id'},
timeout => $opts->{'timeout'},
pod_url => $opts->{'pod-url'},
username => $opts->{'username'},
password => $opts->{'password'},
aspect_ids => \@aspect_ids,
limit => $opts->{'limit'},
) unless $opts->{'fetch-only'};
};
warn "$@" if $@;
# get the id of the most recently fetched tweet, if there is one in the DB
sub get_last_id {
my ( $db_file ) = @_;
my $query_string = "SELECT id FROM tweets ORDER BY timestamp DESC LIMIT 1";
my $dbh = connect_to_db( $db_file );
my $sth = $dbh->prepare( $query_string ) or die "Can't prepare statement: $DBI::errstr";
$sth->execute() or die "Can't execute statement: $DBI::errstr";
my $result = $sth->fetchrow_hashref;
return $result->{'id'} if keys %$result || undef;
}
# get the most recent tweets, within limits
sub get_tweets {
my ( %params ) = @_;
my $query = { exclude_replies => 1 };
my $twit = Net::Twitter::Lite::WithAPIv1_1->new(
access_token => $params{'access_token'},
access_token_secret => $params{'access_token_secret'},
consumer_key => $params{'consumer_key'},
consumer_secret => $params{'consumer_secret'},
user_agent => 'pod_tweeter',
ssl => 1,
);
my $tweets = undef;
$query->{'since_id'} = $params{'since_id'} if defined $params{'since_id'};
$query->{'count'} = $params{'count'} if defined $params{'count'};
$query->{'screen_name'} = $params{'screen_name'} if defined $params{'screen_name'};
eval {
$tweets = $twit->user_timeline( $query );
};
if ( my $err = $@ ) {
die $@ unless blessed $err && $err->isa('Net::Twitter::Lite::Error');
warn "HTTP Response Code: ", $err->code, "\n",
"HTTP Message......: ", $err->message, "\n",
"Twitter error.....: ", $err->error, "\n";
}
return $tweets;
}
# publishes un-posted items in the database
sub publish_feed_items {
my ( %params ) = @_;
my @updates = ();
my $query_string = "SELECT id, timeline_id, text, link, hashtags, posted, timestamp FROM tweets WHERE timeline_id == ? AND posted == 0 AND timestamp > ? ORDER BY timestamp";
my $dbh = connect_to_db( $params{'db_file'} );
# limit the number of items published if limit is specified
$query_string .= " LIMIT $params{'limit'}" if $params{'limit'} > 0;
my $sth = $dbh->prepare( $query_string ) or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $params{'timeline_id'}, time - ( $params{'timeout'} * 3600 ) ) or die "Can't execute statement: $DBI::errstr";
while( my $row = $sth->fetchrow_hashref() ){
push( @updates, $row );
}
foreach my $update ( @updates ){
my $content = '[](' . $update->{'link'} . ')' . $update->{'hashtags'};
print "Publishing $params{'timeline_id'}\t$update->{'id'}\n";
my $post = publish_post( $content, %params );
# mark the item as successfully posted
if( $post->is_success ){
$sth = $dbh->prepare( "UPDATE tweets SET posted = 1 WHERE id = ?" ) or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $update->{'id'} ) or die "Can't execute statement: $DBI::errstr";
}
else {
warn $post->code . ' ' . $post->message;
}
# Now, don't be hasty, master Meriadoc
sleep 1;
}
$dbh->disconnect();
}
# adds new feed items to the database
sub update_tweets {
my ( $tweets, %params ) = @_;
my $dbh = connect_to_db( $params{'db_file'} );
foreach my $tweet ( @$tweets ){
# check to see if it exists already
my $sth = $dbh->prepare("SELECT id FROM tweets WHERE id == ? LIMIT 1") or die "Can't prepare statement: $DBI::errstr";
$sth->execute( $tweet->{'id'} ) or die "Can't execute statement: $DBI::errstr";
my $row = $sth->fetch();
# and if not, insert it
unless( defined $row ){
# extract the hashtags from the tweet
my @hashtags = ();
foreach my $tag( @{$tweet->{'entities'}->{'hashtags'}} ){
push ( @hashtags, $tag->{'text'} );
}
# add user-specified tags
push( @hashtags, @{$params{'auto_tags'}} ) if defined $params{'auto_tags'};
# convert the created date to an epoch timestamp
my $months = { Jan => 1, Feb => 2, Mar => 3, Apr => 4, May => 5, Jun => 6, Jul => 7, Aug => 8, Sep => 9, Oct => 10, Nov => 11, Dec => 12 };
# example: 'created_at' => 'Sat Oct 04 00:47:10 +0000 2014',
$tweet->{'created_at'} =~ m/^[A-Za-z]{3} ([A-Za-z]{3}) ([0-9]{2}) ([0-9]{2}):([0-9]{2}):([0-9]{2}) ([+-][0-9]{4}) ([0-9]{4})$/;
my $dt = DateTime->new(
month => $months->{$1},
day => $2,
hour => $3,
minute => $4,
second => $5,
time_zone => $6,
year => $7,
);
$sth = $dbh->prepare(
"INSERT INTO tweets( id, timeline_id, text, link, hashtags, posted, timestamp ) VALUES( ?, ?, ?, ?, ?, ?, ?)"
) or die "Can't prepare statement: $DBI::errstr";
$sth->execute(
$tweet->{'id'},
$params{'timeline_id'},
$tweet->{'text'},
'https://twitter.com/' . $tweet->{'user'}->{'screen_name'} . '/status/' . $tweet->{'id'},
join( ' ', @{hashtagify(\@hashtags)} ),
0,
$dt->epoch(),
) or die "Can't execute statement: $DBI::errstr";
}
}
$dbh->disconnect();
}
sub connect_to_db {
my ( $db_file ) = @_;
my $dbh = DBI->connect("dbi:SQLite:dbname=$db_file", '', '', { RaiseError => 1 } ) or die $DBI::errstr;
return $dbh;
}
# sanitize and de-dupe tags
sub hashtagify {
my ( $list_ref ) = @_;
my %hashtags = ();
my @list = @$list_ref;
foreach my $item ( @list ){
# remove non-alphanumerics
$item =~ s/[^(\p{Letter}|\p{Number})]//g;
# drop stop words
next if length( $item ) < 3;
next if lc( $item ) =~ m/^(and|are|but|for|from|how|its|the|this)$/;
# hashtagify it
$item = '#' . $item;
# use a hash here instead of an ordered list for auto-dedupe
$hashtags{ lc( $item ) } = undef;
}
my @deduped = keys %hashtags;
my @sorted = sort @deduped;
return \@sorted;
}
# publish a post to the pod
sub publish_post {
my ( $content, %params ) = @_;
my $posted = 0;
# create our user agent
my $ua = LWP::UserAgent->new( requests_redirectable => [ 'GET', 'HEAD', 'POST' ] );
# initialize an empty cookie jar
$ua->cookie_jar( {} );
# log in
my $login_response = login( $ua, $params{'pod_url'}, $params{'username'}, $params{'password'} ) ;
# if we've logged in successfully, post the message
if( $login_response->is_success ){
my $post = post_message( $ua, $params{'pod_url'}, $content, $params{'aspect_ids'} );
#logout( $ua, $pod_url );
return $post;
}
else {
return $login_response;
}
}
# log in to the pod
sub login {
my ( $ua, $base_url, $username, $password ) = @_;
my $sign_in_url = "$base_url/users/sign_in";
$ua->cookie_jar->clear();
my ( $sign_in, $result ) = get_page( $ua, $sign_in_url );
if( $sign_in ){
my $csrf = extract_token( $result );
my $urlencoded_params = '';
my $params = {
$csrf->{'param'} => $csrf->{'token'},
'utf8' => '%E2%9C%93',
'user[username]' => $username,
'user[password]' => $password,
'user[remember_me]' => 1,
'commit' => 'Sign in'
};
foreach my $key ( keys %$params ){
$urlencoded_params .= uri_escape( $key ) . '=' . uri_escape( $params->{$key} ) . '&';
}
return $ua->post( $sign_in_url, 'Content' => $urlencoded_params, 'Content-Type' => 'application/x-www-form-urlencoded' );
}
else {
return $result;
}
}
# retreive a web page via GET
sub get_page {
my ( $ua, $url ) = @_;
my $response = $ua->get( $url );
if( $response->is_success ){
return ( 1, $response->decoded_content );
}
else {
return ( 0, $response );
}
}
# extract the CSRF token from the page's source code
sub extract_token {
my ( $html ) = @_;
my $csrf = {};
# parse out CSRF param and token
$html =~ m/meta name="csrf-param" content="([^"]+)"/;
$csrf->{'param'} = decode_entities( $1 ) if defined ( $1 );
$html =~ m/meta name="csrf-token" content="([^"]+)"/;
$csrf->{'token'} = decode_entities( $1 ) if defined ( $1 );
return $csrf if defined $csrf->{'param'} and defined $csrf->{'token'};
}
# make any necessary string manipulations to play nice with markdown
sub format_content {
my ( $content ) = @_;
$content =~ s/\n/\n\n/g;
$content .= "\nposted by [pod_tweeter](https://github.com/rev138/pod_feeder)";
return $content;
}
# post a message
sub post_message {
my ( $ua, $base_url, $content, $aspect_ids ) = @_;
my ( $get_stream, $result ) = get_page( $ua, "$base_url/stream" );
if( $get_stream ){
my $csrf = extract_token( $result );
my $post_url = "$base_url/status_messages";
my $message = { status_message => { text => format_content( $content ), provider_display_name => 'pod_tweeter' }, aspect_ids => $aspect_ids };
my $json = JSON->new->allow_nonref;
$json = $json->utf8(0) unless utf8::is_utf8( $message );
my $json_message = $json->encode( $message );
return $ua->post( $post_url, 'Content' => $json_message, 'Content-Type' => 'application/json; charset=UTF-8', 'X-CSRF-Token' => $csrf->{'token'} );
}
}
# create a new sqlite db file with a 'feeds' table if it does not exist already
sub init_database {
my ( $db_file ) = @_;
unless( -e $db_file ){
my $dbh = connect_to_db( $db_file );
my $sth = $dbh->prepare(
'CREATE TABLE tweets(id VARCHAR(20) PRIMARY KEY,timeline_id VARCHAR(127),text VARCHAR(140),link VARCHAR(255),hashtags VARCHAR(255),timestamp INTEGER(10),posted INTEGER(1))'
) or die "Can't prepare statement: $DBI::errstr";
$sth->execute() or die "Can't execute statement: $DBI::errstr";
$dbh->disconnect();
}
}
sub usage {
print "$0\n";
print "usage:\n";
print " -a --aspect-id <id> Aspects to share with. May specify multiple times (default: 'public')\n";
print " -c --consumer-key <string> The twitter API consumer key\n";
print " -d --database <sqlite file> The SQLite file to store feed data (default: 'feed.db')\n";
print " -e --access-token-secret <string> The twitter API access token secret\n";
print " -i --timeline-id <string> An arbitrary identifier to associate database entries with this feed\n";
print " -k --access-token <string> The twitter API access token\n";
print " -l --pod-url <https://...> The pod URL\n";
print " -m --timeout <hours> How long (in hours) to keep attempting failed posts (default: 72)\n";
print " -o --fetch-only Don't publish to Diaspora, just queue the new feed items for later\n";
print " -p --password <********> The D* user password\n";
print " -r --consumer-secret <string> The twitter API consumer secret\n";
print " -s --screen-name <\@screenname> The twitter feed to scrape (default: the user associated with the API keys)\n";
print " -t --auto-tag <#hashtag> Hashtags to add to all posts. May be specified multiple times (default: none)\n";
print " -u --username <user> The D* login username\n";
print " -x --limit <n> Only post n items per script run, to prevent post-spamming (default: no limit)\n";
print "\n";
exit;
}