Logo Questions Linux Laravel Mysql Ubuntu Git Menu
 

Python: Get flattr count

I get a list like this (the numbers are the count of comments)...

  14 http://www.spiegelfechter.com/wordpress/8726/auswege-aus-der-sackgasse
  26 http://www.spiegelfechter.com/wordpress/8722/die-asozialen-hinter-die-asozialen
  77 http://www.spiegelfechter.com/wordpress/8717/in-gesetz-gegossene-verfassungswidrigkeit
  91 http://www.spiegelfechter.com/wordpress/8714/the-same-procedure-as-every-year-europa-lugt-sich-selbst-in-die-tasche
 279 http://www.spiegelfechter.com/wordpress/8709/konstruktionsfehler-des-grundeinkommens

...via...

import urllib2
import re

def main():
    pattern = re.compile('<a href="(.*)#comments".*>(\d+) Kommentare</a>')
    liste = []
    for k in range(2, 3):
        for line in urllib2.urlopen("http://www.spiegelfechter.com/wordpress/page/" + str(k)):
            matcher = pattern.search(line)
            if matcher != None:
                liste.append("%4s" % matcher.group(2) + " " + matcher.group(1))
    for elt in sorted(liste):
        print elt

if __name__ == '__main__':
    main()

Flattr count

I have the 77, but how do I get the 4 in python...? I think the 4 is generated in javascript and I think its hard to handle javascript in python but in this case it might be easy?!

like image 921
qräbnö Avatar asked Sep 25 '26 23:09

qräbnö


2 Answers

This is another solution that uses selenium.

import urllib, selenium.webdriver

driver = selenium.webdriver.Firefox()
driver.get("http://www.spiegelfechter.com/wordpress/page/2")
comments = driver.find_elements_by_xpath('//span[@class="commentbutton"]')
for comment in comments:
    link = comment.find_element_by_xpath('a')
    comment_count = link.text.split()[0]
    url = comment.find_element_by_xpath('iframe').get_attribute('src')
    flattr = urllib.urlopen(url).read()
    flattr = flattr.split('flattr-count"><span>')[1].split('</span>')[0]
    print (flattr, comment_count, link.get_attribute('href'))
driver.close()

output

('1', u'14', u'http://www.spiegelfechter.com/wordpress/8726/auswege-aus-der-sackgasse#comments')
('1', u'26', u'http://www.spiegelfechter.com/wordpress/8722/die-asozialen-hinter-die-asozialen#comments')
('4', u'77', u'http://www.spiegelfechter.com/wordpress/8717/in-gesetz-gegossene-verfassungswidrigkeit#comments')
('1', u'91', u'http://www.spiegelfechter.com/wordpress/8714/the-same-procedure-as-every-year-europa-lugt-sich-selbst-in-die-tasche#comments')
('2', u'279', u'http://www.spiegelfechter.com/wordpress/8709/konstruktionsfehler-des-grundeinkommens#comments')
like image 160
Marwan Alsabbagh Avatar answered Sep 27 '26 12:09

Marwan Alsabbagh


You could use PyQt4's QtWebKit module to inspect the HTML after the JavaScript has been executed. You could then use an HTML parser like lxml.html to scrape the desired information.

For example,

import urllib2
import lxml.html as LH
from PyQt4 import QtGui, QtCore, QtWebKit
import sys

class Render(QtWebKit.QWebPage):
    def __init__(self, url):
        self.app = QtGui.QApplication(sys.argv)
        QtWebKit.QWebPage.__init__(self)
        self.loadFinished.connect(self._loadFinished)
        self.mainFrame().load(QtCore.QUrl(url))
        self.app.exec_()

    def _loadFinished(self, result):
        self.frame = self.mainFrame()
        self.app.quit()

def main():
    liste = []
    for k in range(2, 3):
        url = "http://www.spiegelfechter.com/wordpress/page/" + str(k)
        r = Render(url)
        content = unicode(r.frame.toHtml())
        doc = LH.fromstring(content)
        for span in doc.xpath('//span[@class="commentbutton"]'):
            a = span.xpath('a')[0]
            post = a.attrib['href']
            kommentare = a.text_content()
            # kommentare is expected to be a string such as '14 Kommentare'
            comments = int(kommentare.split()[0])

            iframe = span.xpath('iframe')[0]
            flattr_url = (iframe.attrib['src'])
            flattr_doc = LH.parse(flattr_url)
            span = flattr_doc.xpath('//span[@class="flattr-count"]')[0]
            flattr_count = int(span.text_content())
            liste.append((comments, flattr_count, post))
        for elt in sorted(liste):
            print(elt)

if __name__ == '__main__':
    main()

yields

(14, 1, 'http://www.spiegelfechter.com/wordpress/8726/auswege-aus-der-sackgasse#comments')
(26, 1, 'http://www.spiegelfechter.com/wordpress/8722/die-asozialen-hinter-die-asozialen#comments')
(77, 4, 'http://www.spiegelfechter.com/wordpress/8717/in-gesetz-gegossene-verfassungswidrigkeit#comments')
(91, 1, 'http://www.spiegelfechter.com/wordpress/8714/the-same-procedure-as-every-year-europa-lugt-sich-selbst-in-die-tasche#comments')
(279, 2, 'http://www.spiegelfechter.com/wordpress/8709/konstruktionsfehler-des-grundeinkommens#comments')
like image 24
unutbu Avatar answered Sep 27 '26 11:09

unutbu