dosage/scripts/keenspot.py

#!/usr/bin/env python3
# SPDX-License-Identifier: MIT
# Copyright (C) 2004-2008 Tristan Seligmann and Jonathan Jacobs
# Copyright (C) 2012-2014 Bastian Kleineidam
# Copyright (C) 2015-2020 Tobias Gruetzmacher
# Copyright (C) 2019-2020 Daniel Ring
"""
Script to get a list of KeenSpot comics and save the info in a
JSON file for further processing.
"""

from urllib.parse import urlsplit

from scriptutil import ComicListUpdater
from dosagelib.util import check_robotstxt


class KeenSpotUpdater(ComicListUpdater):
    dup_templates = ('Creators/%s', 'GoComics/%s', 'ComicGenesis/%s')

    # names of comics to exclude
    excluded_comics = (
        # non-standard navigation
        'BrawlInTheFamily',
        'Flipside',
        'LastBlood',
        'TheGodChild',
        'Twokinds',
        'Yirmumah',
    )

    extra = {
        'CrowScare': "last='20111031'",
        'Dreamless': "last='20100726'",
        'GeneCatlow': "last='20170412'",
        'MysticRevolution': "path='?cid=%s'",
        'PunchAnPie': "path='daily/%s.html'",
        'ShockwaveDarkside': "path='2d/%s.html'",
    }

    def collect_results(self):
        """Parse the front page."""
        data = self.get_url('http://keenspot.com/')

        for comiclink in data.xpath('//td[@id]/a'):
            comicurl = comiclink.attrib['href']
            name = comiclink.xpath('string()')
            try:
                if '/d/' not in comicurl:
                    check_robotstxt(comicurl + 'd/', self.session)
                else:
                    check_robotstxt(comicurl, self.session)
            except IOError as e:
                print('[%s] INFO: robots.txt denied: %s' % (name, e))
                continue

            self.add_comic(name, comicurl)

    def get_entry(self, name, url):
        sub = urlsplit(url).hostname.split('.', 1)[0]
        if name in self.extra:
            extra = ', ' + self.extra[name]
        else:
            extra = ''
        return u"cls('%s', '%s'%s)," % (name, sub, extra)


if __name__ == '__main__':
    KeenSpotUpdater(__file__).run()
Update file headers The default encoding for source files is UTF-8 since Python 3, so we can drop all encoding headers. While we are at it, just replace them with SPDX headers. 2020-04-18 11:45:44 +00:00			`#!/usr/bin/env python3`
			`# SPDX-License-Identifier: MIT`
Fixup copyright years. 2016-10-28 22:21:41 +00:00			`# Copyright (C) 2004-2008 Tristan Seligmann and Jonathan Jacobs`
Updated copyright. 2014-01-05 15:50:57 +00:00			`# Copyright (C) 2012-2014 Bastian Kleineidam`
Add self to authors list, update copyright headers 2020-01-13 06:34:05 +00:00			`# Copyright (C) 2015-2020 Tobias Gruetzmacher`
			`# Copyright (C) 2019-2020 Daniel Ring`
Add keenspot. 2013-03-11 21:03:17 +00:00			`"""`
			`Script to get a list of KeenSpot comics and save the info in a`
			`JSON file for further processing.`
			`"""`
Clean up update helper scripts. 2016-04-12 22:52:16 +00:00
Drop Python 2 support: six & other imports 2020-02-03 00:03:31 +00:00			`from urllib.parse import urlsplit`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00
			`from scriptutil import ComicListUpdater`
			`from dosagelib.util import check_robotstxt`


			`class KeenSpotUpdater(ComicListUpdater):`
Mark KeenSpot/GeneCatlow completed :( 2019-06-13 02:26:17 +00:00			`dup_templates = ('Creators/%s', 'GoComics/%s', 'ComicGenesis/%s')`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00
			`# names of comics to exclude`
			`excluded_comics = (`
			`# non-standard navigation`
Mark KeenSpot/GeneCatlow completed :( 2019-06-13 02:26:17 +00:00			`'BrawlInTheFamily',`
			`'Flipside',`
			`'LastBlood',`
			`'TheGodChild',`
			`'Twokinds',`
			`'Yirmumah',`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`)`

			`extra = {`
Mark KeenSpot/GeneCatlow completed :( 2019-06-13 02:26:17 +00:00			`'CrowScare': "last='20111031'",`
			`'Dreamless': "last='20100726'",`
			`'GeneCatlow': "last='20170412'",`
			`'MysticRevolution': "path='?cid=%s'",`
			`'PunchAnPie': "path='daily/%s.html'",`
			`'ShockwaveDarkside': "path='2d/%s.html'",`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`}`

			`def collect_results(self):`
			`"""Parse the front page."""`
			`data = self.get_url('http://keenspot.com/')`

			`for comiclink in data.xpath('//td[@id]/a'):`
			`comicurl = comiclink.attrib['href']`
Mark KeenSpot/GeneCatlow completed :( 2019-06-13 02:26:17 +00:00			`name = comiclink.xpath('string()')`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`try:`
Mark KeenSpot/GeneCatlow completed :( 2019-06-13 02:26:17 +00:00			`if '/d/' not in comicurl:`
			`check_robotstxt(comicurl + 'd/', self.session)`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`else:`
			`check_robotstxt(comicurl, self.session)`
			`except IOError as e:`
Mark KeenSpot/GeneCatlow completed :( 2019-06-13 02:26:17 +00:00			`print('[%s] INFO: robots.txt denied: %s' % (name, e))`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`continue`
Add keenspot. 2013-03-11 21:03:17 +00:00
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`self.add_comic(name, comicurl)`
Add keenspot. 2013-03-11 21:03:17 +00:00
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`def get_entry(self, name, url):`
			`sub = urlsplit(url).hostname.split('.', 1)[0]`
			`if name in self.extra:`
			`extra = ', ' + self.extra[name]`
			`else:`
			`extra = ''`
			`return u"cls('%s', '%s'%s)," % (name, sub, extra)`
Add keenspot. 2013-03-11 21:03:17 +00:00

			`if __name__ == '__main__':`
Rework/fix KeenSpot modules. 2016-10-13 22:14:53 +00:00			`KeenSpotUpdater(__file__).run()`