adapter_twilightednet.py - This Python code defines a class…

/fanficdownloader/adapters/adapter_twilightednet.py

https://code.google.com/p/fanficdownloader/ · Python · 253 lines · 150 code · 54 blank · 49 comment · 36 complexity · 21335bc32ca0c26d56d6f22ca54a3ad3 MD5 · raw file

# -*- coding: utf-8 -*-



# Copyright 2011 Fanficdownloader team

#

# Licensed under the Apache License, Version 2.0 (the "License");

# you may not use this file except in compliance with the License.

# You may obtain a copy of the License at

#

#     http://www.apache.org/licenses/LICENSE-2.0

#

# Unless required by applicable law or agreed to in writing, software

# distributed under the License is distributed on an "AS IS" BASIS,

# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.

# See the License for the specific language governing permissions and

# limitations under the License.

#



# Software: eFiction

import time

import logging

logger = logging.getLogger(__name__)

import re

import urllib

import urllib2



from .. import BeautifulSoup as bs

from ..htmlcleanup import stripHTML

from .. import exceptions as exceptions



from base_adapter import BaseSiteAdapter,  makeDate



class TwilightedNetSiteAdapter(BaseSiteAdapter):



    def __init__(self, config, url):

        BaseSiteAdapter.__init__(self, config, url)

        self.story.setMetadata('siteabbrev','tw')

        self.decode = ["Windows-1252",

                       "utf8"] # 1252 is a superset of iso-8859-1.

                               # Most sites that claim to be

                               # iso-8859-1 (and some that claim to be

                               # utf8) are really windows-1252.

        self.username = "NoneGiven" # if left empty, site doesn't return any message at all.

        self.password = ""

        

        # get storyId from url--url validation guarantees query is only sid=1234

        self.story.setMetadata('storyId',self.parsedUrl.query.split('=',)[1])

        

        

        # normalized story URL.

        self._setURL('http://' + self.getSiteDomain() + '/viewstory.php?sid='+self.story.getMetadata('storyId'))



            

    @staticmethod

    def getSiteDomain():

        return 'www.twilighted.net'



    @classmethod

    def getAcceptDomains(cls):

        return ['www.twilighted.net','twilighted.net']



    @classmethod

    def getSiteExampleURLs(cls):

        return "http://www.twilighted.net/viewstory.php?sid=1234"



    def getSiteURLPattern(self):

        return re.escape("http://")+r"(www\.)?"+re.escape("twilighted.net/viewstory.php?sid=")+r"\d+$"



    def needToLoginCheck(self, data):

        if 'Registered Users Only' in data \

                or 'There is no such account on our website' in data \

                or "That password doesn't match the one in our database" in data:

          return True

        else:

          return False



    def performLogin(self, url):

        params = {}



        if self.password:

            params['penname'] = self.username

            params['password'] = self.password

        else:

            params['penname'] = self.getConfig("username")

            params['password'] = self.getConfig("password")

        params['cookiecheck'] = '1'

        params['submit'] = 'Submit'

    

        loginUrl = 'http://' + self.getSiteDomain() + '/user.php?action=login'

        logger.debug("Will now login to URL (%s) as (%s)" % (loginUrl,

                                                              params['penname']))

    

        d = self._fetchUrl(loginUrl, params)

    

        if "Member Account" not in d : #Member Account

            logger.info("Failed to login to URL %s as %s" % (loginUrl,

                                                              params['penname']))

            raise exceptions.FailedToLogin(url,params['penname'])

            return False

        else:

            return True



    def extractChapterUrlsAndMetadata(self):



        url = self.url+'&index=1'

        logger.debug("URL: "+url)



        try:

            data = self._fetchUrl(url)

        except urllib2.HTTPError, e:

            if e.code == 404:

                raise exceptions.StoryDoesNotExist(self.url)

            else:

                raise e



        if self.needToLoginCheck(data):

            # need to log in for this one.

            self.performLogin(url)

            data = self._fetchUrl(url)



        if "Access denied. This story has not been validated by the adminstrators of this site." in data:

            raise exceptions.FailedToDownload(self.getSiteDomain() +" says: Access denied. This story has not been validated by the adminstrators of this site.")

            

        # problems with some stories, but only in calibre.  I suspect

        # issues with different SGML parsers in python.  This is a

        # nasty hack, but it works.

        # twilighted isn't writing <body> ??? wtf?

        data = "<html><body>"+data[data.index("</head>"):] 

        

        # use BeautifulSoup HTML parser to make everything easier to find.

        soup = bs.BeautifulSoup(data)



        ## Title

        a = soup.find('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"$"))

        self.story.setMetadata('title',stripHTML(a))

        

        # Find authorid and URL from... author url.

        a = soup.find('a', href=re.compile(r"viewuser.php"))

        self.story.setMetadata('authorId',a['href'].split('=')[1])

        self.story.setMetadata('authorUrl','http://'+self.host+'/'+a['href'])

        self.story.setMetadata('author',a.string)



        # Find the chapters:

        for chapter in soup.findAll('a', href=re.compile(r'viewstory.php\?sid='+self.story.getMetadata('storyId')+"&chapter=\d+$")):

            # just in case there's tags, like <i> in chapter titles.

            self.chapterUrls.append((stripHTML(chapter),'http://'+self.host+'/'+chapter['href']))



        self.story.setMetadata('numChapters',len(self.chapterUrls))



        def defaultGetattr(d,k):

            try:

                return d[k]

            except:

                return ""

        

        # <span class="label">Rated:</span> NC-17<br /> etc

        labels = soup.findAll('span',{'class':'label'})

        for labelspan in labels:

            value = labelspan.nextSibling

            label = labelspan.string



            if 'Summary' in label:

                ## Everything until the next span class='label'

                svalue = ""

                while not defaultGetattr(value,'class') == 'label':

                    svalue += str(value)

                    value = value.nextSibling

                self.setDescription(url,svalue)



            if 'Rated' in label:

                self.story.setMetadata('rating', value)



            if 'Word count' in label:

                self.story.setMetadata('numWords', value)



            if 'Categories' in label:

                cats = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=categories'))

                catstext = [cat.string for cat in cats]

                for cat in catstext:

                    self.story.addToList('category',cat.string)



            if 'Characters' in label:

                chars = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=characters'))

                charstext = [char.string for char in chars]

                for char in charstext:

                    self.story.addToList('characters',char.string)



            ## twilighted.net doesn't use genre.

            # if 'Genre' in label:

            #     genres = labelspan.parent.findAll('a',href=re.compile(r'browse.php\?type=class'))

            #     genrestext = [genre.string for genre in genres]

            #     self.genre = ', '.join(genrestext)

            #     for genre in genrestext:

            #         self.story.addToList('genre',genre.string)



            if 'Completed' in label:

                if 'Yes' in value:

                    self.story.setMetadata('status', 'Completed')

                else:

                    self.story.setMetadata('status', 'In-Progress')



            if 'Published' in label:

                self.story.setMetadata('datePublished', makeDate(value.strip(), "%B %d, %Y"))

            

            if 'Updated' in label:

                # there's a stray [ at the end.

                #value = value[0:-1]

                self.story.setMetadata('dateUpdated', makeDate(value.strip(), "%B %d, %Y"))



        try:

            # Find Series name from series URL.

            a = soup.find('a', href=re.compile(r"viewseries.php\?seriesid=\d+"))

            series_name = a.string

            series_url = 'http://'+self.host+'/'+a['href']

            

            # use BeautifulSoup HTML parser to make everything easier to find.

            seriessoup = bs.BeautifulSoup(self._fetchUrl(series_url))

            storyas = seriessoup.findAll('a', href=re.compile(r'^viewstory.php\?sid=\d+$'))

            i=1

            for a in storyas:

                if a['href'] == ('viewstory.php?sid='+self.story.getMetadata('storyId')):

                    self.setSeries(series_name, i)

                    self.story.setMetadata('seriesUrl',series_url)

                    break

                i+=1

        

        except:

            # I find it hard to care if the series parsing fails

            pass



    def getChapterText(self, url):



        logger.debug('Getting chapter text from: %s' % url)



        data = self._fetchUrl(url)

        # problems with some stories, but only in calibre.  I suspect

        # issues with different SGML parsers in python.  This is a

        # nasty hack, but it works.

        # twilighted isn't writing <body> ??? wtf?

        data = "<html><body>"+data[data.index("</head>"):] 

        

        soup = bs.BeautifulStoneSoup(data,

                                     selfClosingTags=('br','hr')) # otherwise soup eats the br/hr tags.

        

        span = soup.find('div', {'id' : 'story'})



        if None == span:

            raise exceptions.FailedToDownload("Error downloading Chapter: %s!  Missing required element!" % url)

    

        return self.utf8FromSoup(url,span)



def getClass():

    return TwilightedNetSiteAdapter
Summary ✨

This Python code defines a class TwilightedNetSiteAdapter that adapts data from the Twilighted.net website to a format suitable for e-book platforms. It extracts metadata such as title, author, rating, word count, and chapter text, and stores it in an object’s attributes. The class is designed to work with Calibre, a popular e-book management software.
Tech Fingerprint

Alerts (15)

'def' Ensure functions have docstrings for documentation
54 58 62 68 76 102 149 230 251
Complexity hotspot; lines 69 to 71 (total complexity: 3)
69 70 71
'except:' Avoid catching all exceptions; specify exception types to catch only expected errors
152 226
'try:' Ensure try blocks have corresponding except or finally blocks
209