177 lines
6.1 KiB
Python
Raw Normal View History

2016-08-09 16:36:30 +02:00
# -*- coding: utf-8 -*-
2017-03-17 09:42:59 +01:00
# Copyright 2016-2017 Mike Fährmann
2016-08-09 16:36:30 +02:00
#
# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License version 2 as
# published by the Free Software Foundation.
"""Extract images from http://seiga.nicovideo.jp"""
from .common import Extractor, Message
from .. import text, util, exception
2016-08-09 16:36:30 +02:00
from ..cache import cache
2017-02-01 00:53:19 +01:00
2017-01-04 14:20:37 +01:00
class SeigaExtractor(Extractor):
"""Base class for seiga extractors"""
2016-08-09 16:36:30 +02:00
category = "seiga"
cookiedomain = ".nicovideo.jp"
2016-08-09 16:36:30 +02:00
def __init__(self):
Extractor.__init__(self)
2017-12-03 01:38:24 +01:00
self.start_image = 0
2016-08-09 16:36:30 +02:00
def items(self):
2017-01-04 14:20:37 +01:00
self.login()
images = iter(self.get_images())
data = next(images)
2016-08-09 16:36:30 +02:00
yield Message.Version, 1
yield Message.Directory, data
2017-12-03 01:38:24 +01:00
for image in util.advance(images, self.start_image):
2017-01-04 14:20:37 +01:00
data.update(image)
data["extension"] = None
yield Message.Url, self.get_image_url(data["image_id"]), data
2017-01-04 14:20:37 +01:00
def get_images(self):
"""Return iterable containing metadata and images"""
2016-08-09 16:36:30 +02:00
def get_image_url(self, image_id):
"""Get url for an image with id 'image_id'"""
url = "http://seiga.nicovideo.jp/image/source/{}".format(image_id)
2016-08-09 16:36:30 +02:00
response = self.session.head(url)
2016-08-29 17:02:53 +02:00
if response.status_code == 404:
raise exception.NotFoundError("image")
2016-08-09 16:36:30 +02:00
return response.headers["Location"].replace("/o/", "/priv/", 1)
2017-01-04 14:20:37 +01:00
def login(self):
"""Login and set necessary cookies"""
2017-07-25 14:59:41 +02:00
if not self._check_cookies(("user_session",)):
username, password = self._get_auth_info()
self.session.cookies = self._login_impl(username, password)
2017-01-04 14:20:37 +01:00
@cache(maxage=7*24*60*60, keyarg=1)
2017-01-04 14:20:37 +01:00
def _login_impl(self, username, password):
"""Actual login implementation"""
2017-03-17 09:42:59 +01:00
self.log.info("Logging in as %s", username)
2016-08-09 16:36:30 +02:00
url = "https://account.nicovideo.jp/api/v1/login"
data = {"mail_tel": username, "password": password}
self.request(url, method="POST", data=data)
2016-08-09 16:36:30 +02:00
if "user_session" not in self.session.cookies:
raise exception.AuthenticationError()
del self.session.cookies["nicosid"]
return self.session.cookies
2017-01-04 14:20:37 +01:00
class SeigaUserExtractor(SeigaExtractor):
"""Extractor for images of a user from seiga.nicovideo.jp"""
subcategory = "user"
directory_fmt = ["{category}", "{user[id]}"]
filename_fmt = "{category}_{user[id]}_{image_id}.{extension}"
2017-01-04 14:20:37 +01:00
pattern = [(r"(?:https?://)?(?:www\.|seiga\.)?nicovideo\.jp/"
r"user/illust/(\d+)(?:\?(?:[^&]+&)*sort=([^&#]+))?")]
2017-01-04 14:20:37 +01:00
test = [
("http://seiga.nicovideo.jp/user/illust/39537793", {
"pattern": r"https://lohas\.nicoseiga\.jp/priv/[0-9a-f]+/\d+/\d+",
"count": 2,
2017-01-04 14:20:37 +01:00
}),
("http://seiga.nicovideo.jp/user/illust/79433", {
"exception": exception.NotFoundError,
2017-01-04 14:20:37 +01:00
}),
(("http://seiga.nicovideo.jp/user/illust/39537793"
"?sort=image_view&target=illust_all"), None),
2017-01-04 14:20:37 +01:00
]
def __init__(self, match):
SeigaExtractor.__init__(self)
self.user_id, self.order = match.groups()
2017-12-03 01:38:24 +01:00
self.start_page = 1
def skip(self, num):
pages, images = divmod(num, 40)
2017-12-03 01:38:24 +01:00
self.start_page += pages
self.start_image += images
return num
def get_metadata(self, page):
"""Collect metadata from 'page'"""
data = text.extract_all(page, (
("name" , '<img alt="', '"'),
("msg" , '<li class="user_message">', '</li>'),
(None , '<span class="target_name">すべて</span>', ''),
("count", '<span class="count ">', '</span>'),
))[0]
if not data["name"] and "ユーザー情報が取得出来ませんでした" in page:
raise exception.NotFoundError("user")
return {
"user": {
"id": util.safe_int(self.user_id),
"name": data["name"],
"message": (data["msg"] or "").strip(),
},
"count": util.safe_int(data["count"]),
}
2017-01-04 14:20:37 +01:00
def get_images(self):
url = "http://seiga.nicovideo.jp/user/illust/" + self.user_id
2017-12-03 01:38:24 +01:00
params = {"sort": self.order, "page": self.start_page,
"target": "illust_all"}
while True:
cnt = 0
page = self.request(url, params=params).text
2017-12-03 01:38:24 +01:00
if params["page"] == self.start_page:
yield self.get_metadata(page)
for info in text.extract_iter(
page, '<li class="list_item', '</a></li> '):
data = text.extract_all(info, (
("image_id", '/seiga/im', '"'),
("title" , '<li class="title">', '</li>'),
("views" , '</span>', '</li>'),
("comments", '</span>', '</li>'),
("clips" , '</span>', '</li>'),
))[0]
for key in ("image_id", "views", "comments", "clips"):
data[key] = util.safe_int(data[key])
yield data
cnt += 1
if cnt < 40:
return
params["page"] += 1
2017-01-04 14:20:37 +01:00
class SeigaImageExtractor(SeigaExtractor):
"""Extractor for single images from seiga.nicovideo.jp"""
subcategory = "image"
filename_fmt = "{category}_{image_id}.{extension}"
2017-01-04 14:20:37 +01:00
pattern = [(r"(?:https?://)?(?:www\.|seiga\.)?nicovideo\.jp/"
r"(?:seiga/im|image/source/)(\d+)"),
(r"(?:https?://)?lohas\.nicoseiga\.jp/"
r"(?:priv|o)/[^/]+/\d+/(\d+)")]
test = [
("http://seiga.nicovideo.jp/seiga/im5977527", {
"keyword": "f66ba5de33d4ce2cb57f23bb37e1e847e0771c10",
2017-01-04 14:20:37 +01:00
"content": "d9202292012178374d57fb0126f6124387265297",
}),
("http://seiga.nicovideo.jp/seiga/im123", {
"exception": exception.NotFoundError,
}),
]
def __init__(self, match):
SeigaExtractor.__init__(self)
self.image_id = match.group(1)
def skip(self, num):
2017-12-03 01:38:24 +01:00
self.start_image += num
return num
2017-01-04 14:20:37 +01:00
def get_images(self):
return ({}, {"image_id": util.safe_int(self.image_id)})