import json import re from datetime import datetime, timedelta from typing import List, Optional, Tuple import requests from bs4 import BeautifulSoup def _fallback_niconico_user_name(user_id: str) -> str: safe_user_id = re.sub(r"[^A-Za-z0-9_.-]+", "_", str(user_id)).strip("_") or "unknown" return f"niconico_user_{safe_user_id}" def _extract_niconico_owner_name(videos: List[dict]) -> Optional[str]: for video in videos: essential = video.get('essential', {}) if isinstance(video, dict) else {} owner = essential.get('owner', {}) if isinstance(essential, dict) else {} name = owner.get('name') if isinstance(owner, dict) else None if name: return name return None def _page_title(soup: BeautifulSoup) -> str: if soup.title and soup.title.string: return soup.title.string.strip() return "N/A" def get_niconico_user_and_video_info(user_id: str, existing_video_ids: set = None, force_full_scan: bool = False) -> Tuple[Optional[str], List[Tuple[str, str, str]]]: print(f"[DEBUG] get_niconico_user_and_video_info called with user_id={user_id}, existing_count={len(existing_video_ids) if existing_video_ids else 0}, force_full_scan={force_full_scan}") headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36" } all_video_info = [] user_name = None page = 1 found_existing = False first_page_parsed = False while True: # ページ番号付きのURLを構築 if page == 1: url = f"https://www.nicovideo.jp/user/{user_id}/video" else: url = f"https://www.nicovideo.jp/user/{user_id}/video?page={page}" print(f"[DEBUG] Fetching page {page}: {url}") import time start_time = time.time() response = requests.get(url, headers=headers, timeout=30) response.encoding = 'utf-8' print(f"[DEBUG] Request completed in {time.time() - start_time:.2f} seconds, status={response.status_code}") if response.status_code != 200: # 2ページ目以降で404の場合は、そのページが存在しないということなので終了 if page > 1 and response.status_code == 404: print(f"Page {page} does not exist, finishing.") break print(f"Error: User ID {user_id} - Status code {response.status_code}") if page == 1: return None, [] break soup = BeautifulSoup(response.text, 'html.parser') json_data_element = soup.find('div', id='js-initial-userpage-data') if not json_data_element or 'data-initial-data' not in json_data_element.attrs: # データが見つからない場合、ページが存在しない可能性がある if page > 1: print(f"No data found on page {page}, finishing.") break print(f"Error: Could not find initial data for user {user_id}") return None, [] json_data = json_data_element['data-initial-data'] try: data = json.loads(json_data) except json.JSONDecodeError as exc: print(f"Error: Could not parse initial data for user {user_id}: {exc}") if page == 1: return None, [] break if page == 1: first_page_parsed = True videos = [] if 'nvapi' in data: nvapi_entries = data.get('nvapi') or [] nvapi_data = nvapi_entries[0] if isinstance(nvapi_entries, list) and nvapi_entries else {} if not nvapi_data: print(f"[WARN] Niconico user {user_id} page {page}: nvapi entries are missing.") elif 'body' in nvapi_data and 'data' in nvapi_data['body']: items = nvapi_data['body']['data'].get('items', []) videos = items if isinstance(items, list) else [] # 初回のみユーザー名を取得 if videos and user_name is None: user_name = _extract_niconico_owner_name(videos) # ビデオが見つからない場合は終了 if not videos: if page == 1: print(f"No videos found for user {user_id}") print(f"[DEBUG] Niconico user {user_id} page title: {_page_title(soup)}") else: print(f"No more videos found on page {page}, finishing.") break # 動画情報を追加 for video in videos: video_id = video['essential']['id'] video_title = video['essential']['title'] registered_at = video['essential']['registeredAt'] registered_at_jst = datetime.fromisoformat(registered_at.replace('Z', '+00:00')) + timedelta(hours=9) all_video_info.append((video_id, video_title, registered_at_jst.strftime('%Y-%m-%d %H:%M:%S'))) # 既存の動画IDが見つかった場合フラグを立てる(1ページ目のみ) if page == 1 and existing_video_ids and video_id in existing_video_ids: found_existing = True print(f"Found existing video {video_id} on page 1") print(f"Found {len(videos)} videos on page {page}") # 1ページ目で既存動画が見つかった場合、2ページ目以降はスキップ(force_full_scanがFalseの場合のみ) if page == 1 and found_existing and not force_full_scan: print(f"Found existing videos on page 1, skipping remaining pages to save time") break elif page == 1 and found_existing and force_full_scan: print(f"Found existing videos on page 1, but force_full_scan is enabled, continuing to scan all pages") # 100件未満の場合は最後のページ if len(videos) < 100: print(f"Less than 100 videos on page {page}, this is the last page.") break # 次のページへ page += 1 if not user_name and first_page_parsed: user_name = _fallback_niconico_user_name(user_id) if all_video_info: print( f"[INFO] Niconico user {user_id} display name is hidden or unavailable; " f"using fallback username {user_name}." ) else: print( f"[INFO] Niconico user {user_id} page was parsed but no display name was available; " f"using fallback username {user_name}." ) print(f"Total videos found: {len(all_video_info)}") return user_name, all_video_info def get_youtube_channel_and_video_info(channel_id: str) -> Tuple[Optional[str], List[Tuple[str, str, str]]]: url = f"https://www.youtube.com/@{channel_id}/videos" headers = { "User-Agent": "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36" } response = requests.get(url, headers=headers) if response.status_code != 200: print(f"Error: Channel ID {channel_id} - Status code {response.status_code}") return None, [] soup = BeautifulSoup(response.text, 'html.parser') # Extract JSON data pattern = re.compile(r'var ytInitialData = (.+?);', re.DOTALL) matches = pattern.search(response.text) if not matches: print(f"Error: Could not find ytInitialData for channel {channel_id}") return None, [] data = json.loads(matches.group(1)) # Extract channel name channel_name = soup.find('meta', property='og:title')['content'] # Extract video information video_info = [] items = data['contents']['twoColumnBrowseResultsRenderer']['tabs'][1]['tabRenderer']['content']['richGridRenderer']['contents'] current_date = datetime.now().strftime('%Y年%m月%d日') for item in items: if 'richItemRenderer' in item: video_data = item['richItemRenderer']['content']['videoRenderer'] video_id = video_data['videoId'] title = video_data['title']['runs'][0]['text'] upload_date = video_data.get('publishedTimeText', {}).get('simpleText', 'N/A') # 元の表記に現在日時を追記 if upload_date != 'N/A': upload_date = f"{upload_date} ({current_date} 取得)" video_info.append((video_id, title, upload_date)) return channel_name, video_info