diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..f18a128 --- /dev/null +++ b/.gitignore @@ -0,0 +1,168 @@ +# Byte-compiled / optimized / DLL files +__pycache__/ +*.py[cod] +*$py.class + +# C extensions +*.so + +# Distribution / packaging +.Python +build/ +develop-eggs/ +dist/ +downloads/ +eggs/ +.eggs/ +lib/ +lib64/ +parts/ +sdist/ +var/ +wheels/ +share/python-wheels/ +*.egg-info/ +.installed.cfg +*.egg +MANIFEST + +# PyInstaller +# Usually these files are written by a python script from a template +# before PyInstaller builds the exe, so as to inject date/other infos into it. +*.manifest +*.spec + +# Installer logs +pip-log.txt +pip-delete-this-directory.txt + +# Unit test / coverage reports +htmlcov/ +.tox/ +.nox/ +.coverage +.coverage.* +.cache +nosetests.xml +coverage.xml +*.cover +*.py,cover +.hypothesis/ +.pytest_cache/ +cover/ + +# Translations +*.mo +*.pot + +# Django stuff: +*.log +local_settings.py +db.sqlite3 +db.sqlite3-journal + +# Flask stuff: +instance/ +.webassets-cache + +# Scrapy stuff: +.scrapy + +# Sphinx documentation +docs/_build/ + +# PyBuilder +.pybuilder/ +target/ + +# Jupyter Notebook +.ipynb_checkpoints + +# IPython +profile_default/ +ipython_config.py + +# pyenv +# For a library or package, you might want to ignore these files since the code is +# intended to run in multiple environments; otherwise, check them in: +# .python-version + +# pipenv +# According to pypa/pipenv#598, it is recommended to include Pipfile.lock in version control. +# However, in case of collaboration, if having platform-specific dependencies or dependencies +# having no cross-platform support, pipenv may install dependencies that don't work, or not +# install all needed dependencies. +#Pipfile.lock + +# poetry +# Similar to Pipfile.lock, it is generally recommended to include poetry.lock in version control. +# This is especially recommended for binary packages to ensure reproducibility, and is more +# commonly ignored for libraries. +# https://python-poetry.org/docs/basic-usage/#commit-your-poetrylock-file-to-version-control +#poetry.lock + +# pdm +# Similar to Pipfile.lock, it is generally recommended to include pdm.lock in version control. +#pdm.lock +# pdm stores project-wide configurations in .pdm.toml, but it is recommended to not include it +# in version control. +# https://pdm.fming.dev/#use-with-ide +.pdm.toml + +# PEP 582; used by e.g. github.com/David-OConnor/pyflow and github.com/pdm-project/pdm +__pypackages__/ + +# Celery stuff +celerybeat-schedule +celerybeat.pid + +# SageMath parsed files +*.sage.py + +# Environments +.env +.venv +env/ +venv/ +ENV/ +env.bak/ +venv.bak/ + +# Spyder project settings +.spyderproject +.spyproject + +# Rope project settings +.ropeproject + +# mkdocs documentation +/site + +# mypy +.mypy_cache/ +.dmypy.json +dmypy.json + +# Pyre type checker +.pyre/ + +# pytype static type analyzer +.pytype/ + +# Cython debug symbols +cython_debug/ + +# PyCharm +# JetBrains specific template is maintained in a separate JetBrains.gitignore that can +# be found at https://github.com/github/gitignore/blob/main/Global/JetBrains.gitignore +# and can be added to the global gitignore or merged into this file. For a more nuclear +# option (not recommended) you can uncomment the following to ignore the entire idea folder. +.idea/ + + + +# RVM +#tencent_uploader/*.json +#youtube_uploader/*.json +#douyin_uploader/*.json +#bilibili_uploader/*.json \ No newline at end of file diff --git a/README.MD b/README.MD index 8708b69..4a1548a 100644 --- a/README.MD +++ b/README.MD @@ -1,123 +1,69 @@ -# 自动化上传视频到社交媒体 -包含:抖音、视频号、bilibili、小红书等平台 -1. 自动化上传 -2. 视频立即发布 or 定时发布 -3. 自动填写标题、hashtag等 +# social-auto-upload +social-auto-upload 该项目旨在自动化发布视频到各个社交媒体平台 -## install -```pip install -r requeirements.txt -i https://pypi.tuna.tsinghua.edu.cn/simple``` +## 💡Feature +- 中国主流社交媒体平台: + - 抖音 + - 视频号 + - bilibili + - 小红书 -## 核心模块解释 +- 部分国外社交媒体: + - tiktok + - youtube +- 自动化上传(schedule) +- 定时上传(cron) +- cookie 管理 +- 国外平台proxy 设置 +- 多线程上传 +- slack 推送 -### 文件目录结构 -filepath 为本地视频目录,目录包含视频文件、视频meta信息txt文件 +# 💾Installation +``` +pip install -r requirements.txt -i https://pypi.tuna.tsinghua.edu.cn/simple +playwright install Chromium +``` + +# 核心模块解释 + +## 视频文件准备 +filepath 本地视频目录,目录包含 +- 视频文件 +- 视频meta信息txt文件 + +举例: file:2023-08-24_16-29-52 - 这位勇敢的男子为了心爱之人每天坚守 .mp4 meta_file:2023-08-24_16-29-52 - 这位勇敢的男子为了心爱之人每天坚守 .txt meta_file 内容: ```angular2html ----Youtube title--- 这位勇敢的男子为了心爱之人每天坚守 🥺❤️‍🩹 #坚持不懈 #爱情执着 #奋斗使者 #短视频 ``` -视频发布规则: -1. 第一条视频立即发送 -2. 第二条开始 每日11:00, 16:00定时发布 -举例: -filepath 5个视频 -第一个视频立即发布,第二个视频在第二天11点发布,第三个视频在第二天16点发布,第四个视频在第三天11点发布.... - - -### bilibili -1. 设置好视频目录结构 -2. 替换filepath本地目录(视频目录) -3. 获取bilibili 相应cookie: sessdata, bili_jct, buvid3, dedeuserid, ac_time_value(用于刷新cookie, 见图2) -```javascript -// 获取 cookie 中的值 -function getCookie(name) { - const value = "; " + document.cookie; - const parts = value.split("; " + name + "="); - if (parts.length == 2) return parts.pop().split(";").shift(); -} - -// 获取 localStorage 中的值 -function getLocalStorage(name) { - return localStorage.getItem(name); -} - -// 获取所需的值 -const values = { - sessdata: '加密信息, 请通过浏览器获取', - bili_jct: getCookie('bili_jct'), - buvid3: getCookie('buvid3'), - dedeuserid: getCookie('dedeuserid'), - ac_time_value: getLocalStorage('ac_time_value') -}; - -// 输出值 -console.log(values); - -``` - -4. 拉取子仓库 `git submodule add https://github.com/Nemo2011/bilibili-api.git bilibili_uploader/bilibili_api` - -![Alt text](media/e0df568f16d6447c8a66f672ba37af2f.jpg) -![Alt text](media/2023-10-09_105553.png) - -其他部分解释: -```angular2html -random_emoji() 标题后增加随机emoji,避免出错后抛出相同标题不允许上传 -time.sleep(30) -为什么没使用pypi包,因为有个bug,仓库代码修改了但是pypi并没有更新 -分区可以在VideoZoneTypes 文件获取 -``` -todo: -- [ ] tid 分区id指定 -- [ ] 多账号cookie管理 https://nemo2011.github.io/bilibili-api/#/refresh_cookies -- [ ] cookie 失效预警 - -参考项目: -- https://github.com/Nemo2011/bilibili-api - - api 文档: https://nemo2011.github.io/bilibili-api/#/homepage - - 分区 文档:https://biliup.github.io/tid-ref.html -- https://github.com/biliup/biliup-rs - -### 小红书 -1. 目录结构同上 -2. cookie获取,可使用chrome插件:EditThisCookie -- 设置导出格式 -![Alt text](media/20231009111131.png) -- 导出 -![Alt text](media/20231009111214.png) - -其他部分解释: -``` -遇到签名问题,可尝试更新cdn.jsdelivr.net_gh_requireCool_stealth.min.js_stealth.min.js文件 -https://github.com/requireCool/stealth.min.js -参考: -- https://reajason.github.io/xhs/basic -``` -todo: -- [ ] 多账号cookie管理 -- [ ] cookie 失效预警 - -参考项目:https://github.com/ReaJason/xhs - ### 抖音 -1. 目录结构同上 -2. cookie获取:运行程序,检测无cookie,弹出浏览器,登录后,关于。将会保存cookie,文件名为account1.json +使用playwright模拟浏览器行为 +> 抖音前端实现,诸多css class id 均为随机数,故项目中locator多采用相对定位,而非固定定位 +1. 准备视频目录结构 +2. cookie获取:get_douyin_cookie.py 扫码登录 +3. 上传视频:upload_video_to_douyin.py + + 其他部分解释: ``` -laywright.chromium.launch 中的headless可以设置为true,不显示浏览器 +douyin_setup handle 参数为True,为手动获取cookie False 则是校验cookie有效性 + +generate_schedule_time_next_day 默认从第二天开始(此举为避免选择时间的意外错误) +参数解释: +- total_videos 本次上传视频个数 +- videos_per_day 每日上传视频数量 +- daily_times 视频发布时间 默认6、11、14、16、22点 +- start_days 从第N天开始 ``` -todo: -- [ ] cookie 失效预警 -- [ ] 多账号cookie管理 参考项目: - https://github.com/wanghaisheng/tiktoka-studio-uploader @@ -125,45 +71,9 @@ todo: - https://github.com/lishang520/DouYin-Auto-Upload.git -### 视频号 -1. 目录结构同上 -2. cookie获取:运行程序,检测无cookie,弹出浏览器,登录后,关于。将会保存cookie,文件名为account1.json - -其他部分解释: -``` -laywright.chromium.launch 中的headless可以设置为true,不显示浏览器 -chromium 不支持h264编码,需要使用chrome,我这里下载的是Canary -https://www.google.com/intl/en_sg/chrome/canary/ - -这里设置启动的浏览器地址(你chrome的地址) -browser = await playwright.chromium.launch(headless=False, - executable_path="C:/Users/dream/AppData/Local/Google/Chrome SxS/Application/chrome.exe") -``` - -todo: -- [ ] cookie 失效预警 -- [ ] 多账号cookie管理 - - -### tiktok -1. 目录结构同上 -2. cookie获取:sessionid -3. 设置proxy 代理(最好干净代理,避免封号) -4. url_prefix 设置tiktok区,default: us, The request domain. Different countries require different domain configurations. - -proxy = {'socks': 'socks://172.16.22.73:10808'} -url_prefix = "www" -其他部分解释: -``` -Note that you cannot schedule a video more than 10 days in advance. -Note that your TikTok sessionid cookie needs to be updated every 2 months. -``` - - - -todo: -- [ ] cookie 失效预警 -- [ ] 多账号cookie管理 - -参考项目:https://github.com/546200350/TikTokUploder +### 其余部分(todo) +整理后上传 +# 联系我 +探讨自动化上传、自动制作视频 +- 微信:rstar019 diff --git a/conf.py b/conf.py new file mode 100644 index 0000000..e5797dc --- /dev/null +++ b/conf.py @@ -0,0 +1,3 @@ +from pathlib import Path + +BASE_DIR = Path(__file__).parent.resolve() diff --git a/douyin_uploader/__init__.py b/douyin_uploader/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/douyin_uploader/main.py b/douyin_uploader/main.py new file mode 100644 index 0000000..73abdf7 --- /dev/null +++ b/douyin_uploader/main.py @@ -0,0 +1,200 @@ +# -*- coding: utf-8 -*- +import pathlib +from datetime import datetime + +from playwright.async_api import Playwright, async_playwright +import os +import asyncio + + +async def cookie_auth(account_file): + async with async_playwright() as playwright: + browser = await playwright.chromium.launch(headless=True) + context = await browser.new_context(storage_state=account_file) + # 创建一个新的页面 + page = await context.new_page() + # 访问指定的 URL + await page.goto("https://creator.douyin.com/creator-micro/content/upload") + try: + await page.wait_for_selector("div.boards-more h3:text('抖音排行榜')", timeout=5000) # 等待5秒 + print("[+] 等待5秒 cookie 失效") + return False + except: + print("[+] cookie 有效") + return True + + +async def douyin_setup(account_file, handle=False): + if not os.path.exists(account_file) or not await cookie_auth(account_file): + if not handle: + # Todo alert message + return False + print('[+] cookie文件不存在或已失效,即将自动打开浏览器,请扫码登录,登陆后会自动生成cookie文件') + await douyin_cookie_gen(account_file) + return True + + +async def douyin_cookie_gen(account_file): + async with async_playwright() as playwright: + options = { + 'headless': False + } + # Make sure to run headed. + browser = await playwright.chromium.launch(**options) + # Setup context however you like. + context = await browser.new_context() # Pass any options + # Pause the page, and start recording manually. + page = await context.new_page() + await page.goto("https://www.douyin.com/") + await page.pause() + # 点击调试器的继续,保存cookie + await context.storage_state(path=account_file) + + +class DouYinVideo(object): + def __init__(self, title, file_path, tags, publish_date: datetime, account_file): + self.title = title # 视频标题 + self.file_path = file_path + self.tags = tags + self.publish_date = publish_date + self.account_file = account_file + self.date_format = '%Y年%m月%d日 %H:%M' + + async def set_schedule_time_douyin(self, page, publish_date): + # 选择包含特定文本内容的 label 元素 + label_element = page.locator("label.radio--4Gpx6:has-text('定时发布')") + # 在选中的 label 元素下点击 checkbox + await label_element.click() + await asyncio.sleep(1) + publish_date_hour = publish_date.strftime("%Y-%m-%d %H:%M") + + await asyncio.sleep(1) + await page.locator('.semi-input[placeholder="日期和时间"]').click() + await page.keyboard.press("Control+KeyA") + await page.keyboard.type(str(publish_date_hour)) + await page.keyboard.press("Enter") + + await asyncio.sleep(1) + + async def handle_upload_error(self, page): + print("视频出错了,重新上传中") + await page.locator('div.progress-div [class^="upload-btn-input"]').set_input_files(self.file_path) + + async def upload(self, playwright: Playwright) -> None: + # 使用 Chromium 浏览器启动一个浏览器实例 + browser = await playwright.chromium.launch(headless=False) + # 创建一个浏览器上下文,使用指定的 cookie 文件 + context = await browser.new_context(storage_state=f"{self.account_file}") + + # 创建一个新的页面 + page = await context.new_page() + # 访问指定的 URL + await page.goto("https://creator.douyin.com/creator-micro/content/upload") + print('[+]正在上传-------{}.mp4'.format(self.title)) + # 等待页面跳转到指定的 URL,没进入,则自动等待到超时 + print('[-] 正在打开主页...') + await page.wait_for_url("https://creator.douyin.com/creator-micro/content/upload") + # 点击 "上传视频" 按钮 + await page.locator(".upload-btn--9eZLd").set_input_files(self.file_path) + + # 等待页面跳转到指定的 URL + while True: + # 判断是是否进入视频发布页面,没进入,则自动等待到超时 + try: + await page.wait_for_url( + "https://creator.douyin.com/creator-micro/content/publish?enter_from=publish_page") + break + except: + print(" [-] 正在等待进入视频发布页面...") + await asyncio.sleep(0.1) + + # 填充标题和话题 + # 检查是否存在包含输入框的元素 + # 这里为了避免页面变化,故使用相对位置定位:作品标题父级右侧第一个元素的input子元素 + await asyncio.sleep(1) + print(" [-] 正在填充标题和话题...") + title_container = page.get_by_text('作品标题').locator("..").locator("xpath=following-sibling::div[1]").locator("input") + if await title_container.count(): + await title_container.fill(self.title[:30]) + else: + titlecontainer = page.locator(".notranslate") + await titlecontainer.click() + print("clear existing title") + await page.keyboard.press("Backspace") + await page.keyboard.press("Control+KeyA") + await page.keyboard.press("Delete") + print("filling new title") + await page.keyboard.type(self.title) + await page.keyboard.press("Enter") + css_selector = ".zone-container" + for index, tag in enumerate(self.tags, start=1): + print("正在添加第%s个话题" % index) + await page.type(css_selector, "#" + tag) + await page.press(css_selector, "Space") + + while True: + # 判断重新上传按钮是否存在,如果不存在,代表视频正在上传,则等待 + try: + # 新版:定位重新上传 + number = await page.locator('div label+div:has-text("重新上传")').count() + if number > 0: + print(" [-]视频上传完毕") + break + else: + print(" [-] 正在上传视频中...") + await asyncio.sleep(2) + + if await page.locator('div.progress-div > div:has-text("上传失败")').count(): + print(" [-] 发现上传出错了...") + await self.handle_upload_error(page) + except: + print(" [-] 正在上传视频中...") + await asyncio.sleep(2) + + # 更换可见元素 + await page.locator('div.semi-select span:has-text("输入地理位置")').click() + await asyncio.sleep(1) + print("clear existing location") + await page.keyboard.press("Backspace") + await page.keyboard.press("Control+KeyA") + await page.keyboard.press("Delete") + await page.keyboard.type("杭州市") + await asyncio.sleep(1) + await page.locator('div[role="listbox"] [role="option"]').first.click() + + # 頭條 + if await page.locator('text="今日头条" >> xpath=../following-sibling::div/div[contains(@class, "semi-switch")]').count(): + if 'semi-switch-checked' not in await page.eval_on_selector('text="今日头条" >> xpath=../following-sibling::div/div[contains(@class, "semi-switch")]', 'div => div.className'): + await page.locator('text="今日头条" >> xpath=../following-sibling::div//input[contains(@class, "semi-switch-native-control")]').click() + + if self.publish_date != 0: + await self.set_schedule_time_douyin(page, self.publish_date) + + # 判断视频是否发布成功 + while True: + # 判断视频是否发布成功 + try: + publish_button = page.get_by_role('button', name="发布", exact=True) + if await publish_button.count(): + await publish_button.click() + await page.wait_for_url("https://creator.douyin.com/creator-micro/content/manage", + timeout=1500) # 如果自动跳转到作品页面,则代表发布成功 + print(" [-]视频发布成功") + break + except: + print(" [-] 视频正在发布中...") + await page.screenshot(full_page=True) + await asyncio.sleep(0.5) + + await context.storage_state(path=self.account_file) # 保存cookie + print(' [-]cookie更新完毕!') + await asyncio.sleep(2) # 这里延迟是为了方便眼睛直观的观看 + # 关闭浏览器上下文和浏览器实例 + await context.close() + await browser.close() + + async def main(self): + async with async_playwright() as playwright: + await self.upload(playwright) + + diff --git a/examples/__init__.py b/examples/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/examples/get_douyin_cookie.py b/examples/get_douyin_cookie.py new file mode 100644 index 0000000..861fbc4 --- /dev/null +++ b/examples/get_douyin_cookie.py @@ -0,0 +1,9 @@ +import asyncio +from pathlib import Path + +from conf import BASE_DIR +from douyin_uploader.main import douyin_setup + +if __name__ == '__main__': + account_file = Path(BASE_DIR / "douyin_uploader" / "account.json") + cookie_setup = asyncio.run(douyin_setup(str(account_file), handle=True)) diff --git a/examples/upload_video_to_douyin.py b/examples/upload_video_to_douyin.py new file mode 100644 index 0000000..ac9e123 --- /dev/null +++ b/examples/upload_video_to_douyin.py @@ -0,0 +1,26 @@ +import asyncio +from pathlib import Path + +from conf import BASE_DIR +from douyin_uploader.main import douyin_setup, DouYinVideo +from utils.files_times import generate_schedule_time_next_day, get_title_and_hashtags + + +if __name__ == '__main__': + filepath = Path(BASE_DIR) / "videos" + account_file = Path(BASE_DIR / "douyin_uploader" / "account.json") + # 获取视频目录 + folder_path = Path(filepath) + # 获取文件夹中的所有文件 + files = list(folder_path.glob("*.mp4")) + file_num = len(files) + publish_datetimes = generate_schedule_time_next_day(file_num, 1, daily_times=[16]) + cookie_setup = asyncio.run(douyin_setup(account_file, handle=False)) + for index, file in enumerate(files): + title, tags = get_title_and_hashtags(str(file)) + # 打印视频文件名、标题和 hashtag + print(f"视频文件名:{file}") + print(f"标题:{title}") + print(f"Hashtag:{tags}") + app = DouYinVideo(title, file, tags, publish_datetimes[index], account_file) + asyncio.run(app.main(), debug=False) \ No newline at end of file diff --git a/requirements.txt b/requirements.txt new file mode 100644 index 0000000..7b32751 --- /dev/null +++ b/requirements.txt @@ -0,0 +1,10 @@ +requests +playwright +xhs +psycopg2-binary +eventlet +slack_sdk +biliup +schedule +cf_clearance +qrcode \ No newline at end of file diff --git a/utils/__init__.py b/utils/__init__.py new file mode 100644 index 0000000..e69de29 diff --git a/utils/files_times.py b/utils/files_times.py new file mode 100644 index 0000000..a1a6ccc --- /dev/null +++ b/utils/files_times.py @@ -0,0 +1,83 @@ +from datetime import timedelta + +from datetime import datetime +from pathlib import Path + +from conf import BASE_DIR + + +def get_absolute_path(relative_path: str, base_dir: str = None) -> str: + # Convert the relative path to an absolute path + absolute_path = Path(BASE_DIR) / base_dir / relative_path + return str(absolute_path) + + +def get_title_and_hashtags(filename): + """ + 获取视频标题和 hashtag + + Args: + filename: 视频文件名 + + Returns: + 视频标题和 hashtag 列表 + """ + + # 获取视频标题和 hashtag txt 文件名 + txt_filename = filename.replace(".mp4", ".txt") + + # 读取 txt 文件 + with open(txt_filename, "r", encoding="utf-8") as f: + content = f.read() + + # 获取标题和 hashtag + splite_str = content.strip().split("\n") + title = splite_str[0] + hashtags = splite_str[1].replace("#", "").split(" ") + + return title, hashtags + + +def generate_schedule_time_next_day(total_videos, videos_per_day, daily_times=None, timestamps=False, start_days=0): + """ + Generate a schedule for video uploads, starting from the next day. + + Args: + - total_videos: Total number of videos to be uploaded. + - videos_per_day: Number of videos to be uploaded each day. + - daily_times: Optional list of specific times of the day to publish the videos. + - timestamps: Boolean to decide whether to return timestamps or datetime objects. + - start_days: Start from after start_days. + + Returns: + - A list of scheduling times for the videos, either as timestamps or datetime objects. + """ + if videos_per_day <= 0: + raise ValueError("videos_per_day should be a positive integer") + + if daily_times is None: + # Default times to publish videos if not provided + daily_times = [6, 11, 14, 16, 22] + + if videos_per_day > len(daily_times): + raise ValueError("videos_per_day should not exceed the length of daily_times") + + # Generate timestamps + schedule = [] + current_time = datetime.now() + + for video in range(total_videos): + day = video // videos_per_day + start_days + 1 # +1 to start from the next day + daily_video_index = video % videos_per_day + + # Calculate the time for the current video + hour = daily_times[daily_video_index] + time_offset = timedelta(days=day, hours=hour - current_time.hour, minutes=-current_time.minute, + seconds=-current_time.second, microseconds=-current_time.microsecond) + timestamp = current_time + time_offset + + schedule.append(timestamp) + + if timestamps: + schedule = [int(time.timestamp()) for time in schedule] + return schedule diff --git a/videos/demo.mp4 b/videos/demo.mp4 new file mode 100644 index 0000000..c36300f Binary files /dev/null and b/videos/demo.mp4 differ diff --git a/videos/demo.txt b/videos/demo.txt new file mode 100644 index 0000000..94b58ea --- /dev/null +++ b/videos/demo.txt @@ -0,0 +1,2 @@ +这位勇敢的男子为了心爱之人每天坚守 🥺❤️‍🩹 +#坚持不懈 #爱情执着 #奋斗使者 #短视频 \ No newline at end of file