update Database to use default in init
This commit is contained in:
+17
-8
@@ -8,9 +8,9 @@ class Database:
|
||||
def __init__(self, db_path: str = "database.db"):
|
||||
self.columns = "id, title, created_at, view_count, duration, url, thumbnail_url, game_id, game_name, stream_id, creator_name, clip_is, downloaded, uploaded_yt, uploaded_yt_chats, uploaded_yt_shorts"
|
||||
|
||||
file_exists = self.db_path.is_file()
|
||||
file_exists = Path(db_path).is_file()
|
||||
|
||||
self.conn = sqlite3.connect("database.db")
|
||||
self.conn = sqlite3.connect(db_path)
|
||||
self.cursor = self.conn.cursor()
|
||||
|
||||
if not file_exists:
|
||||
@@ -54,10 +54,10 @@ class Database:
|
||||
self.conn.commit()
|
||||
self.conn.close()
|
||||
|
||||
def __mark_as(self, record_id: str, set_sql: str):
|
||||
def __mark_as(self, record_id: str, set_row: str, mark: str = "1" ):
|
||||
self.cursor.execute(
|
||||
f"UPDATE twitch_videos SET {set_sql} = 1 WHERE id = ?",
|
||||
(record_id,))
|
||||
f"UPDATE twitch_videos SET {set_row} = ? WHERE id = ?",
|
||||
(mark, record_id))
|
||||
self.conn.commit()
|
||||
|
||||
def mark_as_uploaded_shorts(self, record_id: str):
|
||||
@@ -76,9 +76,13 @@ class Database:
|
||||
"""Flags a specific row record to downloaded (1)."""
|
||||
self.__mark_as(record_id, "downloaded")
|
||||
|
||||
def __get_unuploaded(self, set_sql: str) -> list[Any]:
|
||||
def unmark_as_download(self, record_id: str):
|
||||
"""Flags a specific row record to downloaded (0)."""
|
||||
self.__mark_as(record_id, "downloaded", "0")
|
||||
|
||||
def __get_unuploaded(self, set_row: str) -> list[Any]:
|
||||
"""Retrieve all rows that were download but not uploaded"""
|
||||
self.cursor.execute(f"SELECT {self.columns} FROM twitch_videos WHERE downloaded = 1 AND {set_sql} = 0")
|
||||
self.cursor.execute(f"SELECT {self.columns} FROM twitch_videos WHERE downloaded = 1 AND {set_row} = 0")
|
||||
return self.cursor.fetchall()
|
||||
|
||||
def get_unuploaded_shorts(self) -> list[Any]:
|
||||
@@ -98,6 +102,11 @@ class Database:
|
||||
self.cursor.execute(f"SELECT {self.columns} FROM twitch_videos WHERE downloaded = 0")
|
||||
return self.cursor.fetchall()
|
||||
|
||||
def get_clips(self) -> list[Any]:
|
||||
"""Retrieves all rows that are clip_is"""
|
||||
self.cursor.execute(f"SELECT {self.columns} FROM twitch_videos WHERE clip_is = 1")
|
||||
return self.cursor.fetchall()
|
||||
|
||||
def insert_video_record(
|
||||
self, id: str, title: str, created_at: str, view_count: int, duration: str,
|
||||
url: str, thumbnail_url: str, game_id: int, game_name: str, stream_id: str,
|
||||
@@ -127,5 +136,5 @@ class Database:
|
||||
self.conn.commit()
|
||||
except Exception as e:
|
||||
# Prevent silent failures if the database connection drops
|
||||
print(f"Database insertion failed: {e}")
|
||||
print(f"❌ Database insertion failed: {e}")
|
||||
self.conn.rollback()
|
||||
|
||||
@@ -3,41 +3,7 @@ import sys
|
||||
import shlex
|
||||
import subprocess
|
||||
|
||||
def run_command_ffmpeg(cmd_str: str, progress_prefix: str = "Progress") -> bool:
|
||||
"""Runs a system command and streams its output live to the console."""
|
||||
# shlex safely handles quotes and paths inside the command string
|
||||
args = shlex.split(cmd_str)
|
||||
|
||||
# Redirect stderr to stdout because FFmpeg outputs status updates to stderr
|
||||
process = subprocess.Popen(
|
||||
args,
|
||||
stdout=subprocess.PIPE,
|
||||
stderr=subprocess.STDOUT,
|
||||
text=True,
|
||||
bufsize=1
|
||||
)
|
||||
|
||||
print(f"Executing: {cmd_str[:90]}...")
|
||||
|
||||
# Stream the output live to the terminal
|
||||
while True:
|
||||
line = process.stdout.readline()
|
||||
if not line and process.poll() is not None:
|
||||
break
|
||||
if line:
|
||||
clean_line = line.strip()
|
||||
# Only print updates that show progress metrics to keep terminal clean
|
||||
if any(metric in clean_line for metric in ["frame=", "time=", "fps=", "Rendering frame"]):
|
||||
sys.stdout.write(f"\r[{progress_prefix}] {clean_line}")
|
||||
sys.stdout.flush()
|
||||
elif "Error" in clean_line or "failed" in clean_line:
|
||||
print(f"\n[Alert] {clean_line}")
|
||||
|
||||
print("\n") # New line after process finishes
|
||||
return process.returncode == 0
|
||||
|
||||
def run_command(cmd_str: str, progress_prefix: str = "Progress", look_for: list = ["frame=", "time=", "fps=", "Rendering frame"]) -> bool:
|
||||
pass
|
||||
"""Runs a system command and streams its output live to the console."""
|
||||
# shlex safely handles quotes and paths inside the command string
|
||||
args = shlex.split(cmd_str)
|
||||
@@ -67,19 +33,4 @@ def run_command(cmd_str: str, progress_prefix: str = "Progress", look_for: list
|
||||
print(f"\n[Alert] {clean_line}")
|
||||
|
||||
print("\n") # New line after process finishes
|
||||
return process.returncode == 0
|
||||
|
||||
def run_command_old(command: str):
|
||||
"""Executes a Linux command, waits for completion, and returns output."""
|
||||
try:
|
||||
# shell=True allows running full command strings with pipes/wildcards
|
||||
# text=True returns strings instead of bytes
|
||||
result = subprocess.run(
|
||||
command, shell=True, check=True, capture_output=True, text=True
|
||||
)
|
||||
|
||||
return {"success": True, "stdout": result.stdout, "stderr": result.stderr}
|
||||
|
||||
except subprocess.CalledProcessError as e:
|
||||
# Handles errors if the Linux command returns a non-zero exit code
|
||||
return {"success": False, "stdout": e.stdout, "stderr": e.stderr}
|
||||
return process.returncode == 0
|
||||
@@ -0,0 +1,35 @@
|
||||
#!/usr/bin/env python3
|
||||
import database
|
||||
|
||||
DB = None
|
||||
|
||||
def redownload_clips():
|
||||
clips = DB.get_clips()
|
||||
|
||||
for clip in clips:
|
||||
# Unpack variables clearly
|
||||
(
|
||||
id,
|
||||
title,
|
||||
created_at,
|
||||
view_count,
|
||||
duration,
|
||||
url,
|
||||
thumbnail_url,
|
||||
game_id,
|
||||
game_name,
|
||||
stream_id,
|
||||
creator_name,
|
||||
clip_is,
|
||||
downloaded,
|
||||
uploaded_yt,
|
||||
uploaded_yt_chats,
|
||||
uploaded_yt_shorts,
|
||||
) = clip
|
||||
|
||||
DB.unmark_as_download(id)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
DB = database.Database()
|
||||
redownload_clips()
|
||||
+4
-2
@@ -18,7 +18,7 @@ def extract_audio(video_path: str):
|
||||
# -map_chapters -1 removes chapter layouts that break the parser.
|
||||
# -sn strips text/subtitle streams that crash MoviePy.
|
||||
# -c copy copies video and audio instantly without quality loss.
|
||||
linux.run_command_ffmpeg(f"ffmpeg -y -i {video_path} -map_chapters -1 -sn -c copy {sanitized_video_path}")
|
||||
linux.run_command(f"ffmpeg -y -i {video_path} -map_chapters -1 -sn -c copy {sanitized_video_path}")
|
||||
|
||||
print("Extracting uncompressed WAV audio...")
|
||||
try:
|
||||
@@ -32,6 +32,8 @@ def extract_audio(video_path: str):
|
||||
)
|
||||
except Exception as e:
|
||||
print(f"❌ Error: {e}")
|
||||
if os.path.exists(audio_temp_path):
|
||||
os.remove(audio_temp_path)
|
||||
finally:
|
||||
# Always clean up the temporary sanitized video on Windows 11
|
||||
if os.path.exists(sanitized_video_path):
|
||||
@@ -57,4 +59,4 @@ def transcribe_to_srt(video_path: str):
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
transcribe_to_srt("download/clips/AbnegateAgitatedGrassPJSalt/AbnegateAgitatedGrassPJSalt.mp4")
|
||||
transcribe_to_srt("download/videos/2813112936/2813112936.mp4")
|
||||
+1
-1
@@ -78,7 +78,7 @@ def upload_video(file_path: str, title: str, category: CategoryId, description:
|
||||
try:
|
||||
# Check if file exists
|
||||
if not os.path.exists(file_path):
|
||||
print(f"Error: File not found: {file_path}")
|
||||
print(f"❌ Error: File not found: {file_path}")
|
||||
return False
|
||||
|
||||
# Load credentials
|
||||
|
||||
+1
-1
@@ -183,7 +183,7 @@ def get_clip_slugs(channel_name):
|
||||
return clip_slugs
|
||||
|
||||
except Exception as e:
|
||||
print(f"An unexpected error occurred: {e}")
|
||||
print(f"❌ An unexpected error occurred: {e}")
|
||||
return []
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
+48
-22
@@ -1,15 +1,18 @@
|
||||
#!/usr/bin/env python3
|
||||
import re
|
||||
import json
|
||||
import nltk
|
||||
import string
|
||||
from collections import Counter
|
||||
from nltk.corpus import stopwords
|
||||
from nltk.tokenize import word_tokenize
|
||||
import nltk
|
||||
|
||||
# Download necessary NLTK data
|
||||
# Download necessary NLTK data modules
|
||||
nltk.download('punkt', quiet=True)
|
||||
nltk.download('stopwords', quiet=True)
|
||||
nltk.download('punkt_tab', quiet=True)
|
||||
nltk.download('averaged_perceptron_tagger', quiet=True) # Required for POS tagging
|
||||
nltk.download('averaged_perceptron_tagger_eng', quiet=True)
|
||||
|
||||
def extract_text_from_srt(file_path: str):
|
||||
with open(file_path, 'r', encoding='utf-8') as file:
|
||||
@@ -19,32 +22,55 @@ def extract_text_from_srt(file_path: str):
|
||||
clean_text = re.sub(r'\d+', '', clean_text)
|
||||
return clean_text
|
||||
|
||||
def get_top_hashtags(srt_file_path: str, top_n: int = 10):
|
||||
raw_text = extract_text_from_srt(srt_file_path)
|
||||
def extract_text_from_json(file_path: str):
|
||||
with open(file_path, "r", encoding="utf-8") as file:
|
||||
data = json.load(file)
|
||||
|
||||
messages = []
|
||||
for comment in data.get("comments", []):
|
||||
message_text = comment.get("message", {}).get("body", "")
|
||||
messages.append(message_text)
|
||||
|
||||
# Lowercase and remove punctuation
|
||||
raw_text = raw_text.lower()
|
||||
raw_text = raw_text.translate(str.maketrans('', '', string.punctuation))
|
||||
# Return a single merged string of all chat text
|
||||
return " ".join(messages)
|
||||
|
||||
def get_top_nouns(file_path: str, top_n: int = 10):
|
||||
# 1. Extract raw text
|
||||
if file_path.endswith(".srt"):
|
||||
raw_text = extract_text_from_srt(file_path)
|
||||
else:
|
||||
raw_text = extract_text_from_json(file_path)
|
||||
|
||||
# 2. Basic cleanup (Keep original case for proper noun accuracy)
|
||||
# Strip basic punctuation but leave words intact
|
||||
clean_text = raw_text.translate(str.maketrans('', '', string.punctuation))
|
||||
|
||||
# Tokenize and remove stopwords
|
||||
words = word_tokenize(raw_text)
|
||||
# 3. Tokenize words
|
||||
words = word_tokenize(clean_text)
|
||||
|
||||
# 4. Part-of-Speech Tagging
|
||||
tagged_words = nltk.pos_tag(words)
|
||||
|
||||
# 5. Filter for Nouns (NN = Singular Noun, NNP = Proper Noun, NNS = Plural Noun)
|
||||
stop_words = set(stopwords.words('english'))
|
||||
nouns = []
|
||||
|
||||
# Filter for alphabetical words longer than 3 characters that aren't stop words
|
||||
filtered_words = [
|
||||
word for word in words
|
||||
if word.isalpha() and word not in stop_words and len(word) > 3
|
||||
]
|
||||
for word, tag in tagged_words:
|
||||
word_lower = word.lower()
|
||||
# Filter out short fragments and standard stopwords
|
||||
if tag in ['NN', 'NNP', 'NNS'] and len(word_lower) > 2 and word_lower not in stop_words:
|
||||
nouns.append(word_lower)
|
||||
|
||||
# 6. Count frequencies
|
||||
noun_counts = Counter(nouns)
|
||||
return [noun for noun, count in noun_counts.most_common(top_n)]
|
||||
|
||||
# Get frequency and create hashtags
|
||||
word_counts = Counter(filtered_words)
|
||||
top_words = word_counts.most_common(top_n)
|
||||
|
||||
hashtags = [f"{word[0]}" for word in top_words]
|
||||
return hashtags
|
||||
|
||||
if __name__ == "__main__":
|
||||
# Example usage
|
||||
# Replace 'your_video.srt' with the path to your file
|
||||
results = get_top_hashtags('your_video.srt', top_n=10)
|
||||
print("Trending Hashtags:", results)
|
||||
results = get_top_nouns('download/videos/2813112936/2813112936.srt', top_n=10)
|
||||
print("Trending Hashtags SRT:", results)
|
||||
|
||||
results = get_top_nouns("download/videos/2813112936/2813112936_chat.json", top_n=10)
|
||||
print("Trending Hashtags JSON:", results)
|
||||
|
||||
Reference in New Issue
Block a user