Spaces:

flzta
/

data

Paused

App Files Files Community

flzta commited on Mar 27, 2025

Commit

752618a

verified ·

1 Parent(s): efc3e07

Update sync_data.sh

Browse files

Files changed (1) hide show

sync_data.sh +106 -148

sync_data.sh CHANGED Viewed

@@ -1,180 +1,162 @@
 #!/bin/bash
-# 检查 Hugging Face Token 和 Dataset ID 环境变量
 if [[ -z "$HF_TOKEN" ]] || [[ -z "$DATASET_ID" ]]; then
     echo "Starting without backup functionality - missing HF_TOKEN or DATASET_ID"
     exec /opt/cloudreve/cloudreve -c /opt/cloudreve/config.ini
     exit 0
 fi
 # 激活虚拟环境
 source /opt/venv/bin/activate
-# 定义 Cloudreve 主程序目录
-CLOUDREVE_DIR="/opt/cloudreve"
-BACKUP_PREFIX="cloudreve_backup"
-HF_DATA_DIR="/opt/cloudreve/hf_uploaded" # 用于记录已上传到 HF 的文件
-# 创建 HF 上传记录目录
-mkdir -p "$HF_DATA_DIR"
-# Python 函数: 上传文件到 Hugging Face Dataset
-upload_file_to_dataset() {
-    local_file_path="$1"
-    relative_path=$(echo "$local_file_path" | sed "s|$CLOUDREVE_DIR/data/||")
-    token="$HF_TOKEN"
-    repo_id="$DATASET_ID"
-    echo "Preparing to upload file: $local_file_path to Dataset: $repo_id at path: $relative_path"
-    python3 -c "
-from huggingface_hub import HfApi
-import os
-api = HfApi(token='$token')
-try:
-    repo_id_val = os.environ.get('DATASET_ID')
-    print(f'Uploading file: '$local_file_path' to {repo_id_val} as '$relative_path'')
-    api.upload_file(
-        path_or_fileobj='$local_file_path',
-        path_in_repo='$relative_path',
-        repo_id=repo_id_val,
-        repo_type='dataset'
-    )
-    print(f'Successfully uploaded '$relative_path'')
-except Exception as e:
-    print(f'Error uploading file: {str(e)}')
-"
-}
-# Python 函数: 上传备份
 upload_backup() {
     file_path="$1"
     file_name="$2"
     token="$HF_TOKEN"
     repo_id="$DATASET_ID"
-    echo "Preparing to upload backup file: $file_path as $file_name to Dataset: $repo_id"
     python3 -c "
 from huggingface_hub import HfApi
 import sys
 import os
-print(f'HF_TOKEN is set: {os.environ.get(\"HF_TOKEN\") is not None}')
-print(f'DATASET_ID is set: {os.environ.get(\"DATASET_ID\") is not None}')
-def manage_backups(api, repo_id_val, max_files=50):
-    print('Managing old backups...')
-    files = api.list_repo_files(repo_id=repo_id_val, repo_type='dataset')
-    backup_files = [f for f in files if f.startswith('$BACKUP_PREFIX') and f.endswith('.tar.gz')]
     backup_files.sort()
     if len(backup_files) >= max_files:
-        print(f'Found {len(backup_files)} backup files, maximum allowed is {max_files}.')
         files_to_delete = backup_files[:(len(backup_files) - max_files + 1)]
         for file_to_delete in files_to_delete:
             try:
-                print(f'Deleting old backup: {file_to_delete}')
-                api.delete_file(path_in_repo=file_to_delete, repo_id=repo_id_val, repo_type='dataset')
-                print(f'Successfully deleted: {file_to_delete}')
             except Exception as e:
                 print(f'Error deleting {file_to_delete}: {str(e)}')
-    else:
-        print('Number of backup files is within the limit.')
 api = HfApi(token='$token')
 try:
-    repo_id_val = os.environ.get('DATASET_ID') # 从环境变量中获取 repo_id
-    print(f'Uploading file: $file_path to {repo_id_val} as $file_name')
     api.upload_file(
-        path_or_fileobj='$file_path',
-        path_in_repo='$file_name',
-        repo_id=repo_id_val,
         repo_type='dataset'
     )
-    print(f'Successfully uploaded $file_name')
 except Exception as e:
     print(f'Error uploading file: {str(e)}')
 "
 }
-# Python 函数: 下载最新备份
 download_latest_backup() {
-  token="$HF_TOKEN"
-  repo_id="$DATASET_ID"
-  echo "Preparing to download the latest backup from Dataset: $repo_id"
-  python3 -c "
 from huggingface_hub import HfApi
 import sys
 import os
 import tarfile
 import tempfile
-print(f'HF_TOKEN is set: {os.environ.get(\"HF_TOKEN\") is not None}')
-print(f'DATASET_ID is set: {os.environ.get(\"DATASET_ID\") is not None}')
 api = HfApi(token='$token')
 try:
-    repo_id_val = os.environ.get('DATASET_ID') # 从环境变量中获取 repo_id
-    print(f'Listing files in Dataset: {repo_id_val}')
-    files = api.list_repo_files(repo_id=repo_id_val, repo_type='dataset')
-    backup_files = [f for f in files if f.startswith('$BACKUP_PREFIX') and f.endswith('.tar.gz')]
     if not backup_files:
-        print('No backup files found in the Dataset.')
         sys.exit()
     latest_backup = sorted(backup_files)[-1]
-    print(f'Latest backup file found: {latest_backup}')
     with tempfile.TemporaryDirectory() as temp_dir:
-        filepath = api.hf_hub_download(
-            repo_id=repo_id_val,
             filename=latest_backup,
             repo_type='dataset',
             local_dir=temp_dir
         )
-        if filepath and os.path.exists(filepath):
-            print(f'Successfully downloaded backup to temporary directory: {filepath}')
-            print(\"Before restoring backup:\")
-            import subprocess
-            subprocess.run(['ls', '-l', \"$CLOUDREVE_DIR\"], shell=True, check=False)
-            # 删除现有的 Cloudreve 目录和配置文件
-            import shutil
-            cloudreve_path = os.path.join(\"$CLOUDREVE_DIR\", \"cloudreve\")
-            cloudreve_db_path = os.path.join(\"$CLOUDREVE_DIR\", \"cloudreve.db\")
-            config_ini_path = os.path.join(\"$CLOUDREVE_DIR\", \"config.ini\")
-            data_path = os.path.join(\"$CLOUDREVE_DIR\", \"data\")
-            if os.path.exists(cloudreve_path):
-                print(f'Deleting: {cloudreve_path}')
-                shutil.rmtree(cloudreve_path, ignore_errors=True)
-            if os.path.exists(cloudreve_db_path):
-                print(f'Deleting: {cloudreve_db_path}')
-                os.remove(cloudreve_db_path)
-            if os.path.exists(config_ini_path):
-                print(f'Deleting: {config_ini_path}')
-                os.remove(config_ini_path)
-            if os.path.exists(data_path):
-                print(f'Deleting: {data_path}')
-                shutil.rmtree(data_path, ignore_errors=True)
-            print(\"Deletion complete.\")
-            print(f'Extracting backup archive: {filepath} to $CLOUDREVE_DIR')
-            import tarfile
-            with tarfile.open(filepath, 'r:gz') as tar:
-                tar.extractall(\"$CLOUDREVE_DIR\")
             print(f'Successfully restored backup from {latest_backup}')
-            print(\"After restoring backup:\")
-            subprocess.run(['ls', '-l', \"$CLOUDREVE_DIR\"], shell=True, check=False)
-        else:
-            print('Error during file download.')
 except Exception as e:
     print(f'Error downloading backup: {str(e)}')
@@ -182,61 +164,37 @@ except Exception as e:
 }
 # 首次启动时下载最新备份
-echo "Downloading latest backup from HuggingFace..."
 download_latest_backup
 # 同步函数
 sync_data() {
-    echo "SYNC_DATA FUNCTION IS RUNNING"
     while true; do
         echo "Starting sync process at $(date)"
-        # 检查 Cloudreve data 目录并上传新文件
-        if [ -d "$CLOUDREVE_DIR/data" ]; then
-            find "$CLOUDREVE_DIR/data" -type f -print0 | while IFS= read -r -d $'\0' file; do
-                if [ ! -f "$HF_DATA_DIR/$(basename "$file")" ]; then
-                    echo "New file found: $file"
-                    upload_file_to_dataset "$file"
-                    touch "$HF_DATA_DIR/$(basename "$file")" # 创建已上传标记
-                fi
-            done
-        fi
-        if [ -d "$CLOUDREVE_DIR" ]; then
-            echo "Before compression:"
-            ls -l \"$CLOUDREVE_DIR\"
             timestamp=$(date +%Y%m%d_%H%M%S)
-            backup_file="${BACKUP_PREFIX}_${timestamp}.tar.gz"
-            backup_path="/tmp/${backup_file}"
-            echo "Compressing Cloudreve directory (including database and config) to: $backup_path"
-            tar -czf "$backup_path" -C "$CLOUDREVE_DIR" cloudreve cloudreve.db config.ini
-            echo "Compression complete."
-            echo "After compression:"
-            ls -l "$backup_path"
             echo "Uploading backup to HuggingFace..."
-            upload_backup "$backup_path" "${backup_file}"
-            rm -f "$backup_path"
         else
             echo "Cloudreve directory does not exist yet, waiting for next sync..."
         fi
-        SYNC_INTERVAL=${SYNC_INTERVAL:-300} # 默认同步间隔改为 5 分钟
         echo "Next sync in ${SYNC_INTERVAL} seconds..."
         sleep $SYNC_INTERVAL
     done
 }
-# 延迟启动同步脚本，给 Cloudreve 一些启动时间
-sleep 30
 # 后台启动同步进程
 sync_data &
-# 启动 Cloudreve (这里需要启动 Cloudreve)
-echo "Starting Cloudreve..."
 exec /opt/cloudreve/cloudreve -c /opt/cloudreve/config.ini

 #!/bin/bash
+# 检查环境变量
 if [[ -z "$HF_TOKEN" ]] || [[ -z "$DATASET_ID" ]]; then
     echo "Starting without backup functionality - missing HF_TOKEN or DATASET_ID"
     exec /opt/cloudreve/cloudreve -c /opt/cloudreve/config.ini
     exit 0
 fi
+# 设置解密密钥 (请务必设置一个长且随机的字符串)
+ENCRYPTION_KEY=${ENCRYPTION_KEY:-"请在此处设置您的加密密钥，这是一个长且随机的字符串"}
 # 激活虚拟环境
 source /opt/venv/bin/activate
+# 上传备份
 upload_backup() {
     file_path="$1"
     file_name="$2"
     token="$HF_TOKEN"
     repo_id="$DATASET_ID"
+    encryption_key="$ENCRYPTION_KEY"
     python3 -c "
 from huggingface_hub import HfApi
 import sys
 import os
+import base64
+from cryptography.fernet import Fernet
+from cryptography.hazmat.primitives import hashes
+from cryptography.hazmat.primitives.kdf.pbkdf2 import PBKDF2HMAC
+import io
+def generate_key(password, salt=b'cloudreve_salt'):
+    kdf = PBKDF2HMAC(
+        algorithm=hashes.SHA256(),
+        length=32,
+        salt=salt,
+        iterations=100000,
+    )
+    key = base64.urlsafe_b64encode(kdf.derive(password.encode()))
+    return key
+def encrypt_file(file_path, key):
+    f = Fernet(key)
+    with open(file_path, 'rb') as file:
+        file_data = file.read()
+    encrypted_data = f.encrypt(file_data)
+    encrypted_file_path = file_path + '.enc'
+    with open(encrypted_file_path, 'wb') as file:
+        file.write(encrypted_data)
+    return encrypted_file_path
+def manage_backups(api, repo_id, max_files=10):
+    files = api.list_repo_files(repo_id=repo_id, repo_type='dataset')
+    backup_files = [f for f in files if f.startswith('cloudreve_backup_') and f.endswith('.tar.gz.enc')]
     backup_files.sort()
     if len(backup_files) >= max_files:
         files_to_delete = backup_files[:(len(backup_files) - max_files + 1)]
         for file_to_delete in files_to_delete:
             try:
+                api.delete_file(path_in_repo=file_to_delete, repo_id=repo_id, repo_type='dataset')
+                print(f'Deleted old backup: {file_to_delete}')
             except Exception as e:
                 print(f'Error deleting {file_to_delete}: {str(e)}')
 api = HfApi(token='$token')
 try:
+    # 生成加密密钥
+    key = generate_key('$encryption_key')
+    # 加密文件
+    encrypted_file_path = encrypt_file('$file_path', key)
+    # 上传加密文件
     api.upload_file(
+        path_or_fileobj=encrypted_file_path,
+        path_in_repo='$file_name.enc',
+        repo_id='$repo_id',
         repo_type='dataset'
     )
+    print(f'Successfully uploaded encrypted $file_name')
+    # 删除临时加密文件
+    os.remove(encrypted_file_path)
+    # 管理备份文件数量
+    manage_backups(api, '$repo_id')
 except Exception as e:
     print(f'Error uploading file: {str(e)}')
 "
 }
+# 下载最新备份
 download_latest_backup() {
+    token="$HF_TOKEN"
+    repo_id="$DATASET_ID"
+    encryption_key="$ENCRYPTION_KEY"
+    python3 -c "
 from huggingface_hub import HfApi
 import sys
 import os
 import tarfile
 import tempfile
+import base64
+from cryptography.fernet import Fernet
+from cryptography.hazmat.primitives import hashes
+from cryptography.hazmat.primitives.kdf.pbkdf2 import PBKDF2HMAC
+def generate_key(password, salt=b'cloudreve_salt'):
+    kdf = PBKDF2HMAC(
+        algorithm=hashes.SHA256(),
+        length=32,
+        salt=salt,
+        iterations=100000,
+    )
+    key = base64.urlsafe_b64encode(kdf.derive(password.encode()))
+    return key
+def decrypt_file(encrypted_file_path, key):
+    f = Fernet(key)
+    with open(encrypted_file_path, 'rb') as file:
+        encrypted_data = file.read()
+    decrypted_data = f.decrypt(encrypted_data)
+    decrypted_file_path = encrypted_file_path[:-4]  # 移除 .enc 后缀
+    with open(decrypted_file_path, 'wb') as file:
+        file.write(decrypted_data)
+    return decrypted_file_path
 api = HfApi(token='$token')
 try:
+    files = api.list_repo_files(repo_id='$repo_id', repo_type='dataset')
+    backup_files = [f for f in files if f.startswith('cloudreve_backup_') and f.endswith('.tar.gz.enc')]
     if not backup_files:
+        print('No backup files found')
         sys.exit()
     latest_backup = sorted(backup_files)[-1]
     with tempfile.TemporaryDirectory() as temp_dir:
+        # 下载加密的备份文件
+        encrypted_filepath = api.hf_hub_download(
+            repo_id='$repo_id',
             filename=latest_backup,
             repo_type='dataset',
             local_dir=temp_dir
         )
+        if encrypted_filepath and os.path.exists(encrypted_filepath):
+            # 生成解密密钥
+            key = generate_key('$encryption_key')
+            # 解密文件
+            decrypted_filepath = decrypt_file(encrypted_filepath, key)
+            # 解压缩到目标目录
+            with tarfile.open(decrypted_filepath, 'r:gz') as tar:
+                tar.extractall('/opt/cloudreve')
             print(f'Successfully restored backup from {latest_backup}')
+            # 清理临时文件
+            os.remove(decrypted_filepath)
 except Exception as e:
     print(f'Error downloading backup: {str(e)}')
 }
 # 首次启动时下载最新备份
+echo "Checking for latest backup from HuggingFace..."
 download_latest_backup
 # 同步函数
 sync_data() {
     while true; do
         echo "Starting sync process at $(date)"
+        if [ -d /opt/cloudreve ]; then
             timestamp=$(date +%Y%m%d_%H%M%S)
+            backup_file="cloudreve_backup_${timestamp}.tar.gz"
+            # 压缩整个 Cloudreve 目录
+            tar -czf "/tmp/${backup_file}" -C /opt/cloudreve .
             echo "Uploading backup to HuggingFace..."
+            upload_backup "/tmp/${backup_file}" "${backup_file}"
+            rm -f "/tmp/${backup_file}"
         else
             echo "Cloudreve directory does not exist yet, waiting for next sync..."
         fi
+        SYNC_INTERVAL=${SYNC_INTERVAL:-7200} # 默认同步间隔改为 2 小时
         echo "Next sync in ${SYNC_INTERVAL} seconds..."
         sleep $SYNC_INTERVAL
     done
 }
 # 后台启动同步进程
 sync_data &
+# 启动 Cloudreve
 exec /opt/cloudreve/cloudreve -c /opt/cloudreve/config.ini