Skills Performance Optimization

When a Skill processes large files or is called frequently, performance issues directly affect user experience.

This article introduces common techniques to improve processing speed at the script level.


Common Sources of Performance Issues

SourceTypical SymptomsOptimization Direction
Loading large files all at onceReading a 100MB CSV takes several secondsChunked reading (chunk)
Repeatedly performing the same calculationRecalculating statistics on every callCaching results
Single-threaded serial processingProcessing 1000 files takes a long timeParallel processing
Frequent disk I/OWriting files line by line in a loopBatch writing
Loading unnecessary dependenciesimport takes more than 2 secondsLazy import

Chunked Reading of Large Files

pandas'chunksizeparameter allows large files to be divided into several batches and processed chunk by chunk, avoiding excessive memory usage at once.

Example

# File path: scripts/chunked_reader.py
import pandas as pd
import json
import sys

def process_large_csv(file_path: str, chunk_size: int = 10000) -> dict:
    """
Read and process large CSV files in chunks

Parameters:
file_path: CSV file path
chunk_size: number of rows per chunk, default 10000

Returns:
Merged statistical results
    """

    total_rows   = 0
    total_sum    = 0.0
    chunk_count  = 0

    # Read chunk by chunk, without loading the entire file into memory at once
    for chunk in pd.read_csv(file_path, chunksize=chunk_size):
        chunk_count += 1
        total_rows  += len(chunk)

        # Perform statistics on each chunk, accumulate the results
        if "score" in chunk.columns:
            total_sum += chunk["score"].sum()

        # Report progress in real time
        print(f" Processed batch {chunk_count}, accumulated {total_rows} rows", flush=True)

    avg = total_sum / total_rows if total_rows > 0 else 0
    return {
        "status":      "success",
        "total_rows":  total_rows,
        "chunks":      chunk_count,
        "score_avg":   round(avg, 2)
    }

if __name__ == "__main__":
    file_path = sys.argv[1] if len(sys.argv) > 1 else ""
    result    = process_large_csv(file_path)
    print(json.dumps(result, ensure_ascii=False, indent=2))
  已处理第 1 批,累计 10000 行
  已处理第 2 批,累计 20000 行
  已处理第 3 批,累计 28456 行
{
  "status": "success",
  "total_rows": 28456,
  "chunks": 3,
  "score_avg": 82.37
}

Result Caching

For operations where the same input produces the same result, cache the result to a local file to avoid repeated computation.

Example

# File path: scripts/file_cache.py
import hashlib
import json
import os
import time

CACHE_DIR = "/home/claude/.skill_cache"
os.makedirs(CACHE_DIR, exist_ok=True)

def _file_fingerprint(file_path: str) -> str:
    """
Generate a fingerprint based on file path, size, and modification time
Much faster than MD5 of file contents, suitable for large files
    """

    stat = os.stat(file_path)
    raw  = f"{file_path}|{stat.st_size}|{stat.st_mtime}"
    return hashlib.md5(raw.encode()).hexdigest()

def get_cached(file_path: str, operation: str):
    """Get cache, return None if it does not exist"""
    key        = _file_fingerprint(file_path) + "_" + operation
    cache_file = os.path.join(CACHE_DIR, f"{key}.json")
    if os.path.exists(cache_file):
        with open(cache_file) as f:
            return json.load(f)
    return None

def set_cached(file_path: str, operation: str, result: dict):
    """Write cache"""
    key        = _file_fingerprint(file_path) + "_" + operation
    cache_file = os.path.join(CACHE_DIR, f"{key}.json")
    with open(cache_file, "w") as f:
        json.dump(result, f, ensure_ascii=False)

# Usage example
def get_stats(file_path: str) -> dict:
    # Check cache first
    cached = get_cached(file_path, "stats")
    if cached:
        print("Cache hit, skipping repeated computation")
        return cached

    # Cache miss, execute computation
    import pandas as pd
    t0 = time.time()
    df = pd.read_csv(file_path)
    result = {
        "rows":    len(df),
        "cols":    len(df.columns),
        "elapsed": round(time.time() - t0, 3)
    }

    # Write cache for next use
    set_cached(file_path, "stats", result)
    return result

Parallel Processing of Multiple Files

When multiple files need to be processed, useconcurrent.futuresfor parallel execution, which can improve speed several times.

Example

# File path: scripts/parallel_process.py
import concurrent.futures
import os
import json
import sys

def process_single_file(file_path: str) -> dict:
    """Process a single file (will be called in parallel)"""
    try:
        size = os.path.getsize(file_path)
        # Simulate actual processing logic
        return {"file": os.path.basename(file_path),
                "size_kb": size // 1024, "status": "ok"}
    except Exception as e:
        return {"file": file_path, "status": "error", "message": str(e)}

def process_files_parallel(file_paths: list, max_workers: int = 4) -> list:
    """
Parallel processing of multiple files

Parameters:
file_paths: list of file paths
max_workers: maximum concurrency, default 4 (to avoid too many processes competing for resources)
    """

    results = []
    with concurrent.futures.ThreadPoolExecutor(max_workers=max_workers) as executor:
        # Submit all tasks
        future_map = {
            executor.submit(process_single_file, fp): fp
            for fp in file_paths
        }
        # Collect results in order of completion
        for future in concurrent.futures.as_completed(future_map):
            result = future.result()
            results.append(result)
            print(f" Completed: {result['file']}")
    return results

if __name__ == "__main__":
    # Simulate: process all CSV files in the upload directory
    upload_dir = "/mnt/user-data/uploads"
    csv_files  = [
        os.path.join(upload_dir, f)
        for f in os.listdir(upload_dir)
        if f.endswith(".csv")
    ]

    if not csv_files:
        print("No CSV files found")
        sys.exit(0)

    print(f"Starting parallel processing of {len(csv_files)} files...")
    results = process_files_parallel(csv_files)
    print(json.dumps(results, ensure_ascii=False, indent=2))
开始并行处理 3 个文件...
  完成:example_jan.csv
  完成:example_mar.csv
  完成:example_feb.csv
[{"file": "example_jan.csv", "size_kb": 128, "status": "ok"}, ...]

Performance Optimization Checklist

Check ItemBefore OptimizationAfter Optimization
Reading large filespd.read_csv(file)pd.read_csv(file, chunksize=10000)
Repeatedly calculating statistics on the same fileRecalculating every timeUsing file fingerprint to cache results
Processing multiple filesfor loop, serialThreadPoolExecutor, parallel
Writing files line by linefor row: f.write(row)Collect in batches and write at once
Unnecessary importsAll imports at the top of the scriptImport when needed (lazy import)

Measure before optimizing; don't optimize by intuition. Use thetime.perf_counter()or@timeitdecorator to identify the real bottleneck, then optimize accordingly, avoiding unnecessary code complexity from premature optimization.

Other Extensions