OARC-Crawlers Cheat Sheet
pip install oarc-crawlers # Basic install
# or
pip install uv && uv pip install oarc-crawlers # Recommended install
# Windows
set OARC_DATA_DIR=C:\p ath\t o\d ata
# Linux/WSL
export OARC_DATA_DIR=~ /path/to/data
# Or specify in code
from oarc_crawlers import YTCrawler
crawler = YTCrawler(data_dir=" ./my_data_dir" )
# Download video
oarc-crawlers yt download --url " https://www.youtube.com/watch?v=MDbdb-W4x4w&t=330s&ab_channel=MattWilliams"
# Download specific quality
oarc-crawlers yt download --url " https://www.youtube.com/watch?v=MDbdb-W4x4w&t=330s&ab_channel=MattWilliams" --resolution 720p
# Download as audio only
oarc-crawlers yt download --url " https://www.youtube.com/watch?v=MDbdb-W4x4w&t=330s&ab_channel=MattWilliams" --extract-audio --format mp3
# Download playlist (first 10 videos)
oarc-crawlers yt playlist --url " https://www.youtube.com/playlist?list=PLzH6n4zXuckquVnQ0KlMDxyXxiSO2DXOQ"
# Get video captions
oarc-crawlers yt captions --url " https://www.youtube.com/watch?v=MDbdb-W4x4w&t=330s&ab_channel=MattWilliams" --languages " en,es,fr"
# Search videos
oarc-crawlers yt search --query " python tutorials" --limit 5
# Fetch live chat messages
oarc-crawlers yt chat --video-id dQw4w9WgXcQ --max-messages 500 --duration 300
# Clone repository
oarc-crawlers gh clone --url " https://github.com/user/repo"
# Analyze repository
oarc-crawlers gh analyze --url " https://github.com/user/repo"
# Find similar code snippets
oarc-crawlers gh find-similar --url " https://github.com/user/repo" --code " def example():" --language python
# Download paper metadata
oarc-crawlers arxiv download --id " 2310.12123"
# Search papers
oarc-crawlers arxiv search --query " quantum computing" --limit 5
# Get LaTeX source
oarc-crawlers arxiv latex --id " 2310.12123"
# Text search
oarc-crawlers ddg text --query " python async" --max-results 5
# Image search
oarc-crawlers ddg images --query " cute cats" --max-results 10
# News search
oarc-crawlers ddg news --query " technology" --max-results 20
# Crawl webpage
oarc-crawlers web crawl --url " https://docs.llamaindex.ai/en/stable/examples/query_engine/pandas_query_engine/"
# Save to file
oarc-crawlers web crawl --url " https://docs.llamaindex.ai/en/stable/examples/query_engine/pandas_query_engine/" --output-file page.txt
# Get docs
oarc-crawlers web docs --url " https://docs.llamaindex.ai/en/stable/examples/query_engine/pandas_query_engine/"
# Get PyPI info
oarc-crawlers web pypi --package " requests"
# View Parquet file
oarc-crawlers data view ./data/example.parquet --max-rows 20
# Run MCP server for agent integration (default port 3000)
oarc-crawlers mcp run
# Install MCP server for VS Code integration
oarc-crawlers mcp install --name " OARC Tools"
# Build the package
oarc-crawlers build package
# Publish to PyPI
oarc-crawlers publish pypi
# Publish to TestPyPI
oarc-crawlers publish pypi --test
# Enable verbose output
oarc-crawlers --verbose [command]
# Use custom config
oarc-crawlers --config ~ /.oarc/config.ini [command]
# Get help
oarc-crawlers --help
oarc-crawlers [command] --help
from oarc_crawlers import (
YTCrawler ,
GHCrawler ,
ArxivCrawler ,
DDGCrawler ,
WebCrawler ,
ParquetStorage ,
)
import asyncio
from oarc_crawlers import YTCrawler
async def download_video ():
yt = YTCrawler (data_dir = "./data" )
result = await yt .download_video ("https://youtube.com/watch?v=dQw4w9WgXcQ" )
print (f"Video saved to: { result .get ('file_path' , 'N/A' )} " )
print (f"Video title: { result .get ('title' , 'Unknown' )} " )
asyncio .run (download_video ())
Analyze a GitHub repository
import asyncio
from oarc_crawlers import GHCrawler
async def analyze_repo ():
gh = GHCrawler (data_dir = "./data" )
summary = await gh .get_repo_summary ("https://github.com/pytorch/pytorch" )
print (summary )
asyncio .run (analyze_repo ())
Search ArXiv and download LaTeX source
import asyncio
from oarc_crawlers import ArxivCrawler
async def arxiv_example ():
arxiv = ArxivCrawler (data_dir = "./data" )
paper_info = await arxiv .fetch_paper_info ("2103.00020" )
print (f"Title: { paper_info ['title' ]} " )
source = await arxiv .download_source ("2103.00020" )
print (f"Main TeX file: { source .get ('main_tex_file' , 'N/A' )} " )
asyncio .run (arxiv_example ())
import asyncio
from oarc_crawlers import DDGCrawler
async def ddg_example ():
ddg = DDGCrawler (data_dir = "./data" )
results = await ddg .text_search ("python async programming" , max_results = 5 )
print (results )
asyncio .run (ddg_example ())
Crawl a webpage and extract text
import asyncio
from oarc_crawlers import WebCrawler
async def crawl_example ():
web = WebCrawler (data_dir = "./data" )
html = await web .fetch_url_content ("https://www.python.org/" )
text = WebCrawler .extract_text_from_html (html )
print (text [:500 ])
asyncio .run (crawl_example ())
Parquet Storage: Save and load data
from oarc_crawlers import ParquetStorage
data = {"name" : "Example" , "value" : 42 }
ParquetStorage .save_to_parquet (data , "./data/example.parquet" )
df = ParquetStorage .load_from_parquet ("./data/example.parquet" )
print (df )
Variable
Description
Default (Linux)
Default (Windows)
OARC_DATA_DIR
Data storage directory
~/.local/share/oarc/data
%APPDATA%\oarc\data
OARC_CONFIG_DIR
Config files directory
~/.config/oarc
%APPDATA%\oarc\config
OARC_CACHE_DIR
Cache directory
~/.cache/oarc
%LOCALAPPDATA%\oarc\cache
OARC_LOG_LEVEL
Logging level (DEBUG, INFO, etc.)
INFO
INFO
OARC_MAX_RETRIES
Max network retries
3
3
OARC_TIMEOUT
Network timeout (seconds)
30
30
OARC_USER_AGENT
User agent string for requests
oarc-crawlers/VERSION
oarc-crawlers/VERSION
OARC_PROXY
HTTP/HTTPS proxy URL
None
None
OARC_NO_PROGRESS
Disable progress bars (set to 1 to disable)
0
0
Use --verbose for detailed logs.
Check docs/Troubleshoot.md for common issues.
For CLI help: oarc-crawlers [command] --help
For API reference: see docs/API.md