{"owner":"adbar","github":"https://github.com/adbar","claimed":false,"inventory":[],"indexed":[{"repo":"adbar/trafilatura","github":"https://github.com/adbar/trafilatura","description":"Python & Command-line tool to gather text and metadata on the Web: Crawling, scraping, extraction, output as CSV, JSON, HTML, MD, TXT, XML","language":"Python","stars":6647,"topics":["article-extractor","corpus-builder","corpus-tools","crawler","html-to-markdown","html2text","llm","news-aggregator","news-crawler","nlp","rag","readability","rss-feed","scraping","tei","text-cleaning","text-extraction","text-mining","text-preprocessing","web-scraping"],"license":"Apache-2.0","category":"data_extraction"},{"repo":"adbar/courlan","github":"https://github.com/adbar/courlan","description":"Clean, filter and sample URLs to optimize data collection – Python & command-line – Deduplication, spam, content and language filters","language":"Python","stars":178,"topics":["url","url-parsing","crawler","tld","uri","url-validation","url-parser","recon","crawling","url-checker"],"license":"Apache-2.0","category":"scrapers-browser-automation"},{"repo":"adbar/htmldate","github":"https://github.com/adbar/htmldate","description":"Fast and robust date extraction from web pages, with Python or on the command-line","language":"Python","stars":155,"topics":["metadata-extraction","date-parser","entity-extraction","natural-language-processing","nlp","web-scraping","webscraping","date","datetime","metadata"],"license":"Apache-2.0","category":"scrapers-browser-automation"}],"how_to_buy":"GET /r/adbar/<repo> (Accept: application/json) for any listed repo here: tree, README, price and the checkout to pay (x402; rehearse first at its test twin, simulated money). Repos under 'indexed' are free: clone them from GitHub."}