{"owner":"StabRise","github":"https://github.com/StabRise","claimed":false,"inventory":[],"indexed":[{"repo":"StabRise/spark-pdf","github":"https://github.com/StabRise/spark-pdf","description":"PDF DataSource for Apache Spark, allow to read PDF files directly to the DataFrame and ocr it","language":"Scala","stars":81,"topics":["ocr","ocr-recognition","pdf","pdf-document","pdf-document-processor","spark","spark-datasource","big-data","data-engineering","data-extraction"],"license":"AGPL-3.0","category":"media-processing"},{"repo":"StabRise/ScaleDP","github":"https://github.com/StabRise/ScaleDP","description":"ScaleDP is an Open-Source extension of Apache Spark for Document Processing","language":"Python","stars":19,"topics":["pdf","pdf-document-processor","spark","ocr","ocr-python","ocr-recognition","easyocr","huggingface-models","machine-learning","nlp"],"license":"AGPL-3.0","category":"media-processing"}],"how_to_buy":"GET /r/StabRise/<repo> (Accept: application/json) for any listed repo here: tree, README, price and the checkout to pay (x402; rehearse first at its test twin, simulated money). Repos under 'indexed' are free: clone them from GitHub."}