{"owner":"lechmazur","github":"https://github.com/lechmazur","claimed":false,"inventory":[],"indexed":[{"repo":"lechmazur/writing","github":"https://github.com/lechmazur/writing","description":"This benchmark tests how well LLMs incorporate a set of 10 mandatory story elements (characters, objects, core concepts, attributes, motivations, etc.) in a short creative story","language":null,"stars":423,"topics":["claude","llama","llm","o1","gpt-4-5","claude-3-7-sonnet","deepseek","deepseek-r1","gemini"],"license":null,"category":"ai-agents"},{"repo":"lechmazur/confabulations","github":"https://github.com/lechmazur/confabulations","description":"Hallucinations (Confabulations) Document-Based Benchmark for RAG. Includes human-verified questions and answers.","language":"HTML","stars":249,"topics":["benchmark","claude","gemini","hallucinations","leaderboard","llama","llm","rag","confabulations","ai-evaluation"],"license":null,"category":"ai-agents"},{"repo":"lechmazur/nyt-connections","github":"https://github.com/lechmazur/nyt-connections","description":"Benchmark that evaluates LLMs using 759 NYT Connections puzzles extended with extra trick words","language":"Python","stars":238,"topics":["benchmark","evaluation","llm","llms-benchmarking","puzzles","reasoning","testing","claude","gemini-pro","gpt-5"],"license":null,"category":"ai-agents"},{"repo":"lechmazur/bazaar","github":"https://github.com/lechmazur/bazaar","description":"The BAZAAR challenges LLMs to navigate the double-auction marketplace, where buyers and sellers must make strategic decisions with incomplete information. Each agent receives a private value and must decide how to quote based solely on the history of previous rounds. A realistic test of market intuition and strategic adaptation.","language":null,"stars":38,"topics":["claude","gemini","grok","llama","llm","lrm","o3","o4-mini","opus","qwen"],"license":null,"category":"ai-agents"},{"repo":"lechmazur/divergent","github":"https://github.com/lechmazur/divergent","description":"LLM Divergent Thinking Creativity Benchmark. LLMs generate 25 unique words that start with a given letter with no connections to each other or to 50 initial random words.","language":null,"stars":35,"topics":["benchmark","claude","deepseek","gemini","gpt-4o","leaderboard","llama","llm","o1","qwen"],"license":null,"category":"ai-agents"},{"repo":"lechmazur/deception","github":"https://github.com/lechmazur/deception","description":"Benchmark evaluating LLMs on their ability to create and resist disinformation. Includes comprehensive testing across major models (Claude, GPT-4, Gemini, Llama, etc.) with standardized evaluation metrics.","language":null,"stars":33,"topics":["ai-benchmarks","ai-evaluation","ai-safety","ai-security","claude","disinformation","gemini","gpt4o","language-model","llama"],"license":null,"category":"ai-agents"},{"repo":"lechmazur/debate","github":"https://github.com/lechmazur/debate","description":"Adversarial multi-turn benchmark for LLM debate quality, using side-swapped matchups and multi-model judging to rank models by judged debate performance.","language":null,"stars":30,"topics":["benchmark","claude","debates","gemini","gpt","leaderboard","llm"],"license":null,"category":"ai-agents"}],"how_to_buy":"GET /r/lechmazur/<repo> (Accept: application/json) for any listed repo here: tree, README, price and the checkout to pay (x402; rehearse first at its test twin, simulated money). Repos under 'indexed' are free: clone them from GitHub."}