{
 "categories": [
  {
   "id": "personal-assistants",
   "title": "Open-source personal assistants",
   "blurb": "Self-hostable AI assistants that act on your mail, calendar, docs and chat on your behalf.",
   "projects": [
    "openclaw",
    "hermes-agent",
    "khoj",
    "leon",
    "qm",
    "inbox-zero",
    "openbot",
    "rakazo",
    "waku-agent",
    "agenvoy"
   ],
   "queued": [],
   "questions": [
    {
     "qid": "architecture",
     "title": "How is the assistant architected?",
     "covers": "Agent loop and runtime; frontend/backend split; main packages; how a user request flows to an action.",
     "url": "/api/v1/questions/personal-assistants/architecture.json"
    },
    {
     "qid": "integrations",
     "title": "How are integrations (email, calendar, chat, docs) implemented?",
     "covers": "Which services; API clients vs MCP; OAuth flow; token storage; sync vs on-demand fetch.",
     "url": "/api/v1/questions/personal-assistants/integrations.json"
    },
    {
     "qid": "memory",
     "title": "How is memory and user context stored and retrieved?",
     "covers": "Storage (DB, vector store, files); what is remembered; how it is injected into prompts; summarization.",
     "url": "/api/v1/questions/personal-assistants/memory.json"
    },
    {
     "qid": "action-safety",
     "title": "How are actions on the user's behalf gated?",
     "covers": "Approval / human-in-the-loop flows; permission scopes; dry-run or draft modes; audit trail.",
     "url": "/api/v1/questions/personal-assistants/action-safety.json"
    },
    {
     "qid": "models",
     "title": "How are LLM providers selected and configured?",
     "covers": "Supported providers; config surface; tool-calling / structured-output usage; local-model support.",
     "url": "/api/v1/questions/personal-assistants/models.json"
    },
    {
     "qid": "deploy",
     "title": "How is it deployed and self-hosted?",
     "covers": "Runtime dependencies (DB, queues, browsers); Docker/one-click paths; required external accounts.",
     "url": "/api/v1/questions/personal-assistants/deploy.json"
    }
   ]
  },
  {
   "id": "browser-control",
   "title": "Browser & computer control",
   "blurb": "Frameworks that let an LLM drive a browser (or a whole desktop/phone) to complete tasks.",
   "projects": [
    "browser-use",
    "cua",
    "stagehand",
    "skyvern",
    "jev-ultrafast",
    "browser-harness",
    "workflow-use",
    "agent-desktop",
    "hyperagent",
    "typesafe-computer-use"
   ],
   "queued": [
    "omdsh-dev/dsh-browser",
    "jkudish/jev-browser",
    "droidrun/mobile-jev",
    "ekzhang/openjev-sglang",
    "savka777/jev-use",
    "razaanstha/ulka"
   ],
   "questions": [
    {
     "qid": "page-perception",
     "title": "How is the page represented to the model?",
     "covers": "DOM serialization, accessibility tree, screenshots, set-of-marks/element indexes; size limits and pruning.",
     "url": "/api/v1/questions/browser-control/page-perception.json"
    },
    {
     "qid": "action-execution",
     "title": "How are actions executed and how are elements targeted?",
     "covers": "CDP / Playwright / OS-level input; selectors vs indexes vs coordinates; typing, scrolling, file upload, tabs.",
     "url": "/api/v1/questions/browser-control/action-execution.json"
    },
    {
     "qid": "agent-loop",
     "title": "How is the agent loop / planning implemented?",
     "covers": "Step loop; planner vs executor; tool-call schema; stop conditions; memory between steps.",
     "url": "/api/v1/questions/browser-control/agent-loop.json"
    },
    {
     "qid": "reliability",
     "title": "How are failures, retries and self-healing handled?",
     "covers": "Error classes caught; retries; replanning; caching of successful actions or workflows; timeouts.",
     "url": "/api/v1/questions/browser-control/reliability.json"
    },
    {
     "qid": "models",
     "title": "Which models are supported and how are they called?",
     "covers": "Providers; vision requirement; structured output / tool calling; small or specialised models.",
     "url": "/api/v1/questions/browser-control/models.json"
    },
    {
     "qid": "sessions",
     "title": "How are browser sessions, profiles, auth and anti-bot handled?",
     "covers": "Local vs remote/cloud browsers; persistent profiles and cookies; stealth; proxies; CAPTCHA handling.",
     "url": "/api/v1/questions/browser-control/sessions.json"
    }
   ]
  },
  {
   "id": "connectors",
   "title": "API layer & connectors",
   "blurb": "Platforms that give agents and apps authenticated access to hundreds of third-party APIs.",
   "projects": [
    "composio",
    "activepieces",
    "nango",
    "pipedream",
    "klavis",
    "aci",
    "mcp-context-forge",
    "executor",
    "metorial",
    "merge-mcp"
   ],
   "queued": [],
   "questions": [
    {
     "qid": "auth",
     "title": "How is third-party authentication implemented?",
     "covers": "OAuth2 flows, API keys, token refresh, credential storage and encryption, multi-tenant connected accounts.",
     "url": "/api/v1/questions/connectors/auth.json"
    },
    {
     "qid": "definition",
     "title": "How is an integration / connector defined?",
     "covers": "Manifest or schema format; code vs config; codegen from OpenAPI; versioning; how many integrations ship.",
     "url": "/api/v1/questions/connectors/definition.json"
    },
    {
     "qid": "agent-exposure",
     "title": "How are integrations exposed to LLM agents?",
     "covers": "MCP servers, function-calling schemas, SDKs per framework; tool search / dynamic tool loading.",
     "url": "/api/v1/questions/connectors/agent-exposure.json"
    },
    {
     "qid": "execution",
     "title": "How is a tool call executed?",
     "covers": "Proxy vs direct; sandboxing; rate limiting; retries; pagination; error mapping.",
     "url": "/api/v1/questions/connectors/execution.json"
    },
    {
     "qid": "sync-triggers",
     "title": "How are data sync, webhooks and triggers implemented?",
     "covers": "Scheduled syncs; incremental cursors; webhook ingestion; event triggers for agents.",
     "url": "/api/v1/questions/connectors/sync-triggers.json"
    },
    {
     "qid": "self-host",
     "title": "How is it self-hosted and what is open vs proprietary?",
     "covers": "Licence; which components are in the repo; required services; what needs the hosted cloud.",
     "url": "/api/v1/questions/connectors/self-host.json"
    }
   ]
  },
  {
   "id": "ai-scraping",
   "title": "AI web scraping",
   "blurb": "Crawlers, extractors and stealth browsers that turn web pages into LLM-ready data.",
   "projects": [
    "firecrawl",
    "scrapling",
    "crawl4ai",
    "scrapegraph-ai",
    "browserless",
    "autoscraper",
    "omniparse",
    "llm-scraper",
    "trafilatura",
    "cyberscraper-2077"
   ],
   "queued": [
    "raznem/parsera",
    "scraperai/scraperai",
    "BrowserBox/BrowserBox",
    "vinyzu-archive/Botright",
    "ttlns/Selenium-Driverless",
    "crawlab-team/crawlab",
    "ssssssss-team/spider-flow",
    "mixmark-io/turndown"
   ],
   "questions": [
    {
     "qid": "fetching",
     "title": "How are pages fetched and rendered?",
     "covers": "Plain HTTP vs headless browser; JS rendering; waiting strategy; supported content types (PDF, images).",
     "url": "/api/v1/questions/ai-scraping/fetching.json"
    },
    {
     "qid": "extraction",
     "title": "How is content extracted or converted?",
     "covers": "HTML→Markdown/text heuristics; readability-style boilerplate removal; selectors; schema-based extraction.",
     "url": "/api/v1/questions/ai-scraping/extraction.json"
    },
    {
     "qid": "llm-usage",
     "title": "How are LLMs used, if at all?",
     "covers": "Prompting; chunking of large pages; structured output / JSON schema; which providers; cost controls.",
     "url": "/api/v1/questions/ai-scraping/llm-usage.json"
    },
    {
     "qid": "anti-bot",
     "title": "How are anti-bot measures, proxies and fingerprinting handled?",
     "covers": "Stealth patches; fingerprint spoofing; proxy rotation; CAPTCHA handling; rate limiting.",
     "url": "/api/v1/questions/ai-scraping/anti-bot.json"
    },
    {
     "qid": "crawling",
     "title": "How is crawling at scale implemented?",
     "covers": "Queues; concurrency; URL dedup; depth/limits; robots.txt and politeness; distributed workers.",
     "url": "/api/v1/questions/ai-scraping/crawling.json"
    },
    {
     "qid": "interface",
     "title": "What is the developer interface?",
     "covers": "Library API, CLI, REST service, MCP server, UI; output formats; language bindings.",
     "url": "/api/v1/questions/ai-scraping/interface.json"
    }
   ]
  },
  {
   "id": "open-source-deepwiki",
   "title": "Open-source DeepWiki",
   "blurb": "Self-hosted generators that turn a code repository into a browsable, AI-written wiki.",
   "projects": [
    "deepwiki-open",
    "opendeepwiki",
    "deepwiki-rs",
    "repoagent",
    "readmex",
    "repowiki",
    "opendeepwiki-go"
   ],
   "queued": [],
   "questions": [
    {
     "qid": "ingestion",
     "title": "How is a repository ingested and chunked?",
     "covers": "Clone or local path; file filters; chunking strategy; supported hosts; large-repo limits.",
     "url": "/api/v1/questions/open-source-deepwiki/ingestion.json"
    },
    {
     "qid": "retrieval",
     "title": "How is retrieval (RAG) implemented?",
     "covers": "Embedding models; vector store; top-k; how retrieved code reaches the prompt; or agentic file reading instead of RAG.",
     "url": "/api/v1/questions/open-source-deepwiki/retrieval.json"
    },
    {
     "qid": "structure",
     "title": "How is the wiki structure (table of contents) determined?",
     "covers": "Prompt/agent that proposes sections and pages; inputs used (file tree, README); output format.",
     "url": "/api/v1/questions/open-source-deepwiki/structure.json"
    },
    {
     "qid": "page-generation",
     "title": "How are individual pages generated?",
     "covers": "Per-page prompts; source-file citations; diagrams (Mermaid); parallelism; caching and regeneration.",
     "url": "/api/v1/questions/open-source-deepwiki/page-generation.json"
    },
    {
     "qid": "providers",
     "title": "How are model providers configured?",
     "covers": "Supported providers; per-stage model selection; OpenAI-compatible endpoints; local models.",
     "url": "/api/v1/questions/open-source-deepwiki/providers.json"
    },
    {
     "qid": "qa",
     "title": "How is interactive Q&A / chat implemented?",
     "covers": "Chat over the repo; deep-research mode; streaming; conversation memory; MCP exposure.",
     "url": "/api/v1/questions/open-source-deepwiki/qa.json"
    }
   ]
  },
  {
   "id": "memory",
   "title": "Agent memory layers",
   "blurb": "Libraries and services that give LLM agents long-term memory — extracting, storing, updating and recalling what matters across sessions.",
   "projects": [
    "claude-mem",
    "mem0",
    "hindsight",
    "graphiti",
    "cognee",
    "supermemory",
    "memori",
    "memos",
    "honcho",
    "memmachine"
   ],
   "queued": [
    "letta-ai/letta"
   ],
   "questions": [
    {
     "qid": "extraction",
     "title": "How are memories extracted from interactions?",
     "covers": "What gets stored (facts, events, preferences); LLM extraction prompts; deduplication and conflict handling at write time.",
     "url": "/api/v1/questions/memory/extraction.json"
    },
    {
     "qid": "storage",
     "title": "How are memories stored?",
     "covers": "Vector, graph, key-value or SQL backends; the memory schema; embeddings used; pluggable stores.",
     "url": "/api/v1/questions/memory/storage.json"
    },
    {
     "qid": "retrieval",
     "title": "How are memories retrieved and injected into the prompt?",
     "covers": "Search strategy (semantic, keyword, graph, temporal); ranking and filtering; how results reach the LLM context.",
     "url": "/api/v1/questions/memory/retrieval.json"
    },
    {
     "qid": "lifecycle",
     "title": "How are memories updated, consolidated or forgotten?",
     "covers": "Update/merge logic; summarisation or consolidation; decay, TTL and deletion; versioning or history.",
     "url": "/api/v1/questions/memory/lifecycle.json"
    },
    {
     "qid": "scoping",
     "title": "How is memory scoped and isolated?",
     "covers": "User / agent / session / tenant scoping; multi-tenancy; access control; privacy controls.",
     "url": "/api/v1/questions/memory/scoping.json"
    },
    {
     "qid": "integration",
     "title": "How do agents integrate with it, and what is self-hostable?",
     "covers": "SDKs, REST API, MCP server, framework plugins; required services; open vs hosted-only parts.",
     "url": "/api/v1/questions/memory/integration.json"
    }
   ]
  },
  {
   "id": "graph-rag",
   "title": "Graph RAG",
   "blurb": "Retrieval-augmented generation over knowledge graphs — extracting entities and relations from documents and querying the graph alongside vectors.",
   "projects": [
    "graphify",
    "lightrag",
    "graphrag",
    "semantica",
    "llm-graph-builder",
    "hipporag",
    "nano-graphrag",
    "autoflow",
    "trustgraph",
    "vector-graph-rag"
   ],
   "queued": [],
   "questions": [
    {
     "qid": "graph-construction",
     "title": "How is the knowledge graph extracted from documents?",
     "covers": "Chunking; entity and relation extraction prompts or models; entity resolution / deduplication; schema or ontology.",
     "url": "/api/v1/questions/graph-rag/graph-construction.json"
    },
    {
     "qid": "graph-storage",
     "title": "Where and how is the graph stored?",
     "covers": "Graph database vs files vs in-memory; node/edge schema; how embeddings sit next to the graph.",
     "url": "/api/v1/questions/graph-rag/graph-storage.json"
    },
    {
     "qid": "communities",
     "title": "Are communities, summaries or hierarchies built over the graph?",
     "covers": "Community detection (e.g. Leiden); hierarchical summaries; when they are computed; if not done, say so.",
     "url": "/api/v1/questions/graph-rag/communities.json"
    },
    {
     "qid": "query",
     "title": "How does query-time retrieval use the graph?",
     "covers": "Local / global / hybrid modes; traversal; combining graph and vector hits; how context is assembled for the LLM.",
     "url": "/api/v1/questions/graph-rag/query.json"
    },
    {
     "qid": "incremental",
     "title": "How are updates and incremental indexing handled?",
     "covers": "Adding or changing documents without a full rebuild; deletion; caching of extraction results.",
     "url": "/api/v1/questions/graph-rag/incremental.json"
    },
    {
     "qid": "cost",
     "title": "How are LLM cost and latency controlled during indexing and query?",
     "covers": "Caching; batching; model choice per stage; token budgets; small-model or non-LLM shortcuts.",
     "url": "/api/v1/questions/graph-rag/cost.json"
    }
   ]
  },
  {
   "id": "rag",
   "title": "RAG engines",
   "blurb": "End-to-end retrieval-augmented generation engines and frameworks — document parsing, chunking, indexing, retrieval, reranking and grounded answers.",
   "projects": [
    "ragflow",
    "anything-llm",
    "llama_index",
    "quivr",
    "pageindex",
    "onyx",
    "haystack",
    "kotaemon",
    "rag-anything",
    "r2r"
   ],
   "queued": [],
   "questions": [
    {
     "qid": "ingestion",
     "title": "How are documents parsed and chunked?",
     "covers": "Supported formats; OCR and layout parsing; table handling; chunking strategy and sizes.",
     "url": "/api/v1/questions/rag/ingestion.json"
    },
    {
     "qid": "indexing",
     "title": "How are embeddings and indexes built and stored?",
     "covers": "Embedding models; vector stores supported; hybrid / keyword (BM25) indexes; metadata.",
     "url": "/api/v1/questions/rag/indexing.json"
    },
    {
     "qid": "retrieval",
     "title": "How is retrieval performed?",
     "covers": "Dense / sparse / hybrid search; reranking; query rewriting or decomposition; filters.",
     "url": "/api/v1/questions/rag/retrieval.json"
    },
    {
     "qid": "generation",
     "title": "How are answers generated and grounded?",
     "covers": "Prompt assembly; citations / source attribution; streaming; agentic or multi-step answering.",
     "url": "/api/v1/questions/rag/generation.json"
    },
    {
     "qid": "evaluation",
     "title": "How is quality evaluated or observed?",
     "covers": "Built-in evals, metrics, tracing / observability hooks; if absent, say so.",
     "url": "/api/v1/questions/rag/evaluation.json"
    },
    {
     "qid": "deploy",
     "title": "How is it deployed and operated?",
     "covers": "Library vs service; UI; API; required infrastructure; scaling and multi-tenancy.",
     "url": "/api/v1/questions/rag/deploy.json"
    }
   ]
  },
  {
   "id": "coding-agents",
   "title": "Open-source coding agents",
   "blurb": "Terminal, IDE and desktop agents that read a codebase, edit files and run commands — how they gather context, apply edits and stay safe.",
   "projects": [
    "opencode",
    "codex",
    "cline",
    "openinterpreter",
    "aider",
    "codewhale",
    "deepseek-reasonix",
    "open-lovable",
    "qwen-code",
    "bolt.diy",
    "pi-desktop",
    "vibesdk"
   ],
   "queued": [
    "google-gemini/gemini-cli",
    "OpenHands/OpenHands",
    "aaif-goose/goose",
    "continuedev/continue",
    "earendil-works/pi"
   ],
   "questions": [
    {
     "qid": "agent-loop",
     "title": "How is the agent loop implemented?",
     "covers": "Planner vs single loop; tool-call schema; turn structure; stop conditions; sub-agents.",
     "url": "/api/v1/questions/coding-agents/agent-loop.json"
    },
    {
     "qid": "context",
     "title": "How is repository context gathered and kept within the context window?",
     "covers": "Repo maps, search/grep tools, embeddings, file-reading strategy; summarisation or compaction of long sessions.",
     "url": "/api/v1/questions/coding-agents/context.json"
    },
    {
     "qid": "editing",
     "title": "How are code edits applied?",
     "covers": "Edit formats (diff, search/replace, whole file, patch); how edits are validated (lint, tests, retries); undo or git integration.",
     "url": "/api/v1/questions/coding-agents/editing.json"
    },
    {
     "qid": "execution-safety",
     "title": "How are shell commands and file writes kept safe?",
     "covers": "Approval modes; sandboxing (containers, seatbelt, landlock); allow/deny lists; network restrictions; checkpoints.",
     "url": "/api/v1/questions/coding-agents/execution-safety.json"
    },
    {
     "qid": "models",
     "title": "Which models are supported and how are they called?",
     "covers": "Providers; local models; tool calling vs text formats; per-model prompt tuning; cost tracking.",
     "url": "/api/v1/questions/coding-agents/models.json"
    },
    {
     "qid": "extensibility",
     "title": "How can it be extended and customised?",
     "covers": "MCP; plugins or custom tools; rules/instruction files (AGENTS.md etc.); hooks; headless/SDK use.",
     "url": "/api/v1/questions/coding-agents/extensibility.json"
    }
   ]
  }
 ]
}