# Awesome-Embodied&MM system configuration. # Everything domain/tuning-related lives here so behavior changes need no code edits. provider: # OpenRouter is OpenAI-compatible. Key read from this env var. base_url: "https://openrouter.ai/api/v1" api_key_env: "OPENROUTER_API_KEY" # Sent as OpenRouter attribution headers (optional but recommended). http_referer: "https://github.com/wzii/Awesome_Embodied_MM" app_title: "Awesome-Embodied&MM" models: # Tiered routing. Cheap = filtering/first-pass (high volume); mid = analysis; strong = # scoring. Chosen via a blind Opus-4.8 ranking of anonymized outputs on real papers (see # scripts/bench_models.py). Latency is NOT a selection criterion for these batch stages # (only interactive Q&A cares about latency, and its model is deferred): # - cheap: deepseek-v4-flash — #1 on the scoring task (best skeptical N/A calibration, # no fabrication) AND cheapest by ~50x; ideal for the high-volume relevance filter. # - mid/strong: glm-5.1 — #1 overall in the blind ranking, low cost; its only weakness # (27s latency) is irrelevant for batch work. # Excluded: deepseek-v4-pro (abdicated scoring -> all-N/A), glm-4.7-flash (4/6 parse # failures), gemini-3.1-flash-lite (fabricated scores). tiers: cheap: "deepseek/deepseek-v4-flash" mid: "z-ai/glm-5.1" strong: "z-ai/glm-5.1" xstrong: "openai/gpt-5.5" # top tier for the most quality-critical stage (scoring) # Optional per-stage overrides (stage name -> tier). Falls back to the tier passed in code. stage_tiers: filter: cheap summarize: cheap analyze: mid extract: cheap # deepseek-v4-flash: fast + JSON-reliable + faithful for table extraction score: xstrong # flagship two-layer scoring -> gpt-5.5 innovation: mid front_summary: strong qa: strong # TODO: pick a low-latency Q&A model in Phase 7 (deferred) defaults: temperature: 0.2 max_tokens: 4096 # USD per 1,000,000 tokens, {input, output}. Used for cost tracking/logging. cost_per_million: "deepseek/deepseek-v4-flash": {input: 0.098, output: 0.197} "z-ai/glm-5.1": {input: 0.98, output: 3.08} "openai/gpt-5.5": {input: 5.00, output: 30.00} # alternates considered: gemini-3.5-flash {1.50, 9.00}, claude-sonnet-4.6 {3.00, 15.00} constants: request_timeout: 90 # seconds per LLM call max_retries: 4 retry_backoff: 3.0 # seconds, exponential base lookback_days: 60 # ignore papers older than this on fetch relevance_threshold: 0.5 # below this -> dropped analyze_cap: 40 # max papers fully analyzed per run (cost guard) logging: level: "INFO" # console level; debug file sink enabled via --debug/WAM_DEBUG log_dir: "logs" email: feature_threshold: 6.0 # core papers with weighted_total >= this get a detailed card max_featured: 8 # cap on featured (detailed) papers max_grouped: 40 # cap on the grouped lower tier group_lower_by_direction: true # group the lower tier by research direction (else flat list) paths: db: "data/wam.db" pdf_cache: "cache/pdfs" index_dir: "data/index" # Two-layer scoring weights. Top-4 WAM metrics weighted 2x; computed over non-N/A metrics. scoring: general_weights: novelty: 1.0 soundness: 1.0 impact: 1.0 wam_weights: generalist: 2.0 inference_speed: 2.0 specialist: 2.0 inference_cost: 2.0 trustworthiness: 1.0 collaborative: 1.0 controlled_generation: 1.0 other: 1.0 # Source + search config is consumed in Phase 2 (fetch). sources: arxiv: enabled: true categories: ["cs.RO", "cs.AI", "cs.LG", "cs.CV"] max_results: 300 semantic_scholar: enabled: true api_key_env: "SEMANTIC_SCHOLAR_API_KEY" # optional; higher rate limits if set papers_with_code: enabled: true news: enabled: true feeds: - "https://www.therobotreport.com/feed/" - "https://spectrum.ieee.org/feeds/topic/robotics.rss" - "https://huggingface.co/blog/feed.xml" # Search keyword sets. The LLM filter makes the final core/adjacent/drop call, # but these seed the source queries (broad net by design). keywords: core: # World Action Models proper - "world action model" - "action world model" - "embodied foundation model" - "robot foundation model" - "generalist robot policy" adjacent: # transferable techniques — kept in the innovation track - "vision-language-action" - "world model" - "video generation" - "action-conditioned video" - "interactive world model" qa: # Knowledge-base Q&A is long-context (no embeddings/vector DB): a compact pack of # paper summaries + scores + leaderboard + authors + trends is built from the DB and # passed to the qa-tier model. See stage_tiers.qa for the model. max_papers: 400 # cap papers in the pack (most-scored first) to bound context include_dropped: true # include filtered-out papers (title-only) so the KB still covers them web_search: true # also search the web (paper original text, related concepts) web_max_results: 5 # web results per query (OpenRouter web plugin; ~$0.02/query) trends: sim_threshold: 0.62 # cosine edge threshold for the direction graph min_front_size: 3 # ignore clusters smaller than this window_days: 30 # momentum = recent window vs prior window people: min_papers: 2 # influential if on >= N tracked papers min_citations: 2000 # (reserved) citation-based influence via Semantic Scholar max_per_run: 30 # cap influential authors processed per run (most prolific first)