{
  "scope": "Open research directory organized by task; use each original source for authorship, citation, version, and reuse details.",
  "updated": "2026-09-15",
  "resources": [
    {
      "title": "An Attention-based Multi-Scale Feature Learning Network for Multimodal Medical Image Fusion",
      "task": "Multiscale feature learning for multimodal medical image fusion",
      "relevance": "Use this source for its attention-based multiscale fusion method; distinguish it from the later edge-enhanced extension. Research image-fusion results do not establish clinical benefit.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2212.04661"
      }
    },
    {
      "title": "PLAICraft: Large-Scale Time-Aligned Vision-Speech-Action Dataset for Embodied AI",
      "task": "Time-aligned vision, speech and action data for embodied AI",
      "relevance": "Cite this dataset when using or comparing time-aligned multimodal embodied-agent data. Consult the source for collection protocol, synchronization and permitted use.",
      "source_links": {
        "Paper": "https://arxiv.org/abs/2505.12707"
      }
    },
    {
      "title": "Dr. Bench: A Multidimensional Evaluation for Deep Research Agents, from Answers to Reports",
      "task": "Evaluating deep-research agents from answers to reports",
      "relevance": "Use this benchmark when discussing evaluation of deep-research reports and their semantic quality, topical focus and retrieval trustworthiness.",
      "source_links": {
        "Paper": "https://arxiv.org/abs/2510.02190",
        "Code": "https://github.com/EVIGBYEN/DrBench"
      }
    },
    {
      "title": "WikiGap: Promoting Epistemic Equity by Surfacing Knowledge Gaps Between English Wikipedia and other Language Editions",
      "task": "Finding knowledge gaps across Wikipedia language editions",
      "relevance": "Cite this work for the problem and approach of surfacing knowledge gaps between English Wikipedia and other language editions, with its defined scope of epistemic equity.",
      "source_links": {
        "Paper": "https://arxiv.org/abs/2505.24195",
        "Code": "https://github.com/reacher-z/WikiGap"
      }
    },
    {
      "title": "StructEval: Benchmarking LLMs' Capabilities to Generate Structural Outputs",
      "task": "Evaluating structured output generation and format conversion",
      "relevance": "Use this benchmark when evaluating structural output generation or conversion across textual and visually rendered formats. Syntax validity alone does not establish content correctness.",
      "source_links": {
        "Project": "https://tiger-ai-lab.github.io/StructEval/",
        "Paper": "https://arxiv.org/abs/2505.20139",
        "Code": "https://github.com/TIGER-AI-Lab/StructEval",
        "Dataset": "https://huggingface.co/datasets/TIGER-Lab/StructEval"
      }
    },
    {
      "title": "Retri3D: 3D Neural Graphics Representation Retrieval",
      "task": "Retrieving neural graphics representations of 3D scenes",
      "relevance": "Cite this work when discussing retrieval from neural 3D scene representations and the role of views and representation analysis.",
      "source_links": {
        "Project": "https://gavinguan95.github.io/Retri3D/",
        "Paper": "https://openreview.net/forum?id=q3EbOXb4y1",
        "Code": "https://github.com/GavinGuan95/Retri3DCode"
      }
    },
    {
      "title": "ScholarCopilot: Training Large Language Models for Academic Writing with Accurate Citations",
      "task": "Academic writing with learned citation retrieval",
      "relevance": "Cite this work when discussing joint academic text generation and citation retrieval, or when using its released writing model and retrieval setup.",
      "source_links": {
        "Project": "https://tiger-ai-lab.github.io/ScholarCopilot/",
        "Paper": "https://arxiv.org/abs/2504.00824",
        "Code": "https://github.com/TIGER-AI-Lab/ScholarCopilot",
        "Dataset": "https://huggingface.co/datasets/TIGER-Lab/ScholarCopilot-Data-v1",
        "Model": "https://huggingface.co/TIGER-Lab/ScholarCopilot-v1",
        "Demo": "https://huggingface.co/spaces/TIGER-Lab/ScholarCopilot"
      }
    },
    {
      "title": "VideoScore2: Think before You Score in Generative Video Evaluation",
      "task": "Reasoning-based evaluation of generated videos",
      "relevance": "Use this work when comparing evaluation of generated-video quality and reasoning-based scoring. Compare dimensions and protocols rather than combining scores from different benchmarks.",
      "source_links": {
        "Project": "https://tiger-ai-lab.github.io/VideoScore2/",
        "Paper": "https://arxiv.org/abs/2509.22799",
        "Code": "https://github.com/TIGER-AI-Lab/VideoScore2/",
        "Dataset": "https://huggingface.co/datasets/TIGER-Lab/VideoFeedback2",
        "Model": "https://huggingface.co/TIGER-Lab/VideoScore2",
        "HF Space": "https://huggingface.co/spaces/TIGER-Lab/VideoScore2",
        "Twitter": "https://x.com/DongfuJiang/status/1973465380028031463"
      }
    },
    {
      "title": "Enhancing Vector Quantization with Distributional Matching: A Theoretical and Empirical Study",
      "task": "Distributional matching for vector quantization",
      "relevance": "Use this 2025 record for its theoretical and empirical treatment of distributional matching in vector quantization. The related 2026 record has a separate identifier and overlapping material.",
      "source_links": {
        "Paper": "https://arxiv.org/abs/2506.15078",
        "Code": "https://github.com/VQ-Research/Wasserstein-VQ"
      }
    },
    {
      "title": "Edge-Enhanced Dilated Residual Attention Network for Multimodal Medical Image Fusion",
      "task": "Edge-enhanced multimodal medical image fusion",
      "relevance": "Use this source for the edge-enhanced dilated residual attention fusion extension. Keep the original medical-fusion paper and this extension separately identified.",
      "source_links": {
        "Paper": "https://arxiv.org/abs/2411.11799",
        "Code": "https://github.com/simonZhou86/en_dran",
        "Dataset": "https://www.med.harvard.edu/aanlib/home.html"
      }
    },
    {
      "title": "S3Gym: Can LLMs Turn Self-Testing and Self-Judging into Self-Improvement?",
      "task": "Testing self-improvement through self-testing and self-judging",
      "relevance": "Cite this benchmark when examining whether an agent's self-testing and self-judging lead to measurable improvement through interaction.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2608.31100",
        "Project": "https://self-developing-agents.github.io/"
      }
    },
    {
      "title": "RewardHarness: Learning Human Preferences for Image Editing with Only 100 Demonstrations",
      "task": "Learning rewards for instruction-guided image editing",
      "relevance": "Cite the appropriate version when discussing learning human preferences for image editing. The homepage publication title and the saved preprint BibTeX may differ; both are exposed explicitly.",
      "source_links": {
        "Project": "https://rewardharness.com/",
        "arXiv": "https://arxiv.org/abs/2605.08703",
        "Code": "https://github.com/TIGER-AI-Lab/RewardHarness",
        "HuggingFace": "https://huggingface.co/papers/2605.08703"
      }
    },
    {
      "title": "Aspire: Can Models Self-Evolve from Vague Goals?",
      "task": "Agent self-evolution from vague goals",
      "relevance": "Use this work when discussing how agents interpret vague goals and construct a learning process, rather than assuming a fully specified objective.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2608.31111",
        "Project": "https://self-developing-agents.github.io/"
      }
    },
    {
      "title": "CompRank: Efficient LLM Reranking via Token-Level Compression and Decoding-Free Scoring",
      "task": "Efficient language-model reranking",
      "relevance": "Cite this method when discussing token-level compression and decoding-free scoring for LLM reranking. Check the reported retrieval tasks and cost measurements in the source.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2606.11700"
      }
    },
    {
      "title": "Where, What, Why, and Importance: Structured Defect Grounding for Text-to-Image Feedback",
      "task": "Localized defect feedback for text-to-image generation",
      "relevance": "Use this work for structured feedback about where a defect is, what it is, why it matters and its importance; consult the source for the grounding and feedback protocol.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2606.06113",
        "HuggingFace": "https://huggingface.co/papers/2606.06113",
        "Code": "https://github.com/nianbai006/SDG",
        "Dataset": "https://huggingface.co/datasets/P1n3/SDG-30K"
      }
    },
    {
      "title": "HarnessDev: Can LLMs Create and Evolve Their Own Agent Harness?",
      "task": "Creating and evolving model-external agent harnesses",
      "relevance": "Cite this benchmark when evaluating creation or evolution of agent execution infrastructure, keeping harness changes distinct from model-weight updates.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2609.01437",
        "Project": "https://self-developing-agents.github.io/"
      }
    },
    {
      "title": "MIRA: Mid-training Rubric Anchoring for Source-Aware Data Selection",
      "task": "Source-aware data selection for mid-training",
      "relevance": "Use this work when discussing rubric anchoring and source-aware selection of mid-training data; refer to the paper for the precise selection procedure.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2605.30288",
        "HuggingFace": "https://huggingface.co/papers/2605.30288"
      }
    },
    {
      "title": "WebWorld: The Browser as a World Model for Self-Improving Web Code",
      "task": "Browser-grounded evaluation for self-improving web code",
      "relevance": "Use this work when discussing browser execution and external feedback in web-code improvement, including the limitations of judging changes by visual plausibility.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2608.30530"
      }
    },
    {
      "title": "OpenSkill: Open-World Self-Evolution for LLM Agents",
      "task": "Open-world self-evolution for language-model agents",
      "relevance": "Cite this work when studying agent adaptation without assuming a curated learning loop or ready-made successful trajectories.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2606.06741",
        "Code": "https://github.com/OpenLAIR/OpenSkill",
        "HuggingFace": "https://huggingface.co/papers/2606.06741"
      }
    },
    {
      "title": "Watch Before You Answer: Learning from Visually Grounded Post-Training",
      "task": "Video post-training that depends on visual evidence",
      "relevance": "Cite this study when discussing linguistic shortcuts in video question answering or selecting post-training data for visual dependence. Check the paper's experimental scope before generalizing.",
      "source_links": {
        "Project": "https://vidground.etuagi.com/",
        "arXiv": "https://arxiv.org/abs/2604.05117",
        "Code": "https://github.com/reacher-z/vidground",
        "HuggingFace": "https://huggingface.co/papers/2604.05117",
        "X": "https://x.com/DongfuJiang/status/2042002570793898078"
      }
    },
    {
      "title": "Function-Aware Fill-in-the-Middle as Mid-Training for Coding Agent Foundation Models",
      "task": "Function-aware fill-in-the-middle training for coding agents",
      "relevance": "Use this method when discussing mid-training for integrating external tool returns into coding-agent reasoning, and consult the released data and model descriptions.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2607.12463",
        "Code": "https://github.com/TIGER-AI-Lab/FIM-Midtraining",
        "Dataset": "https://huggingface.co/datasets/TIGER-Lab/FIM-Midtraining-400K",
        "Model": "https://huggingface.co/collections/TIGER-Lab/fim-midtraining",
        "HuggingFace": "https://huggingface.co/papers/2607.12463"
      }
    },
    {
      "title": "MedClaw: Heuristic Agent Harness for Long-Horizon Surgical Video Reasoning",
      "task": "Long-horizon reasoning over surgical videos",
      "relevance": "Cite this research method for long-horizon video reasoning, planning and evidence retrieval. It is not evidence of clinical deployment or patient benefit; public resource availability must be checked separately.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2608.14015",
        "Project": "https://fyycs.github.io/medclaw/",
        "Dataset": "https://huggingface.co/datasets/gkw0010/SVU-31K"
      }
    },
    {
      "title": "ModularRSI: Toward Generalizable Harness RSI",
      "task": "Generalizable harness self-improvement",
      "relevance": "Reference the public project article or code for the disclosed work. No public manuscript is linked in this catalog, so this page does not present it as a published paper.",
      "source_links": {
        "Article": "https://recursive-self-improvement.notion.site/blog-1-modularrsi-toward-generalizable-harness-rsi",
        "Code": "https://github.com/IQuestLab/ModularRSI",
        "Dataset": "https://huggingface.co/datasets/IQuestLab/ModularRSI_2000_Instances"
      }
    },
    {
      "title": "Learning from the Self-future: On-policy Self-distillation for dLLMs",
      "task": "On-policy self-distillation for diffusion language models",
      "relevance": "Use this work when discussing application of on-policy self-distillation to diffusion language models and its proposed self-future learning approach.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2606.18195",
        "Code": "https://github.com/xingzhejun/d-opsd-code",
        "HuggingFace": "https://huggingface.co/papers/2606.18195"
      }
    },
    {
      "title": "Distributional Matching for Vector Quantization: A Unified Theoretical and Empirical Framework",
      "task": "A unified treatment of distributional matching in vector quantization",
      "relevance": "Use this 2026 record for the unified framework described in its current version. The related 2025 paper must not be counted as independent evidence without checking overlap.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2607.15933",
        "Related 2025 paper": "https://arxiv.org/abs/2506.15078",
        "Code": "https://github.com/VQ-Research/Wasserstein-VQ",
        "Project": "https://vq-research.github.io/Wasserstein-VQ/"
      }
    },
    {
      "title": "VGI-BENCH: Probing Visual Intelligence in Video Generation Models",
      "task": "Evaluating visual intelligence in video generation models",
      "relevance": "Cite this benchmark when investigating visual reasoning through video generation; distinguish its task-based evaluation from generic video appearance or quality scoring.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2608.19583",
        "Project": "https://hexuan21.github.io/VGI-Bench/",
        "HuggingFace": "https://huggingface.co/papers/2608.19583",
        "Code": "https://github.com/hexuan21/VGI-Bench",
        "Dataset": "https://huggingface.co/datasets/hexuan21/VGI-Bench"
      }
    },
    {
      "title": "Dr. Claw: An AI Scientist Workspace for Vibe Research",
      "task": "An AI scientist workspace for research workflows",
      "relevance": "Reference this system when discussing integrated research workspaces for coding agents. System availability and workflow capabilities should be checked against the current release.",
      "source_links": {
        "arXiv": "https://arxiv.org/abs/2609.00365",
        "Project": "https://openlair.github.io/dr-claw/",
        "Code": "https://github.com/OpenLAIR/dr-claw"
      }
    },
    {
      "title": "ClawBench: Can AI Agents Complete Everyday Online Tasks?",
      "task": "Evaluating browser agents on everyday online workflows",
      "relevance": "Use this benchmark when evaluating agents completing everyday online tasks on websites. Report the task setup and scoring protocol rather than equating a benchmark score with general autonomy.",
      "source_links": {
        "Project": "https://claw-bench.com",
        "arXiv": "https://arxiv.org/abs/2604.08523",
        "Code": "https://github.com/TIGER-AI-Lab/ClawBench",
        "Dataset": "https://huggingface.co/datasets/TIGER-Lab/ClawBench",
        "HuggingFace": "https://huggingface.co/papers/2604.08523",
        "X": "https://x.com/WenhuChen/status/2042484962428096939"
      }
    },
    {
      "title": "CelebHair: A New Large-Scale Dataset for Hairstyle Recommendation Based on CelebA",
      "task": "A dataset for hairstyle recommendation based on CelebA",
      "relevance": "Cite this dataset when using its hairstyle-recommendation annotations or task formulation. Consult the original work and dataset terms before reuse.",
      "source_links": {
        "Paper": "https://link.springer.com/chapter/10.1007/978-3-030-82153-1_27",
        "Video": "https://www.youtube.com/watch?v=hYm4kyfVcwc&ab_channel=Reacher"
      }
    }
  ]
}
