{
  "schema": "vqv.terminal.bucket.v1",
  "generated_at": "2026-09-16T01:23:29.170475+00:00",
  "bucket": {
    "key": "llama-3",
    "name": "Llama-3",
    "type": "models",
    "count": 1,
    "latest_seen_at": "2026-09-12T09:15:59+00:00",
    "strong_count": 1,
    "signals": [
      {
        "title": "Study Finds Local LLM Judges Consistent but Not Always Aligned with Human Ratings",
        "summary": "Researchers evaluated local open-weight LLM judges LLaMA-3-8B and Qwen2.5-7B on 300 responses, finding that while these models produce consistent scores, they do not always agree with human evaluators. This highlights a gap between automated LLM evaluation and human judgment.",
        "why_it_matters": "Using LLMs as judges is faster and cheaper than human evaluation, but discrepancies with human ratings raise concerns about reliability. Understanding these differences is crucial for improving automated evaluation methods in AI development.",
        "why_this_is_here": "Why this is here: This signal is recent, source-backed, and connected to activity readers are already following in Open Source LLMs.",
        "label": "SOURCE-BACKED",
        "signal_strength": 95,
        "score": 77.32,
        "public_interest": {
          "score": 36,
          "editorial_category": "MONEY",
          "version": "v1",
          "reasons": [
            "editorial category: MONEY",
            "recognizable entity: Llama",
            "passes high signal-strength gate"
          ],
          "components": {
            "recognizable_entity_score": 67,
            "practical_impact_score": 8,
            "novelty_interest_score": 48,
            "consequence_score": 18,
            "curiosity_score": 0,
            "shareability_score": 53
          }
        },
        "what_this_means_for_you": "Business readers can use this as a signal of where capital, competition, or market attention is moving.",
        "reader_depth": "GENERAL",
        "editorial_freshness": {
          "band": "archive-level",
          "age_hours": 88.13
        },
        "topic": {
          "name": "Open Source LLMs",
          "slug": "open-source-llms",
          "url": "/t/open-source-llms/"
        },
        "source": {
          "name": "arXiv",
          "type": "arxiv",
          "domain": "arxiv.org",
          "url": "http://arxiv.org/abs/2609.13824v1",
          "source_page": "/source/arxiv/"
        },
        "urls": {
          "signal_page": "/s/XK4Yn/",
          "topic_anchor": "/t/open-source-llms/#signal-2ccc117c34",
          "short_url": "https://vqv.me/XK4Yn"
        },
        "share_text": "Local LLM judges like LLaMA-3-8B show consistent scoring but don\u2019t always match human evaluations, highlighting challenges in automated AI assessment.",
        "reposts": 0,
        "published_at": "2026-09-12T09:15:59+00:00",
        "fetched_at": "2026-09-16T01:19:37+00:00",
        "ai_assisted_summary": true,
        "id": "2ccc117c34",
        "url_hash": "8fbe4a4976ebb57ae867c4cdde51dc1fe396e17dfe43ed4dafe69e529eef37d5",
        "short_code": "XK4Yn",
        "source_title": "When Consistency Does Not Mean Reliability: Evaluating Local LLM Judges Against Human Ratings",
        "source_snippet": "Large language models (LLMs) are increasingly used to evaluate the responses of other language models. This approach, known as LLM-as-a-Judge, is faster and cheaper than human evaluation. However, a judge may produce consistent scores without necessarily agreeing with human evaluators. In this work, we study this...",
        "seen_at": "2026-09-12T09:15:59+00:00",
        "prominence_tier": "primary",
        "buckets": {
          "companies": [],
          "models": [
            "Llama",
            "Llama-3"
          ],
          "products": [],
          "topic": "open-source-llms",
          "reader_depth": "GENERAL",
          "editorial_category": "MONEY"
        }
      }
    ]
  }
}
