<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
  <channel>
    <title>ai on Juanjo&#39;s blog</title>
    <link>https://juanjo.garciaamaya.com/tags/ai/</link>
    <description>Recent content in ai on Juanjo&#39;s blog</description>
    <generator>Hugo -- gohugo.io</generator>
    <language>en-us</language>
    <lastBuildDate>Fri, 07 Aug 2026 22:30:00 +0200</lastBuildDate><atom:link href="https://juanjo.garciaamaya.com/tags/ai/index.xml" rel="self" type="application/rss+xml" />
    <item>
      <title>Serving Local LLMs with llama.cpp, Docker, and OpenCode on an RTX 5060 Ti</title>
      <link>https://juanjo.garciaamaya.com/posts/ai/serving-local-llms-llamacpp-docker-opencode/</link>
      <pubDate>Fri, 07 Aug 2026 22:30:00 +0200</pubDate>
      
      <guid>https://juanjo.garciaamaya.com/posts/ai/serving-local-llms-llamacpp-docker-opencode/</guid>
      <description>Serving Local LLMs with llama.cpp, Docker, and OpenCode This post documents setting up a local LLM inference server on my Linux desktop (RTX 5060 Ti, 16GB VRAM), reusing models already downloaded through LM Studio, and wiring it into OpenCode for agentic coding work.</description>
    </item>
    
  </channel>
</rss>
