<?xml version="1.0" encoding="UTF-8"?>
<urlset xmlns="http://www.sitemaps.org/schemas/sitemap/0.9"
        xmlns:news="http://www.google.com/schemas/sitemap-news/0.9">
  <url>
    <loc>https://ainews.surl.tw/article/aispa-稽核商用-ai-的隱藏系統提示-約四成產品含損害使用者利益的指令-f55f8414</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T18:02:26.696Z</news:publication_date>
      <news:title>AISPA 稽核商用 AI 的隱藏系統提示，約四成產品含損害使用者利益的指令</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/osreward-檢驗-27-款電腦代理裁判-困難軌跡暴露普遍-誤判成功-偏差-c3062a5d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T18:01:32.845Z</news:publication_date>
      <news:title>OSReward 檢驗 27 款電腦代理裁判，困難軌跡暴露普遍「誤判成功」偏差</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/owasp-agentic-skills-top-10-完成-v1-公開審查-將技能供應鏈列為獨立攻擊面-92f15410</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T16:03:07.408Z</news:publication_date>
      <news:title>OWASP Agentic Skills Top 10 完成 v1 公開審查，將技能供應鏈列為獨立攻擊面</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/pocket-將-35b-稀疏-moe-壓至-8-2-gb-標準-llama-cpp-可在手機與純-cpu-執行-ddf86b29</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T16:02:04.729Z</news:publication_date>
      <news:title>POCKET 將 35B 稀疏 MoE 壓至 8.2 GB，標準 llama.cpp 可在手機與純 CPU 執行</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/google-earth-上線影像生成一天即撤回-synthid-未能阻止假衛星圖外流-7ec17f8e</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T12:02:53.218Z</news:publication_date>
      <news:title>Google Earth 上線影像生成一天即撤回，SynthID 未能阻止假衛星圖外流</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/openpangu-2-0-pro-開放-505b-moe-權重-以-18b-啟用參數支援-512k-上下文-a99e272b</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T12:02:01.422Z</news:publication_date>
      <news:title>openPangu 2.0 Pro 開放 505B MoE 權重，以 18B 啟用參數支援 512K 上下文</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/unicon-以-4-160-萬參數統一九類數值系統-凍結權重也能跨領域預測-a0319c98</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T10:02:26.190Z</news:publication_date>
      <news:title>UNICON 以 4,160 萬參數統一九類數值系統，凍結權重也能跨領域預測</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/五款開放權重模型的自報信心近乎常數-低信心優先稽核可能等同隨機抽查-1b0a0d93</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T10:03:46.374Z</news:publication_date>
      <news:title>五款開放權重模型的自報信心近乎常數，低信心優先稽核可能等同隨機抽查</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/orca-bench-重建六天遙測現場-前沿代理的中等難度根因分析最高僅-25-3-75668856</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T08:01:51.433Z</news:publication_date>
      <news:title>ORCA-bench 重建六天遙測現場，前沿代理的中等難度根因分析最高僅 25.3%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/retoken-用單一可學習-token-篩選視覺-kv-cache-長影片問答提高-8-0-點-d05ded77</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T08:02:48.276Z</news:publication_date>
      <news:title>ReToken 用單一可學習 token 篩選視覺 KV cache，長影片問答提高 8.0 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/mistral-一批舊模型於-7-月-31-日退出服務-程式代理與結構化輸出須重做回歸測試-ecb30c49</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T06:02:58.222Z</news:publication_date>
      <news:title>Mistral 一批舊模型於 7 月 31 日退出服務，程式代理與結構化輸出須重做回歸測試</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/sagemaker-hyperpod-inference-v3-1-開放自訂-pod-憑證與逐-pod-請求上限-3e821f5b</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T06:02:00.689Z</news:publication_date>
      <news:title>SageMaker HyperPod Inference v3.1 開放自訂 Pod、憑證與逐 Pod 請求上限</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/mlperf-endpoints-v0-7-上線-為代理推論加入持續提交與可比較結果管線-7691f96d</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T04:02:33.853Z</news:publication_date>
      <news:title>MLPerf Endpoints v0.7 上線，為代理推論加入持續提交與可比較結果管線</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/leancsp-以-lean-驗證限制式重寫與求解器證明-搜尋量最高縮減兩千萬倍-cda8a9d6</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T02:02:42.009Z</news:publication_date>
      <news:title>LeanCSP 以 Lean 驗證限制式重寫與求解器證明，搜尋量最高縮減兩千萬倍</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/pcd-將多模態失敗拆成感知與推理訊號-32b-到-8b-蒸餾平均分升至-61-22-ab391e21</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T02:03:35.591Z</news:publication_date>
      <news:title>PCD 將多模態失敗拆成感知與推理訊號，32B 到 8B 蒸餾平均分升至 61.22</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/歐盟-ai-act-第-50-條開始適用-生成內容須加入機器可讀標記-a9453fe9</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T00:03:33.898Z</news:publication_date>
      <news:title>歐盟 AI Act 第 50 條開始適用，生成內容須加入機器可讀標記</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/deepseek-v4-flash-0731-只重做後訓練-terminal-bench-2-1-官方成績升至-82-7-33bf3334</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-02T00:02:17.290Z</news:publication_date>
      <news:title>DeepSeek-V4-Flash 0731 只重做後訓練，Terminal-Bench 2.1 官方成績升至 82.7</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/glm-rag-把知識圖文字送入圖式-transformer-跨領域多跳檢索優於-gnn-基線-43c605ab</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T22:02:46.979Z</news:publication_date>
      <news:title>GLM-RAG 把知識圖文字送入圖式 Transformer，跨領域多跳檢索優於 GNN 基線</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dfot-量化-llm-對衍生感測值的過度信任-跨模態蒸餾讓修正率提高最多-6-69-點-96edf0df</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T22:01:51.857Z</news:publication_date>
      <news:title>DFOT 量化 LLM 對衍生感測值的過度信任，跨模態蒸餾讓修正率提高最多 6.69 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/dualg-mrag-將多模態知識圖拆成兩層-mmqa-的-r-5-提高至-61-9-9a7c91ce</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T20:02:11.398Z</news:publication_date>
      <news:title>DualG-MRAG 將多模態知識圖拆成兩層，MMQA 的 R@5 提高至 61.9%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/grsd-對照代理自身成敗軌跡分配獎勵-alfworld-未見任務成功率達-83-6-12236df5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T20:03:09.148Z</news:publication_date>
      <news:title>GRSD 對照代理自身成敗軌跡分配獎勵，ALFWorld 未見任務成功率達 83.6%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/gcc-拒收具著作權意義的-llm-衍生內容-生成測試可由維護者例外接受-2964ce6f</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T18:04:03.038Z</news:publication_date>
      <news:title>GCC 拒收具著作權意義的 LLM 衍生內容，生成測試可由維護者例外接受</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/change2task-將歷史-pr-搬到新版程式碼-900-項代理任務通過三階段驗證-3dd58809</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T18:02:47.082Z</news:publication_date>
      <news:title>Change2Task 將歷史 PR 搬到新版程式碼，900 項代理任務通過三階段驗證</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/memharness-在行動前重寫代理記憶-7b-模型於-alfworld-成功率升至-85-2-06d37fd1</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T10:02:42.966Z</news:publication_date>
      <news:title>MemHarness 在行動前重寫代理記憶，7B 模型於 ALFWorld 成功率升至 85.2%</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/grafana-agent-observability-正式上線-以開源-sdk-串接線上評分與代理-ci-8346a016</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T12:02:15.638Z</news:publication_date>
      <news:title>Grafana Agent Observability 正式上線，以開源 SDK 串接線上評分與代理 CI</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/scidatasailor-讓代理直接探索科學資料庫-9b-模型-pass-1-由-14-01-升至-28-99-22563bf5</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T06:03:46.649Z</news:publication_date>
      <news:title>SciDataSailor 讓代理直接探索科學資料庫，9B 模型 Pass@1 由 14.01 升至 28.99</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/repocompliancebench-四款程式代理幾乎不會主動讀取開源專案的-ai-貢獻規則-9b4b343b</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T08:03:50.077Z</news:publication_date>
      <news:title>RepoComplianceBench：四款程式代理幾乎不會主動讀取開源專案的 AI 貢獻規則</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/infoops-bench-每週更新國家宣傳題庫-17-款模型拒答率相差-85-7-點-e85bae61</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T06:02:17.391Z</news:publication_date>
      <news:title>InfoOps Bench 每週更新國家宣傳題庫，17 款模型拒答率相差 85.7 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/openai-評測代理連鎖利用零日漏洞-從隔離環境闖入-hugging-face-生產系統-c79e0909</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T04:02:37.673Z</news:publication_date>
      <news:title>OpenAI 評測代理連鎖利用零日漏洞，從隔離環境闖入 Hugging Face 生產系統</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/manta-在推論途中重組多代理拓撲-五項基準平均提高-5-8-點-0f54c388</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-07-31T22:02:04.435Z</news:publication_date>
      <news:title>MANTA 在推論途中重組多代理拓撲，五項基準平均提高 5.8 點</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/m4-max-本機-llm-實測-解碼快於-gb10-但完整回應仍受-prefill-制約-2b1ecde2</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-08-01T02:02:31.406Z</news:publication_date>
      <news:title>M4 Max 本機 LLM 實測：解碼快於 GB10，但完整回應仍受 prefill 制約</news:title>
    </news:news>
  </url>
  <url>
    <loc>https://ainews.surl.tw/article/svr-讓-2b-模型自行判斷何時停止修訂-平均-2-99-輪完成數學推理-c8046bff</loc>
    <news:news>
      <news:publication>
        <news:name>SURL AI News</news:name>
        <news:language>zh-tw</news:language>
      </news:publication>
      <news:publication_date>2026-07-31T22:02:48.543Z</news:publication_date>
      <news:title>SVR 讓 2B 模型自行判斷何時停止修訂，平均 2.99 輪完成數學推理</news:title>
    </news:news>
  </url>
</urlset>
