<?xml version="1.0" encoding="utf-8" standalone="yes"?>
<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">
	<channel>
		<title>推理优化 on 苏然的博客</title>
		<link>https://bml.asia/tags/%E6%8E%A8%E7%90%86%E4%BC%98%E5%8C%96/</link>
		<description>Recent content in 推理优化 on 苏然的博客</description>
		<generator>Hugo</generator>
		<language>zh-CN</language>
		
		
		
		
			<lastBuildDate>Thu, 17 Sep 2026 04:19:55 +0800</lastBuildDate>
		
			<atom:link href="https://bml.asia/tags/%E6%8E%A8%E7%90%86%E4%BC%98%E5%8C%96/index.xml" rel="self" type="application/rss+xml" />
			<item>
				<title>DeepSeek V4.1 Flash 内测观察：把速度卷到 507 tokens/s 的技术账</title>
				<link>https://bml.asia/posts/deepseek-v41-flash-speed/</link>
				<pubDate>Tue, 08 Sep 2026 19:27:37 +0800</pubDate>
				<guid>https://bml.asia/posts/deepseek-v41-flash-speed/</guid>
				<description>&lt;p&gt;&lt;span style=&#34;text-indent:2em;display:block;&#34;&gt;9 月 8 日下午，DeepSeek 在官方交流群里悄悄放出了 V4.1 Flash 的中间版本内测，没有发布会，也没有官网更新，只是一条群通知：新模型采用了新的结构，原生支持多模态，能力更强、速度更快、成本更低。开发者只要保持 base_url 不变，把模型名换成 deepseek-v4.1-flash-expires-on-0910 就能调用，每个账号限流 20 并发，计费与 V4 Flash 完全持平，到了 9 月 10 日这个模型名会自动过期下线。消息传开后，社交平台上很快出现了实测数据：有人跑出最高 507 tokens/s 的输出速度，也有人让模型生成一段鹈鹕骑自行车的 SVG 动画，稳定在 300 tokens/s 以上。作为参照，人类正常的阅读速度大约每秒 5 到 10 个字，这个速度意味着模型吐字的效率已经远远甩开了人的阅读上限。&lt;/span&gt;&lt;/p&gt;</description>
			</item>
	</channel>
</rss>
