feat(mcp): render voice messages with duration in chat history (#97)

## Problem

Voice messages in `_format_message_text` previously rendered as a bare
`[语音] (local_id=N, ts=T)` because msg_type=34 fell through to the generic
non-text branch with no schema-aware summarizer. LLMs reading chat history
had no way to judge whether a voice clip was worth calling `decode_voice`
on without first inspecting it.

## Fix

New helper `_format_voice_text(content)` parses the embedded
`<voicemsg voicelength="…">` and renders `[语音 N.Ns]` (duration to one
decimal, milliseconds → seconds). Type=34 dispatches through it, then
appends the existing `_id_suffix()` so the local_id annotation is
preserved end-to-end:

    [语音 3.3s] (local_id=72481, ts=1700000000)

Falls back to `[语音]` (still with `_id_suffix()`) when content is empty,
`<voicemsg>` is absent, XML parse fails, or `voicelength` is missing /
zero / non-numeric.

XML parsing routes through the existing `_parse_xml_root` so the
`_XML_UNSAFE_RE` DOCTYPE/ENTITY filter and 200KB size cap are reused —
no new XXE surface.

## Tests

12 new cases in `tests/test_voice_format.py`: happy path, subsecond,
multi-second, missing / zero / non-numeric voicelength, empty / None
content, missing `<voicemsg>` tag, malformed XML, XXE payload, and two
end-to-end cases through `_format_message_text` (with and without
voicelength) to pin the full rendered output including `_id_suffix()`.

Baseline 183 → 195 passing, 0 regressions.

## Scope

- `mcp_server.py`: adds `_format_voice_text` helper and one branch in
  `_format_message_text` (base_type == 34). No public surface change —
  this only affects formatting of messages that previously rendered as
  the bare `[语音]` fallback.
- `tests/test_voice_format.py`: new file, synthetic fixtures only (no
  real PII).
This commit is contained in:
Belugary
2026-05-13 13:31:18 +08:00
committed by GitHub
parent 8bb2d85d8c
commit 187d820bb0
2 changed files with 99 additions and 0 deletions

View File

@@ -0,0 +1,82 @@
"""Tests for `_format_voice_text` (msg_type=34 鉴定).
Voice messages previously rendered as a bare `[语音]` from the generic
non-text branch. LLMs reading chat history had no way to (a) judge whether a
clip was worth transcribing, or (b) call `decode_voice` without first round-
tripping through `get_voice_messages` to look up the `local_id`.
These tests pin the new behaviour: `[语音 Ns]` (duration to 1 decimal) when
the embedded `<voicemsg voicelength="">` is parseable, with graceful
fallback to `[语音]` on missing / zero / malformed length.
"""
import unittest
import mcp_server
def _voice_xml(length_ms):
return (
f'<msg><voicemsg endflag="1" length="2048" voicelength="{length_ms}" '
'clientmsgid="abc" fromusername="wxid_synth_a" '
'cancelflag="0" voiceformat="4" forwardflag="0" /></msg>'
)
class FormatVoiceTextTests(unittest.TestCase):
def test_renders_duration_with_one_decimal(self):
self.assertEqual(mcp_server._format_voice_text(_voice_xml(3300)), "[语音 3.3s]")
def test_subsecond_voice(self):
self.assertEqual(mcp_server._format_voice_text(_voice_xml(800)), "[语音 0.8s]")
def test_long_clip(self):
self.assertEqual(mcp_server._format_voice_text(_voice_xml(62000)), "[语音 62.0s]")
def test_missing_voicelength_falls_back(self):
xml = '<msg><voicemsg endflag="1" length="2048" /></msg>'
self.assertEqual(mcp_server._format_voice_text(xml), "[语音]")
def test_zero_voicelength_falls_back(self):
self.assertEqual(mcp_server._format_voice_text(_voice_xml(0)), "[语音]")
def test_non_numeric_voicelength_falls_back(self):
xml = '<msg><voicemsg voicelength="abc" /></msg>'
self.assertEqual(mcp_server._format_voice_text(xml), "[语音]")
def test_empty_content(self):
self.assertEqual(mcp_server._format_voice_text(""), "[语音]")
self.assertEqual(mcp_server._format_voice_text(None), "[语音]")
def test_missing_voicemsg_tag(self):
self.assertEqual(mcp_server._format_voice_text("<msg></msg>"), "[语音]")
def test_malformed_xml(self):
self.assertEqual(mcp_server._format_voice_text("<msg><voicemsg"), "[语音]")
def test_xxe_payload_rejected(self):
xxe = (
'<!DOCTYPE foo [<!ENTITY x SYSTEM "file:///etc/passwd">]>'
'<msg><voicemsg voicelength="1000" /></msg>'
)
self.assertEqual(mcp_server._format_voice_text(xxe), "[语音]")
def test_end_to_end_format_message_text_with_voicelength(self):
xml = _voice_xml(3300)
_, text = mcp_server._format_message_text(
local_id=72481, local_type=34, content=xml, is_group=False,
chat_username="wxid_synth_a", chat_display_name="A", names={},
create_time=1700000000,
)
self.assertEqual(text, "[语音 3.3s] (local_id=72481, ts=1700000000)")
def test_end_to_end_without_voicelength(self):
_, text = mcp_server._format_message_text(
local_id=99, local_type=34, content="<msg></msg>", is_group=False,
chat_username="wxid_synth_a", chat_display_name="A", names={},
create_time=0,
)
self.assertEqual(text, "[语音] (local_id=99)")
if __name__ == "__main__":
unittest.main()