feat(mcp): render voice messages with duration in chat history (#97)
## Problem
Voice messages in `_format_message_text` previously rendered as a bare
`[语音] (local_id=N, ts=T)` because msg_type=34 fell through to the generic
non-text branch with no schema-aware summarizer. LLMs reading chat history
had no way to judge whether a voice clip was worth calling `decode_voice`
on without first inspecting it.
## Fix
New helper `_format_voice_text(content)` parses the embedded
`<voicemsg voicelength="…">` and renders `[语音 N.Ns]` (duration to one
decimal, milliseconds → seconds). Type=34 dispatches through it, then
appends the existing `_id_suffix()` so the local_id annotation is
preserved end-to-end:
[语音 3.3s] (local_id=72481, ts=1700000000)
Falls back to `[语音]` (still with `_id_suffix()`) when content is empty,
`<voicemsg>` is absent, XML parse fails, or `voicelength` is missing /
zero / non-numeric.
XML parsing routes through the existing `_parse_xml_root` so the
`_XML_UNSAFE_RE` DOCTYPE/ENTITY filter and 200KB size cap are reused —
no new XXE surface.
## Tests
12 new cases in `tests/test_voice_format.py`: happy path, subsecond,
multi-second, missing / zero / non-numeric voicelength, empty / None
content, missing `<voicemsg>` tag, malformed XML, XXE payload, and two
end-to-end cases through `_format_message_text` (with and without
voicelength) to pin the full rendered output including `_id_suffix()`.
Baseline 183 → 195 passing, 0 regressions.
## Scope
- `mcp_server.py`: adds `_format_voice_text` helper and one branch in
`_format_message_text` (base_type == 34). No public surface change —
this only affects formatting of messages that previously rendered as
the bare `[语音]` fallback.
- `tests/test_voice_format.py`: new file, synthetic fixtures only (no
real PII).
This commit is contained in:
@@ -973,6 +973,21 @@ def _format_voip_message_text(content):
|
||||
return f"[通话] {status_map.get(raw_text, raw_text)}"
|
||||
|
||||
|
||||
def _format_voice_text(content):
|
||||
if not content or '<voicemsg' not in content:
|
||||
return "[语音]"
|
||||
root = _parse_xml_root(content)
|
||||
if root is None:
|
||||
return "[语音]"
|
||||
voice = root.find('.//voicemsg')
|
||||
if voice is None:
|
||||
return "[语音]"
|
||||
length_ms = _parse_int(voice.get('voicelength'), 0)
|
||||
if length_ms <= 0:
|
||||
return "[语音]"
|
||||
return f"[语音 {length_ms / 1000:.1f}s]"
|
||||
|
||||
|
||||
def _format_message_text(local_id, local_type, content, is_group, chat_username, chat_display_name, names, create_time=0):
|
||||
sender_from_content, text = _parse_message_content(content, local_type, is_group)
|
||||
base_type, _ = _split_msg_type(local_type)
|
||||
@@ -985,6 +1000,8 @@ def _format_message_text(local_id, local_type, content, is_group, chat_username,
|
||||
|
||||
if base_type == 3:
|
||||
text = f"[图片] {_id_suffix()}"
|
||||
elif base_type == 34:
|
||||
text = f"{_format_voice_text(text)} {_id_suffix()}"
|
||||
elif base_type == 47:
|
||||
text = "[表情]"
|
||||
elif base_type == 50:
|
||||
|
||||
82
tests/test_voice_format.py
Normal file
82
tests/test_voice_format.py
Normal file
@@ -0,0 +1,82 @@
|
||||
"""Tests for `_format_voice_text` (msg_type=34 鉴定).
|
||||
|
||||
Voice messages previously rendered as a bare `[语音]` from the generic
|
||||
non-text branch. LLMs reading chat history had no way to (a) judge whether a
|
||||
clip was worth transcribing, or (b) call `decode_voice` without first round-
|
||||
tripping through `get_voice_messages` to look up the `local_id`.
|
||||
|
||||
These tests pin the new behaviour: `[语音 Ns]` (duration to 1 decimal) when
|
||||
the embedded `<voicemsg voicelength="…">` is parseable, with graceful
|
||||
fallback to `[语音]` on missing / zero / malformed length.
|
||||
"""
|
||||
import unittest
|
||||
|
||||
import mcp_server
|
||||
|
||||
|
||||
def _voice_xml(length_ms):
|
||||
return (
|
||||
f'<msg><voicemsg endflag="1" length="2048" voicelength="{length_ms}" '
|
||||
'clientmsgid="abc" fromusername="wxid_synth_a" '
|
||||
'cancelflag="0" voiceformat="4" forwardflag="0" /></msg>'
|
||||
)
|
||||
|
||||
|
||||
class FormatVoiceTextTests(unittest.TestCase):
|
||||
def test_renders_duration_with_one_decimal(self):
|
||||
self.assertEqual(mcp_server._format_voice_text(_voice_xml(3300)), "[语音 3.3s]")
|
||||
|
||||
def test_subsecond_voice(self):
|
||||
self.assertEqual(mcp_server._format_voice_text(_voice_xml(800)), "[语音 0.8s]")
|
||||
|
||||
def test_long_clip(self):
|
||||
self.assertEqual(mcp_server._format_voice_text(_voice_xml(62000)), "[语音 62.0s]")
|
||||
|
||||
def test_missing_voicelength_falls_back(self):
|
||||
xml = '<msg><voicemsg endflag="1" length="2048" /></msg>'
|
||||
self.assertEqual(mcp_server._format_voice_text(xml), "[语音]")
|
||||
|
||||
def test_zero_voicelength_falls_back(self):
|
||||
self.assertEqual(mcp_server._format_voice_text(_voice_xml(0)), "[语音]")
|
||||
|
||||
def test_non_numeric_voicelength_falls_back(self):
|
||||
xml = '<msg><voicemsg voicelength="abc" /></msg>'
|
||||
self.assertEqual(mcp_server._format_voice_text(xml), "[语音]")
|
||||
|
||||
def test_empty_content(self):
|
||||
self.assertEqual(mcp_server._format_voice_text(""), "[语音]")
|
||||
self.assertEqual(mcp_server._format_voice_text(None), "[语音]")
|
||||
|
||||
def test_missing_voicemsg_tag(self):
|
||||
self.assertEqual(mcp_server._format_voice_text("<msg></msg>"), "[语音]")
|
||||
|
||||
def test_malformed_xml(self):
|
||||
self.assertEqual(mcp_server._format_voice_text("<msg><voicemsg"), "[语音]")
|
||||
|
||||
def test_xxe_payload_rejected(self):
|
||||
xxe = (
|
||||
'<!DOCTYPE foo [<!ENTITY x SYSTEM "file:///etc/passwd">]>'
|
||||
'<msg><voicemsg voicelength="1000" /></msg>'
|
||||
)
|
||||
self.assertEqual(mcp_server._format_voice_text(xxe), "[语音]")
|
||||
|
||||
def test_end_to_end_format_message_text_with_voicelength(self):
|
||||
xml = _voice_xml(3300)
|
||||
_, text = mcp_server._format_message_text(
|
||||
local_id=72481, local_type=34, content=xml, is_group=False,
|
||||
chat_username="wxid_synth_a", chat_display_name="A", names={},
|
||||
create_time=1700000000,
|
||||
)
|
||||
self.assertEqual(text, "[语音 3.3s] (local_id=72481, ts=1700000000)")
|
||||
|
||||
def test_end_to_end_without_voicelength(self):
|
||||
_, text = mcp_server._format_message_text(
|
||||
local_id=99, local_type=34, content="<msg></msg>", is_group=False,
|
||||
chat_username="wxid_synth_a", chat_display_name="A", names={},
|
||||
create_time=0,
|
||||
)
|
||||
self.assertEqual(text, "[语音] (local_id=99)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user