-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathV3RealtimeAsrExample.java
More file actions
149 lines (134 loc) · 6.62 KB
/
Copy pathV3RealtimeAsrExample.java
File metadata and controls
149 lines (134 loc) · 6.62 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
package com.tencent.trtcasr.examples;
import java.nio.file.Files;
import java.nio.file.Path;
import java.time.Duration;
import com.tencent.trtcasr.asr.SpeechRecognitionListener;
import com.tencent.trtcasr.asr.SpeechRecognitionResponse;
import com.tencent.trtcasr.common.ASRException;
import com.tencent.trtcasr.common.Credential;
import com.tencent.trtcasr.v3.SpeechRecognizer;
import com.tencent.trtcasr.v3.Wire;
/**
* Realtime speech recognition example over the v3 protocol (/asr/v3).
*
* <p>Reads a PCM file (16kHz 16bit mono) and streams it in 200ms chunks.
*
* <p>Credentials come from environment variables:
* {@code TRTC_ASR_SDK_APP_ID}, {@code TRTC_ASR_SECRET_KEY}
* (v3 does not need the Tencent Cloud APPID.)
*
* <p>Usage: {@code java ... V3RealtimeAsrExample <audio.pcm> [engine] [lang] [speakerContext]
* [speakerContextId]}
*
* <p>Speaker diarization can be made resumable across connections
* ("断点续传"): keep the printed speaker_context_id and pass it back on the
* next connection, e.g.
* {@code V3RealtimeAsrExample test.pcm bigmodel zh 1} then
* {@code V3RealtimeAsrExample test.pcm bigmodel zh 1 <speaker_context_id>}.
*/
public class V3RealtimeAsrExample {
private static final SpeechRecognitionListener LISTENER = new SpeechRecognitionListener() {
@Override
public void onRecognitionStart(SpeechRecognitionResponse response) {
System.out.println("[start] voice_id=" + response.getVoiceId());
if (response.getSpeakerContinue() != null) {
System.out.println("speaker context: status=\""
+ response.getSpeakerContinue().getContinueStatus()
+ "\" id=" + response.getSpeakerContinue().getSpeakerContextId());
}
}
@Override
public void onSentenceEnd(SpeechRecognitionResponse response) {
System.out.println("[end] " + response.getResult().getVoiceTextStr());
for (SpeechRecognitionResponse.SpeakerSegment seg : response.getResult().getSpeakerSegments()) {
String label = seg.getSpeakerName() == null || seg.getSpeakerName().isEmpty()
? "spk" + seg.getSpeakerId() : seg.getSpeakerName();
System.out.println(" [" + label + "] " + seg.getText()
+ " (" + seg.getStartTime() + "-" + seg.getEndTime() + " ms)");
}
}
@Override
public void onRecognitionResultChange(SpeechRecognitionResponse response) {
System.out.println("[change] " + response.getResult().getVoiceTextStr());
}
@Override
public void onRecognitionComplete(SpeechRecognitionResponse response) {
System.out.println("[complete]");
}
@Override
public void onFail(SpeechRecognitionResponse response, ASRException error) {
System.err.println("[fail] " + error);
}
};
public static void main(String[] args) throws Exception {
long sdkAppId = Long.parseLong(env("TRTC_ASR_SDK_APP_ID", "0"));
String secretKey = env("TRTC_ASR_SECRET_KEY", "");
if (sdkAppId == 0 || secretKey.isEmpty()) {
System.err.println("Set TRTC_ASR_SDK_APP_ID and TRTC_ASR_SECRET_KEY first.");
System.exit(1);
}
String path = args.length > 0 ? args[0] : "examples/test.pcm";
if (args.length <= 1) {
System.err.println("error: engine is required (engine model type, e.g. bigmodel)");
System.exit(2);
}
String engine = args[1];
// Empty lang falls back to server-side language detection.
String lang = args.length > 2 ? args[2] : "";
// The bigmodel engine is best used with an explicit language; every
// other engine falls back to server-side detection unless lang is given.
if (lang.isEmpty() && "bigmodel".equals(engine)) {
lang = "zh";
}
// v3 credentials need only SDKAppID + SecretKey (no Tencent Cloud APPID).
Credential credential = Wire.newCredential(sdkAppId, secretKey);
// arg[3]: resumable speaker diarization (0 off / 1 sync / 2 async);
// arg[4]: the speaker_context_id printed by an earlier run.
int speakerContext = args.length > 3 ? Integer.parseInt(args[3]) : 0;
String speakerContextId = args.length > 4 ? args[4] : "";
SpeechRecognizer recognizer = new SpeechRecognizer(credential, engine, LISTENER);
if (!lang.isEmpty()) {
recognizer.setLanguage(lang); // bigmodel 建议显式指定语种
}
// Speaker-context persistence requires speaker diarization; the SDK
// rejects the combination locally otherwise.
if (speakerContext != 0) {
recognizer.setSpeakerDiarization(Wire.SPEAKER_DIARIZATION_CLUSTER);
recognizer.setEnableSpeakerContext(speakerContext);
}
if (!speakerContextId.isEmpty()) {
recognizer.setSpeakerContextId(speakerContextId);
}
recognizer.setWriteTimeout(Duration.ofSeconds(5));
recognizer.setStopTimeout(Duration.ofSeconds(10));
// start() waits synchronously for the server's ack: authentication
// (4002) and gray-switch (4001) errors are thrown here, not via onFail.
// While resuming a speaker context the ack arrives once the server has
// applied the stored snapshot.
recognizer.start();
// The handshake result is also available without a callback. Persist
// the id: passing it back keeps speaker ids stable across connections.
if (recognizer.getSpeakerContinue() != null) {
System.out.println("speaker context id: "
+ recognizer.getSpeakerContinue().getSpeakerContextId()
+ " (status: \"" + recognizer.getSpeakerContinue().getContinueStatus()
+ "\") — reuse it as the 5th argument");
}
byte[] data = Files.readAllBytes(Path.of(path));
final int sliceSize = 6400; // 200ms of 16kHz 16bit mono PCM
// The server rate-limits to at most 3s of audio per 1s wall-clock
// (error 4000): keep the pacing when enlarging the buffer.
for (int offset = 0; offset < data.length; offset += sliceSize) {
int len = Math.min(sliceSize, data.length - offset);
byte[] chunk = new byte[len];
System.arraycopy(data, offset, chunk, 0, len);
recognizer.write(chunk);
Thread.sleep(200);
}
recognizer.stop();
}
private static String env(String name, String fallback) {
String v = System.getenv(name);
return v == null || v.isEmpty() ? fallback : v;
}
}