오디오 이해
오디오 이해
Gemini는 오디오 입력을 분석하고 텍스트 응답을 생성할 수 있어요.
from google import genai
import base64
client = genai.Client()
uploaded_file = client.files.upload(file="path/to/sample.mp3")
interaction = client.interactions.create(
model="gemini-3.8-flash",
input=[
{"type": "text", "text": "Describe this audio clip"},
{
"type": "audio",
"uri": uploaded_file.uri,
"mime_type": uploaded_file.mime_type
}
]
)
print(interaction.output_text)
import { GoogleGenAI } from "@google/genai";
const client = new GoogleGenAI({});
const uploadedFile = await client.files.upload({
file: "path/to/sample.mp3",
config: { mime_type: "audio/mp3" }
});
const interaction = await client.interactions.create({
model: "gemini-3.8-flash",
input: [
{type: "text", text: "Describe this audio clip"},
{
type: "audio",
uri: uploadedFile.uri,
mime_type: uploadedFile.mimeType
}
]
});
console.log(interaction.output_text);
import com.google.genai.Client;
import com.google.genai.gaos.models.interactions.AudioContent;
import com.google.genai.gaos.models.interactions.AudioContentMimeType;
import com.google.genai.gaos.models.interactions.Content;
import com.google.genai.gaos.models.interactions.CreateModelInteraction;
import com.google.genai.gaos.models.interactions.Interaction;
import com.google.genai.gaos.models.interactions.InteractionsInput;
import com.google.genai.gaos.models.interactions.Model;
import com.google.genai.gaos.models.interactions.TextContent;
import com.google.genai.gaos.models.operations.CreateInteractionRequestBody;
import com.google.genai.types.File;
import com.google.genai.types.UploadFileConfig;
import java.util.Arrays;
import java.util.List;
Client client = new Client();
File uploadedFile =
client.files.upload(
"path/to/sample.mp3", UploadFileConfig.builder().mimeType("audio/mp3").build());
Content textContent = TextContent.builder().text("Describe this audio clip").build();
Content audioContent =
AudioContent.builder()
.uri(uploadedFile.uri().get())
.mimeType(AudioContentMimeType.of(uploadedFile.mimeType().get()))
.build();
List<Content> contents = Arrays.asList(textContent, audioContent);
CreateModelInteraction params =
CreateModelInteraction.builder()
.model(Model.of("gemini-3.8-flash"))
.input(InteractionsInput.ofContent(contents))
.build();
Interaction interaction =
client.interactions.create(CreateInteractionRequestBody.of(params)).interaction().get();
System.out.println(interaction.outputText().orElse(""));
package main
import (
"context"
"fmt"
"log"
"google.golang.org/genai"
"google.golang.org/genai/interactions/models/interactions"
"google.golang.org/genai/interactions/models/operations"
)
func main() {
ctx := context.Background()
client, err := genai.NewClient(ctx, nil)
if err != nil {
log.Fatal(err)
}
uploadedFile, err := client.Files.UploadFromPath(ctx, "path/to/sample.mp3", &genai.UploadFileConfig{
MIMEType: "audio/mp3",
})
if err != nil {
log.Fatal(err)
}
contents := []interactions.Content{
interactions.NewContent(interactions.TextContent{
Text: "Describe this audio clip",
}),
interactions.NewContent(interactions.AudioContent{
URI: genai.Ptr(uploadedFile.URI),
MimeType: interactions.AudioContentMimeType(uploadedFile.MIMEType).ToPointer(),
}),
}
res, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
Model: interactions.Model("gemini-3.8-flash"),
Input: interactions.NewInteractionsInput(contents),
}),
})
if err != nil {
log.Fatal(err)
}
if res.Interaction.OutputText != nil {
fmt.Println(*res.Interaction.OutputText)
}
}
# First upload the file, then use the URI:
curl -X POST "https://generativelanguage.googleapis.com/v1beta/interactions" \
-H "x-goog-api-key: *** \
-H 'Content-Type: application/json' \
-d '{
"model": "gemini-3.8-flash",
"input": [
{"type": "text", "text": "Describe this audio clip"},
{
"type": "audio",
"uri": "YOUR_FILE_URI",
"mime_type": "audio/mp3"
}
]
}'
출처: 원문
본문
개요
Gemini는 오디오 입력을 분석·이해하고 텍스트 응답을 생성하여 다음과 같은 사용 사례를 가능하게 해요.
- 오디오 콘텐츠에 대한 설명, 요약 또는 질문 응답
- 전사 및 번역(음성-텍스트)
- 화자 분리(다른 화자 식별)
- 음성과 음악의 감정 감지
- 타임스탬프로 특정 세그먼트 분석
실시간 음성 및 비디오 상호작용은 Live API를 참조하세요.
실시간 전사를 지원하는 전용 음성-텍스트 모델은 Google Cloud Speech-to-Text API를 사용하세요.
음성을 텍스트로 전사
이 예시는 구조화된 출력을 사용해 타임스탬프, 화자 분리, 감정 감지로 음성을 전사, 번역, 요약하는 방법을 보여줘요.
from google import genai
client = genai.Client()
YOUTUBE_URL = "https://www.youtube.com/watch?v=ku-N-eS1lgM"
prompt = """
Process the audio file and generate a detailed transcription.
Requirements:
1. Identify distinct speakers (e.g., Speaker 1, Speaker 2).
2. Provide accurate timestamps for each segment (Format: MM:SS).
3. Detect the primary language of each segment.
4. If not English, provide the English translation.
5. Identify the primary emotion: Happy, Sad, Angry, or Neutral.
6. Provide a brief summary at the beginning.
"""
response_schema = {
"type": "object",
"properties": {
"summary": {"type": "string"},
"segments": {
"type": "array",
"items": {
"type": "object",
"properties": {
"speaker": {"type": "string"},
"timestamp": {"type": "string"},
"content": {"type": "string"},
"language": {"type": "string"},
"emotion": {
"type": "string",
"enum": ["happy", "sad", "angry", "neutral"]
}
},
"required": ["speaker", "timestamp", "content", "emotion"]
}
}
},
"required": ["summary", "segments"]
}
interaction = client.interactions.create(
model="gemini-3.8-flash",
input=[
{"type": "video", "uri": YOUTUBE_URL, "mime_type": "video/mp4"},
{"type": "text", "text": prompt}
],
response_format=response_schema,
)
print(interaction.output_text)
import { GoogleGenAI } from "@google/genai";
const client = new GoogleGenAI({});
const YOUTUBE_URL = "https://www.youtube.com/watch?v=ku-N-eS1lgM";
const prompt = `
Process the audio file and generate a detailed transcription.
Requirements:
1. Identify distinct speakers (e.g., Speaker 1, Speaker 2).
2. Provide accurate timestamps for each segment (Format: MM:SS).
3. Detect the primary language of each segment.
4. If not English, provide the English translation.
5. Identify the primary emotion: Happy, Sad, Angry, or Neutral.
6. Provide a brief summary at the beginning.
`;
const responseSchema = {
type: "object",
properties: {
summary: { type: "string" },
segments: {
type: "array",
items: {
type: "object",
properties: {
speaker: { type: "string" },
timestamp: { type: "string" },
content: { type: "string" },
language: { type: "string" },
emotion: {
type: "string",
enum: ["happy", "sad", "angry", "neutral"]
}
},
required: ["speaker", "timestamp", "content", "emotion"]
}
}
},
required: ["summary", "segments"]
};
const interaction = await client.interactions.create({
model: "gemini-3.8-flash",
input: [
{ type: "video", uri: YOUTUBE_URL, mime_type: "video/mp4" },
{ type: "text", text: prompt }
],
response_format: responseSchema,
});
console.log(JSON.parse(interaction.output_text));
package main
import (
"context"
"fmt"
"log"
"google.golang.org/genai"
"google.golang.org/genai/interactions/models/interactions"
"google.golang.org/genai/interactions/models/operations"
)
func main() {
ctx := context.Background()
client, err := genai.NewClient(ctx, nil)
if err != nil {
log.Fatal(err)
}
youtubeURL := "https://www.youtube.com/watch?v=ku-N-eS1lgM"
prompt := "Process the audio file and generate a detailed transcription.\n\n" +
"Requirements:\n" +
"1. Identify distinct speakers (e.g., Speaker 1, Speaker 2).\n" +
"2. Provide accurate timestamps for each segment (Format: MM:SS).\n" +
"3. Detect the primary language of each segment.\n" +
"4. If not English, provide the English translation.\n" +
"5. Identify the primary emotion: Happy, Sad, Angry, or Neutral.\n" +
"6. Provide a brief summary at the beginning."
responseSchema := map[string]any{
"type": "object",
"properties": map[string]any{
"summary": map[string]any{"type": "string"},
"segments": map[string]any{
"type": "array",
"items": map[string]any{
"type": "object",
"properties": map[string]any{
"speaker": map[string]any{"type": "string"},
"timestamp": map[string]any{"type": "string"},
"content": map[string]any{"type": "string"},
"language": map[string]any{"type": "string"},
"emotion": map[string]any{
"type": "string",
"enum": []string{"happy", "sad", "angry", "neutral"},
},
},
"required": []string{"speaker", "timestamp", "content", "emotion"},
},
},
},
"required": []string{"summary", "segments"},
}
contents := []interactions.Content{
interactions.NewContent(interactions.VideoContent{
URI: genai.Ptr(youtubeURL),
MimeType: interactions.VideoContentMimeTypeVideoMp4.ToPointer(),
}),
interactions.NewContent(interactions.TextContent{
Text: prompt,
}),
}
res, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
Model: interactions.Model("gemini-3.8-flash"),
Input: interactions.NewInteractionsInput(contents),
ResponseFormat: genai.Ptr(interactions.NewCreateModelInteractionResponseFormat(
interactions.NewResponseFormat(responseSchema),
)),
}),
})
if err != nil {
log.Fatal(err)
}
if res.Interaction.OutputText != nil {
fmt.Println(*res.Interaction.OutputText)
}
}
curl -X POST "https://generativelanguage.googleapis.com/v1beta/interactions" \
-H "x-goog-api-key: *** \
-H 'Content-Type: application/json' \
-d '{
"model": "gemini-3.8-flash",
"input": [
{
"type": "video",
"uri": "https://www.youtube.com/watch?v=ku-N-eS1lgM",
"mime_type": "video/mp4"
},
{
"type": "text",
"text": "Transcribe with speaker diarization and emotion detection."
}
],
"response_format": {
"type": "object",
"properties": {
"summary": {"type": "string"},
"segments": {
"type": "array",
"items": {
"type": "object",
"properties": {
"speaker": {"type": "string"},
"timestamp": {"type": "string"},
"content": {"type": "string"},
"emotion": {"type": "string", "enum": ["happy", "sad", "angry", "neutral"]}
}
}
}
}
}
}'

오디오 입력
오디오 데이터를 다음 방법으로 제공할 수 있어요.
- 요청 전에 오디오 파일 업로드
- 요청과 함께 인라인 오디오 데이터 전달
오디오 파일 업로드
20MB보다 큰 파일은 Files API를 사용하세요.
from google import genai
client = genai.Client()
uploaded_file = client.files.upload(file="path/to/sample.mp3")
interaction = client.interactions.create(
model="gemini-3.8-flash",
input=[
{"type": "text", "text": "Describe this audio clip"},
{
"type": "audio",
"uri": uploaded_file.uri,
"mime_type": uploaded_file.mime_type
}
]
)
print(interaction.output_text)
import { GoogleGenAI } from "@google/genai";
const client = new GoogleGenAI({});
const uploadedFile = await client.files.upload({
file: "path/to/sample.mp3",
config: { mimeType: "audio/mp3" }
});
const interaction = await client.interactions.create({
model: "gemini-3.8-flash",
input: [
{type: "text", text: "Describe this audio clip"},
{
type: "audio",
uri: uploadedFile.uri,
mime_type: uploadedFile.mimeType
}
]
});
console.log(interaction.output_text);
// Upload an audio file using the Files API (recommended for files > 20 MB)
package main
import (
"context"
"fmt"
"log"
"google.golang.org/genai"
"google.golang.org/genai/interactions/models/interactions"
"google.golang.org/genai/interactions/models/operations"
)
func main() {
ctx := context.Background()
client, err := genai.NewClient(ctx, nil)
if err != nil {
log.Fatal(err)
}
uploadedFile, err := client.Files.UploadFromPath(ctx, "path/to/sample.mp3", &genai.UploadFileConfig{
MIMEType: "audio/mp3",
})
if err != nil {
log.Fatal(err)
}
contents := []interactions.Content{
interactions.NewContent(interactions.TextContent{
Text: "Describe this audio clip",
}),
interactions.NewContent(interactions.AudioContent{
URI: genai.Ptr(uploadedFile.URI),
MimeType: interactions.AudioContentMimeType(uploadedFile.MIMEType).ToPointer(),
}),
}
res, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
Model: interactions.Model("gemini-3.8-flash"),
Input: interactions.NewInteractionsInput(contents),
}),
})
if err != nil {
log.Fatal(err)
}
if res.Interaction.OutputText != nil {
fmt.Println(*res.Interaction.OutputText)
}
}
# First upload the file using the Files API, then use the URI:
curl -X POST "https://generativelanguage.googleapis.com/v1beta/interactions" \
-H "x-goog-api-key: *** \
-H 'Content-Type: application/json' \
-d '{
"model": "gemini-3.8-flash",
"input": [
{"type": "text", "text": "Describe this audio clip"},
{
"type": "audio",
"uri": "YOUR_FILE_URI",
"mime_type": "audio/mp3"
}
]
}'
오디오 데이터 인라인 전달
총 요청 크기가 20MB 미만인 작은 오디오 파일의 경우:
from google import genai
import base64
client = genai.Client()
with open('path/to/small-sample.mp3', 'rb') as f:
audio_bytes = f.read()
interaction = client.interactions.create(
model="gemini-3.8-flash",
input=[
{"type": "text", "text": "Describe this audio clip"},
{
"type": "audio",
"data": base64.b64encode(audio_bytes).decode('utf-8'),
"mime_type": "audio/mp3"
}
]
)
print(interaction.output_text)
import { GoogleGenAI } from "@google/genai";
import * as fs from "node:fs";
const client = new GoogleGenAI({});
const audioData = fs.readFileSync("path/to/small-sample.mp3", {
encoding: "base64"
});
const interaction = await client.interactions.create({
model: "gemini-3.8-flash",
input: [
{type: "text", text: "Describe this audio clip"},
{
type: "audio",
data: audioData,
mime_type: "audio/mp3"
}
]
});
console.log(interaction.output_text);
import com.google.genai.Client;
import com.google.genai.gaos.models.interactions.AudioContent;
import com.google.genai.gaos.models.interactions.AudioContentMimeType;
import com.google.genai.gaos.models.interactions.Content;
import com.google.genai.gaos.models.interactions.CreateModelInteraction;
import com.google.genai.gaos.models.interactions.Interaction;
import com.google.genai.gaos.models.interactions.InteractionsInput;
import com.google.genai.gaos.models.interactions.Model;
import com.google.genai.gaos.models.interactions.TextContent;
import com.google.genai.gaos.models.operations.CreateInteractionRequestBody;
import java.nio.file.Files;
import java.nio.file.Paths;
import java.util.Arrays;
import java.util.Base64;
import java.util.List;
Client client = new Client();
byte[] audioBytes = Files.readAllBytes(Paths.get("path/to/small-sample.mp3"));
String base64Audio = Base64.getEncoder().encodeToString(audioBytes);
Content textContent = TextContent.builder().text("Describe this audio clip").build();
Content audioContent =
AudioContent.builder()
.data(base64Audio)
.mimeType(AudioContentMimeType.AUDIO_MP3)
.build();
List<Content> contents = Arrays.asList(textContent, audioContent);
CreateModelInteraction params =
CreateModelInteraction.builder()
.model(Model.of("gemini-3.8-flash"))
.input(InteractionsInput.ofContent(contents))
.build();
Interaction interaction =
client.interactions.create(CreateInteractionRequestBody.of(params)).interaction().get();
System.out.println(interaction.outputText().orElse(""));
package main
import (
"context"
"encoding/base64"
"fmt"
"log"
"os"
"google.golang.org/genai"
"google.golang.org/genai/interactions/models/interactions"
"google.golang.org/genai/interactions/models/operations"
)
func main() {
ctx := context.Background()
client, err := genai.NewClient(ctx, nil)
if err != nil {
log.Fatal(err)
}
audioBytes, err := os.ReadFile("path/to/small-sample.mp3")
if err != nil {
log.Fatal(err)
}
base64Audio := base64.StdEncoding.EncodeToString(audioBytes)
contents := []interactions.Content{
interactions.NewContent(interactions.TextContent{
Text: "Describe this audio clip",
}),
interactions.NewContent(interactions.AudioContent{
Data: genai.Ptr(base64Audio),
MimeType: interactions.AudioContentMimeTypeAudioMp3.ToPointer(),
}),
}
res, err := client.Interactions.Create(ctx, operations.CreateInteractionRequest{
Body: operations.NewCreateInteractionRequestBody(interactions.CreateModelInteraction{
Model: interactions.Model("gemini-3.8-flash"),
Input: interactions.NewInteractionsInput(contents),
}),
})
if err != nil {
log.Fatal(err)
}
if res.Interaction.OutputText != nil {
fmt.Println(*res.Interaction.OutputText)
}
}
AUDIO_PATH="path/to/sample.mp3"
if [[ "$(base64 --version 2>&1)" = *"FreeBSD"* ]]; then
B64FLAGS="--input"
else
B64FLAGS="-w0"
fi
curl -X POST "https://generativelanguage.googleapis.com/v1beta/interactions" \
-H "x-goog-api-key: *** \
-H 'Content-Type: application/json' \
-d '{
"model": "gemini-3.8-flash",
"input": [
{"type": "text", "text": "Describe this audio clip"},
{
"type": "audio",
"data": "'$(base64 $B64FLAGS $AUDIO_PATH)'",
"mime_type": "audio/mp3"
}
]
}'
인라인 오디오 데이터에 대한 참고 사항:
- 최대 요청 크기는 총 20MB야요(프롬프트와 모든 파일 포함)
- 재사용하려면 파일 업로드를 대신 사용하세요
트랜스크립트 얻기
트랜스크립트를 얻으려면 프롬프트에서 요청하세요.
interaction = client.interactions.create(
model="gemini-3.8-flash",
input=[
{"type": "text", "text": "Generate a transcript of the speech."},
{
"type": "audio",
"uri": uploaded_file.uri,
"mime_type": uploaded_file.mime_type
}
]
)
print(interaction.output_text)
const interaction = await client.interactions.create({
model: "gemini-3.8-flash",
input: [
{ type: "text", text: "Generate a transcript of the speech." },
{
type: "audio",
uri: uploadedFile.uri,
mime_type: uploadedFile.mimeType
}
]
});
console.log(interaction.output_text);
import com.google.genai.Client;
import com.google.genai.gaos.models.interactions.AudioContent;
import com.google.genai.gaos.models.interactions.AudioContentMimeType;
import com.google.genai.gaos.models.interactions.Content;
import com.google.genai.gaos.models.interactions.CreateModelInteraction;
import com.google.genai.gaos.models.interactions.Interaction;
import com.google.genai.gaos.models.interactions.InteractionsInput;
import com.google.genai.gaos.models.interactions.Model;
import com.google.genai.gaos.models.interactions.TextContent;
import com.google.genai.gaos.models.operations.CreateInteractionRequestBody;
import com.google.genai.types.File;
import com.google.genai.types.UploadFileConfig;
import java.util.Arrays;
import java.util.List;
Client client = new Client();
File uploadedFile =
client.files.upload(
"path/to/sample.mp3", UploadFileConfig.builder().mimeType("audio/mp3").build());
Content textContent = TextContent.builder().text("Generate a transcript of the speech.").build();
Content audioContent =
AudioContent.builder()
.uri(uploadedFile.uri().get())
.mimeType(AudioContentMimeType.of(uploadedFile.mimeType().get()))
.build();
List<Content> contents = Arrays.asList(textContent, audioContent);
CreateModelInteraction params =
CreateModelInteraction.builder()
.model(Model.of("gemini-3.8-flash"))
.input(InteractionsInput.ofContent(contents))
.build();
Interaction interaction =
client.interactions.create(CreateInteractionRequestBody.of(params)).interaction().get();
System.out.println(interaction.outputText().orElse(""));
타임스탬프 참조
MM:SS 형식을 사용해 특정 구간을 참조하세요.
interaction = client.interactions.create(
model="gemini-3.8-flash",
input=[
{"type": "text", "text": "Provide a transcript from 02:30 to 03:29."},
{
"type": "audio",
"uri": uploaded_file.uri,
"mime_type": uploaded_file.mime_type
}
]
)
const interaction = await client.interactions.create({
model: "gemini-3.8-flash",
input: [
{ type: "text", text: "Provide a transcript from 02:30 to 03:29." },
{ type: "audio", uri: uploadedFile.uri, mime_type: "audio/mp3" }
]
});
Content textContent =
TextContent.builder().text("Provide a transcript from 02:30 to 03:29.").build();
Content audioContent =
AudioContent.builder()
.uri(uploadedFile.uri().get())
.mimeType(AudioContentMimeType.of(uploadedFile.mimeType().get()))
.build();
List<Content> contents = Arrays.asList(textContent, audioContent);
CreateModelInteraction params =
CreateModelInteraction.builder()
.model(Model.of("gemini-3.8-flash"))
.input(InteractionsInput.ofContent(contents))
.build();
토큰 계산
오디오 파일의 토큰을 계산하세요.
response = client.models.count_tokens(
model="gemini-3.8-flash",
contents=[uploaded_file]
)
print(response)
const response = await client.models.countTokens({
model: "gemini-3.8-flash",
contents: [
{ fileData: { fileUri: uploadedFile.uri, mimeType: uploadedFile.mimeType } }
]
});
console.log(response.totalTokens);
import com.google.genai.Client;
import com.google.genai.types.Content;
import com.google.genai.types.CountTokensResponse;
import com.google.genai.types.File;
import com.google.genai.types.Part;
import com.google.genai.types.UploadFileConfig;
import java.util.Arrays;
Client client = new Client();
File uploadedFile =
client.files.upload(
"path/to/sample.mp3", UploadFileConfig.builder().mimeType("audio/mp3").build());
CountTokensResponse response =
client.models.countTokens(
"gemini-3.8-flash",
Arrays.asList(
Content.fromParts(
Part.fromUri(uploadedFile.uri().get(), uploadedFile.mimeType().get()))),
null);
System.out.println(response);
package main
import (
"context"
"fmt"
"log"
"google.golang.org/genai"
)
func main() {
ctx := context.Background()
client, err := genai.NewClient(ctx, nil)
if err != nil {
log.Fatal(err)
}
uploadedFile, err := client.Files.UploadFromPath(ctx, "path/to/sample.mp3", &genai.UploadFileConfig{
MIMEType: "audio/mp3",
})
if err != nil {
log.Fatal(err)
}
response, err := client.Models.CountTokens(
ctx,
"gemini-3.8-flash",
[]*genai.Content{
genai.NewContentFromURI(uploadedFile.URI, uploadedFile.MIMEType, genai.RoleUser),
},
nil,
)
if err != nil {
log.Fatal(err)
}
fmt.Println(response.TotalTokens)
}
지원 오디오 형식
Gemini는 다음 오디오 형식 MIME 유형을 지원해요.
- WAV -
audio/wav - MP3 -
audio/mp3 - AIFF -
audio/aiff - AAC -
audio/aac - OGG -
audio/ogg - FLAC -
audio/flac - MPEG -
audio/mpeg - M4A -
audio/m4a - L16 -
audio/l16 - Opus -
audio/opus - ALAW -
audio/alaw - MULAW -
audio/mulaw - WebM -
audio/webm
지원되는 MIME 유형과 매개변수 스키마의 전체 목록은 Interactions API 참조를 참조하세요.
오디오에 대한 기술적 세부사항
- 토큰: 오디오 1초당 32토큰(1분 = 1,920토큰)
- 비음성: Gemini는 새소리, 사이렌 등 비음성 소리를 이해해요
- 최대 길이: 프롬프트당 최대 9.5시간의 오디오
- 해상도: 16Kbps로 다운샘플링
- 채널: 다중 채널 오디오는 단일 채널로 결합