Question 1mediummultiple choice
Read the full Software Development explanation →NCA-GENL Software Development • Complete Question Bank
Complete NCA-GENL Software Development question bank — all 0 questions with answers and detailed explanations.
{
"model": "llama-3-8b-instruct",
"max_tokens": 512,
"temperature": 0.7,
"n_gpu_layers": 40,
"stream": true
}{
"engine": "tensorrt-llm",
"quantization": "awq",
"batching": "continuous",
"kv_cache_type": "paged"
}config.pbtxt:
instance_group [
{
count: 2
kind: KIND_GPU
}
]model_config.json: { 'quantization': 'int8_sq', 'tensor_parallel': 4, 'pipeline_parallel': 2, 'max_batch_size': 128 }ERROR: [NIM_SERVER] Request failed. Context window exceeded for model: meta-llama-3-70b-instruct. Prompt + Retrieved Context length: 132,044 tokens. Max tokens: 128,000.
{
"policy": "strict",
"quantization": "fp8",
"max_concurrent_requests": 128,
"scheduling": "fcfs"
}{
"model_name": "llama-3",
"precision": "fp16",
"max_batch_size": 8
}config.pbtxt:
backend: "tensorrt"
parameters {
key: "precision_mode"
value: { string_value: "fp8" }
}config_policy: { 'max_tokens': 1024, 'temperature': 0.7, 'top_p': 0.9, 'presence_penalty': 0.0 }log_output: [INFO] Loading model 'llama_v3'... [ERROR] CUDA error: all CUDA-capable devices are busy or unavailable.
{
"model_name": "llama-3-8b",
"max_batch_size": 128,
"precision": "fp16",
"enable_cuda_graph": true
}