fix: schdule type gpu_selctor issue
This commit is contained in:
@@ -211,13 +211,12 @@ export default {
|
|||||||
'models.form.kvCache.tips':
|
'models.form.kvCache.tips':
|
||||||
'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.',
|
'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.',
|
||||||
'models.form.kvCache.tips2':
|
'models.form.kvCache.tips2':
|
||||||
'KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
'Extended KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
||||||
'models.form.scheduling': 'Scheduling',
|
'models.form.scheduling': 'Scheduling',
|
||||||
'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
||||||
'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
||||||
'models.form.ramRatio.tips':
|
'models.form.ramRatio.tips':
|
||||||
'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
||||||
'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
||||||
'models.form.chunkSize.tips':
|
'models.form.chunkSize.tips': 'Number of tokens per KV cache chunk.'
|
||||||
'Number of tokens per KV cache chunk. A larger chunk size may improve throughput but increase memory usage.'
|
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -211,15 +211,14 @@ export default {
|
|||||||
'models.form.kvCache.tips':
|
'models.form.kvCache.tips':
|
||||||
'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.',
|
'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.',
|
||||||
'models.form.kvCache.tips2':
|
'models.form.kvCache.tips2':
|
||||||
'KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
'Extended KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
||||||
'models.form.scheduling': 'Scheduling',
|
'models.form.scheduling': 'Scheduling',
|
||||||
'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
||||||
'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
||||||
'models.form.ramRatio.tips':
|
'models.form.ramRatio.tips':
|
||||||
'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
||||||
'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
||||||
'models.form.chunkSize.tips':
|
'models.form.chunkSize.tips': 'Number of tokens per KV cache chunk.'
|
||||||
'Number of tokens per KV cache chunk. A larger chunk size may improve throughput but increase memory usage.'
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// ========== To-Do: Translate Keys (Remove After Translation) ==========
|
// ========== To-Do: Translate Keys (Remove After Translation) ==========
|
||||||
@@ -265,11 +264,11 @@ export default {
|
|||||||
// 42. 'models.mymodels.status.active': 'Active'
|
// 42. 'models.mymodels.status.active': 'Active'
|
||||||
// 43. 'models.form.remoteURL.tips': 'Refer to the <a href="https://docs.lmcache.ai/api_reference/configurations.html" target="_blank">configuration documentation</a> for details.',
|
// 43. 'models.form.remoteURL.tips': 'Refer to the <a href="https://docs.lmcache.ai/api_reference/configurations.html" target="_blank">configuration documentation</a> for details.',
|
||||||
// 44. 'models.form.kvCache.tips': 'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.'
|
// 44. 'models.form.kvCache.tips': 'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.'
|
||||||
// 45. 'models.form.kvCache.tips2': 'KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
// 45. 'models.form.kvCache.tips2': 'Extended KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
||||||
// 46. 'models.form.scheduling': 'Scheduling',
|
// 46. 'models.form.scheduling': 'Scheduling',
|
||||||
// 47. 'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
// 47. 'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
||||||
// 48. 'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
// 48. 'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
||||||
// 49. 'models.form.ramRatio.tips': 'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
// 49. 'models.form.ramRatio.tips': 'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
||||||
// 50. 'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
// 50. 'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
||||||
// 51. 'models.form.chunkSize.tips': 'Number of tokens per KV cache chunk. A larger chunk size may improve throughput but increase memory usage.'
|
// 51. 'models.form.chunkSize.tips': 'Number of tokens per KV cache chunk.'
|
||||||
// ========== End of To-Do List ==========
|
// ========== End of To-Do List ==========
|
||||||
|
|||||||
@@ -211,15 +211,14 @@ export default {
|
|||||||
'models.form.kvCache.tips':
|
'models.form.kvCache.tips':
|
||||||
'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.',
|
'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.',
|
||||||
'models.form.kvCache.tips2':
|
'models.form.kvCache.tips2':
|
||||||
'KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
'Extended KV cache is only supported when using built-in inference backends (vLLM or SGLang).',
|
||||||
'models.form.scheduling': 'Scheduling',
|
'models.form.scheduling': 'Scheduling',
|
||||||
'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
||||||
'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
||||||
'models.form.ramRatio.tips':
|
'models.form.ramRatio.tips':
|
||||||
'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
||||||
'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
||||||
'models.form.chunkSize.tips':
|
'models.form.chunkSize.tips': 'Number of tokens per KV cache chunk.'
|
||||||
'Number of tokens per KV cache chunk. A larger chunk size may improve throughput but increase memory usage.'
|
|
||||||
};
|
};
|
||||||
|
|
||||||
// ========== To-Do: Translate Keys (Remove After Translation) ==========
|
// ========== To-Do: Translate Keys (Remove After Translation) ==========
|
||||||
@@ -228,11 +227,11 @@ export default {
|
|||||||
// 4. 'models.mymodels.status.active': 'Active'
|
// 4. 'models.mymodels.status.active': 'Active'
|
||||||
// 5. 'models.form.remoteURL.tips': 'Refer to the <a href="https://docs.lmcache.ai/api_reference/configurations.html" target="_blank">configuration documentation</a> for details.',
|
// 5. 'models.form.remoteURL.tips': 'Refer to the <a href="https://docs.lmcache.ai/api_reference/configurations.html" target="_blank">configuration documentation</a> for details.',
|
||||||
// 6. 'models.form.kvCache.tips': 'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.'
|
// 6. 'models.form.kvCache.tips': 'Available only with built-in backends (vLLM / SGLang) — switch backend in <span class="bold-text">Advanced</span> to enable.'
|
||||||
// 7. 'models.form.kvCache.tips2': 'KV cache is only supported when using built-in inference backends (vLLM or SGLang).';
|
// 7. 'models.form.kvCache.tips2': 'Extended KV cache is only supported when using built-in inference backends (vLLM or SGLang).';
|
||||||
// 8. 'models.form.scheduling': 'Scheduling',
|
// 8. 'models.form.scheduling': 'Scheduling',
|
||||||
// 9. 'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
// 9. 'models.form.ramRatio': 'RAM-to-VRAM Ratio',
|
||||||
// 10. 'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
// 10. 'models.form.ramSize': 'Maximum RAM Size (GiB)',
|
||||||
// 11. 'models.form.ramRatio.tips': 'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
// 11. 'models.form.ramRatio.tips': 'Ratio of system RAM to GPU VRAM used for KV cache. For example, 2.0 means the cache in RAM can be twice as large as the GPU VRAM.',
|
||||||
// 12. 'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
// 12. 'models.form.ramSize.tips': `Maximum size of the KV cache stored in system memory (GiB). If set, this value overrides "{content}".`,
|
||||||
// 13. 'models.form.chunkSize.tips': 'Number of tokens per KV cache chunk. A larger chunk size may improve throughput but increase memory usage.'
|
// 13. 'models.form.chunkSize.tips': 'Number of tokens per KV cache chunk.'
|
||||||
// ========== End of To-Do List ==========
|
// ========== End of To-Do List ==========
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ import _ from 'lodash';
|
|||||||
import { backendOptionsMap } from '../config/backend-parameters';
|
import { backendOptionsMap } from '../config/backend-parameters';
|
||||||
import { FormData } from './types';
|
import { FormData } from './types';
|
||||||
|
|
||||||
|
// generate the gpu_selector field for form initial values, when eidting a model
|
||||||
export const generateGPUSelector = (data: any, gpuOptions: any[]) => {
|
export const generateGPUSelector = (data: any, gpuOptions: any[]) => {
|
||||||
const gpu_ids = _.get(data, 'gpu_selector.gpu_ids', []);
|
const gpu_ids = _.get(data, 'gpu_selector.gpu_ids', []);
|
||||||
if (gpu_ids.length === 0) {
|
if (gpu_ids.length === 0) {
|
||||||
@@ -34,13 +35,13 @@ export const generateGPUSelector = (data: any, gpuOptions: any[]) => {
|
|||||||
};
|
};
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* before submit the form, generate the gpu_selector field
|
* before submit the form, generate the gpu_selector field, and clear worker_selector if needed
|
||||||
* @param data
|
* @param data
|
||||||
* @returns
|
* @returns
|
||||||
*/
|
*/
|
||||||
export const generateGPUIds = (data: FormData) => {
|
export const generateGPUIds = (data: FormData) => {
|
||||||
const gpu_ids = _.get(data, 'gpu_selector.gpu_ids', []);
|
const gpu_ids = _.get(data, 'gpu_selector.gpu_ids', []);
|
||||||
console.log('generateGPUIds', gpu_ids);
|
|
||||||
if (!gpu_ids.length) {
|
if (!gpu_ids.length) {
|
||||||
return {
|
return {
|
||||||
gpu_selector: null
|
gpu_selector: null
|
||||||
@@ -63,10 +64,8 @@ export const generateGPUIds = (data: FormData) => {
|
|||||||
return {
|
return {
|
||||||
gpu_selector: {
|
gpu_selector: {
|
||||||
gpu_ids: result || [],
|
gpu_ids: result || [],
|
||||||
gpus_per_replica:
|
gpus_per_replica: data.gpu_selector?.gpus_per_replica || null
|
||||||
data.gpu_selector?.gpus_per_replica === -1
|
},
|
||||||
? null
|
worker_selector: null
|
||||||
: data.gpu_selector?.gpus_per_replica
|
|
||||||
}
|
|
||||||
};
|
};
|
||||||
};
|
};
|
||||||
|
|||||||
@@ -158,7 +158,7 @@ const DataForm: React.FC<DataFormProps> = forwardRef((props, ref) => {
|
|||||||
return {
|
return {
|
||||||
gpu_selector: {
|
gpu_selector: {
|
||||||
gpu_ids: [gpuids[0]],
|
gpu_ids: [gpuids[0]],
|
||||||
gpus_per_replica: -1
|
gpus_per_replica: null
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
}
|
}
|
||||||
@@ -352,7 +352,7 @@ const DataForm: React.FC<DataFormProps> = forwardRef((props, ref) => {
|
|||||||
name="deployModel"
|
name="deployModel"
|
||||||
form={form}
|
form={form}
|
||||||
onFinish={handleOk}
|
onFinish={handleOk}
|
||||||
preserve={true}
|
preserve={false}
|
||||||
clearOnDestroy={true}
|
clearOnDestroy={true}
|
||||||
onValuesChange={handleOnValuesChange}
|
onValuesChange={handleOnValuesChange}
|
||||||
onFinishFailed={handleOnFinishFailed}
|
onFinishFailed={handleOnFinishFailed}
|
||||||
|
|||||||
@@ -2,7 +2,7 @@ import CheckboxField from '@/components/seal-form/checkbox-field';
|
|||||||
import SealInputNumber from '@/components/seal-form/input-number';
|
import SealInputNumber from '@/components/seal-form/input-number';
|
||||||
import { useIntl } from '@umijs/max';
|
import { useIntl } from '@umijs/max';
|
||||||
import { Form } from 'antd';
|
import { Form } from 'antd';
|
||||||
import { useMemo } from 'react';
|
import { useMemo, useRef } from 'react';
|
||||||
import { backendOptionsMap } from '../config/backend-parameters';
|
import { backendOptionsMap } from '../config/backend-parameters';
|
||||||
import { useFormContext } from '../config/form-context';
|
import { useFormContext } from '../config/form-context';
|
||||||
import { FormData } from '../config/types';
|
import { FormData } from '../config/types';
|
||||||
@@ -13,18 +13,20 @@ const KVCacheForm = () => {
|
|||||||
const { onValuesChange, backendOptions } = useFormContext();
|
const { onValuesChange, backendOptions } = useFormContext();
|
||||||
const kvCacheEnabled = Form.useWatch(['extended_kv_cache', 'enabled'], form);
|
const kvCacheEnabled = Form.useWatch(['extended_kv_cache', 'enabled'], form);
|
||||||
const backend = Form.useWatch('backend', form);
|
const backend = Form.useWatch('backend', form);
|
||||||
|
const configCacheRef = useRef<any>({});
|
||||||
|
|
||||||
const handleOnChange = async (e: any) => {
|
const handleOnChange = async (e: any) => {
|
||||||
const extendedKVCache = form.getFieldValue('extended_kv_cache');
|
|
||||||
if (e.target.checked) {
|
if (e.target.checked) {
|
||||||
form.setFieldsValue({
|
form.setFieldsValue({
|
||||||
extended_kv_cache: {
|
extended_kv_cache: {
|
||||||
enabled: true,
|
enabled: true,
|
||||||
chunk_size: extendedKVCache?.chunk_size,
|
chunk_size: configCacheRef.current?.chunk_size,
|
||||||
ram_ratio: extendedKVCache?.ram_ratio || 1.2,
|
ram_ratio: configCacheRef.current?.ram_ratio || 1.2,
|
||||||
ram_size: extendedKVCache?.ram_size
|
ram_size: configCacheRef.current?.ram_size
|
||||||
}
|
}
|
||||||
});
|
});
|
||||||
|
} else {
|
||||||
|
configCacheRef.current = form.getFieldValue('extended_kv_cache');
|
||||||
}
|
}
|
||||||
await new Promise((resolve) => {
|
await new Promise((resolve) => {
|
||||||
setTimeout(resolve, 200);
|
setTimeout(resolve, 200);
|
||||||
|
|||||||
@@ -77,7 +77,7 @@ const ScheduleTypeForm: React.FC = () => {
|
|||||||
return;
|
return;
|
||||||
}
|
}
|
||||||
if (value === ScheduleValueMap.Manual) {
|
if (value === ScheduleValueMap.Manual) {
|
||||||
form.setFieldValue(['gpu_selector', 'gpus_per_replica'], -1);
|
form.setFieldValue(['gpu_selector', 'gpus_per_replica'], null);
|
||||||
}
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
@@ -180,10 +180,11 @@ const ScheduleTypeForm: React.FC = () => {
|
|||||||
label={intl.formatMessage({
|
label={intl.formatMessage({
|
||||||
id: 'models.form.gpusperreplica'
|
id: 'models.form.gpusperreplica'
|
||||||
})}
|
})}
|
||||||
|
allowNull
|
||||||
options={[
|
options={[
|
||||||
{
|
{
|
||||||
label: intl.formatMessage({ id: 'common.options.auto' }),
|
label: intl.formatMessage({ id: 'common.options.auto' }),
|
||||||
value: -1
|
value: null
|
||||||
},
|
},
|
||||||
{ label: '1', value: 1 },
|
{ label: '1', value: 1 },
|
||||||
{ label: '2', value: 2 },
|
{ label: '2', value: 2 },
|
||||||
|
|||||||
Reference in New Issue
Block a user