389 lines
13 KiB
JSON
389 lines
13 KiB
JSON
{
|
|
"schema_version": 1,
|
|
"job_id": 1609,
|
|
"base_model": "trained-job-1608",
|
|
"base_model_id": "Qwen/Qwen2.5-0.5B-Instruct",
|
|
"modality": "text",
|
|
"training_strategy": "slm",
|
|
"artifact_type": "lora_adapter",
|
|
"model_path": "/app/backend/data/artifacts/job_1609/adapter",
|
|
"tokenizer_path": "/app/backend/data/artifacts/job_1609/adapter",
|
|
"context_length": 8192,
|
|
"phase": {
|
|
"enabled": true,
|
|
"index": 2,
|
|
"number": 3,
|
|
"total": 3,
|
|
"sample_start": 7000,
|
|
"sample_end": 10000,
|
|
"target_samples": 10000
|
|
},
|
|
"created_at": "2026-08-07T21:41:36Z",
|
|
"test_supported": true,
|
|
"ownership": {
|
|
"owner": "Reallexi LLC",
|
|
"copyright": "Copyright (c) 2026 Reallexi LLC. All rights reserved.",
|
|
"core_backlink": "https://llm.reallexi.io",
|
|
"license": "Apache-2.0 unless third-party base model or dataset terms require more restrictions.",
|
|
"notice": "This artifact was produced by Reallexi LLC AI Model Builder. Review upstream model, adapter, dataset and source licenses before redistribution."
|
|
},
|
|
"lineage": {
|
|
"mission": "incremental",
|
|
"parent_job_id": 1608,
|
|
"lineage_root_job_id": 256,
|
|
"base_artifact_job_id": 1608,
|
|
"base_artifact_path": "/app/backend/data/artifacts/job_1608",
|
|
"resume_from_job_id": null,
|
|
"resume_checkpoint_path": "",
|
|
"rerun_policy": "",
|
|
"continue_policy": "train-next-phase-from-completed-artifact",
|
|
"history": [
|
|
{
|
|
"timestamp": "2026-08-07T21:22:59.565539",
|
|
"action": "incremental",
|
|
"trigger": "automatic",
|
|
"from_job_id": 1606,
|
|
"from_status": "completed",
|
|
"base_model": "trained-job-1606",
|
|
"phase_index": 1,
|
|
"phase_sample_start": 3500,
|
|
"phase_sample_end": 7000
|
|
},
|
|
{
|
|
"timestamp": "2026-08-07T21:33:35.341886",
|
|
"action": "incremental",
|
|
"trigger": "automatic",
|
|
"from_job_id": 1608,
|
|
"from_status": "completed",
|
|
"base_model": "trained-job-1608",
|
|
"phase_index": 2,
|
|
"phase_sample_start": 7000,
|
|
"phase_sample_end": 10000
|
|
}
|
|
]
|
|
},
|
|
"adapter": {
|
|
"enabled": true,
|
|
"method": "peft-lora-adapter",
|
|
"type": "auto",
|
|
"label": "Auto LoRA",
|
|
"description": "Choose PEFT target modules from the loaded model architecture.",
|
|
"target_modules": "",
|
|
"rank": 8,
|
|
"alpha": 16,
|
|
"model_class": "PeftModelForCausalLM"
|
|
},
|
|
"dataset_preparation": {
|
|
"modality": "text",
|
|
"source": "hub",
|
|
"dataset": "AzharAli05/Resume-Screening-Dataset",
|
|
"max_samples": 10000,
|
|
"phase": {
|
|
"enabled": true,
|
|
"number": 3,
|
|
"total": 3,
|
|
"sample_start": 7000,
|
|
"sample_end": 10000,
|
|
"phase_samples": 3000
|
|
},
|
|
"external_sources": [],
|
|
"temporary_external_cache": true,
|
|
"text_column": "text",
|
|
"split": "train",
|
|
"conversion": "dataset rows are normalized into a text column before tokenization."
|
|
},
|
|
"exports": {
|
|
"requested": [
|
|
"huggingface",
|
|
"safetensors",
|
|
"gguf",
|
|
"onnx",
|
|
"pytorch-bin"
|
|
],
|
|
"formats": [
|
|
{
|
|
"id": "huggingface",
|
|
"label": "Native Training Output",
|
|
"extension": "folder",
|
|
"generated": true,
|
|
"description": "The original Transformers model or PEFT adapter folder saved when training completes.",
|
|
"available": true,
|
|
"status": "generated"
|
|
},
|
|
{
|
|
"id": "safetensors",
|
|
"label": "Standalone SafeTensors",
|
|
"extension": ".zip",
|
|
"generated": false,
|
|
"build_on_demand": true,
|
|
"description": "Complete merged SafeTensors model built and verified after adapter training.",
|
|
"available": false,
|
|
"status": "build-on-demand"
|
|
},
|
|
{
|
|
"id": "gguf",
|
|
"label": "GGUF Q8_0",
|
|
"extension": ".gguf",
|
|
"generated": false,
|
|
"build_on_demand": true,
|
|
"description": "Complete GGUF model built automatically with the pinned official llama.cpp converter.",
|
|
"available": false,
|
|
"status": "build-on-demand"
|
|
},
|
|
{
|
|
"id": "onnx",
|
|
"label": "ONNX",
|
|
"extension": ".onnx",
|
|
"generated": false,
|
|
"requires_converter": "optimum export onnx",
|
|
"description": "ONNX runtime export; requires Optimum or an equivalent exporter.",
|
|
"available": false,
|
|
"status": "converter-required"
|
|
},
|
|
{
|
|
"id": "pytorch-bin",
|
|
"label": "PyTorch Bin",
|
|
"extension": ".bin",
|
|
"generated": false,
|
|
"description": "Legacy PyTorch binary export; use only when a target runtime requires it.",
|
|
"available": false,
|
|
"status": "converter-required"
|
|
}
|
|
],
|
|
"artifact_type": "lora_adapter"
|
|
},
|
|
"proof": {
|
|
"source": "hub",
|
|
"dataset": "AzharAli05/Resume-Screening-Dataset",
|
|
"split": "train",
|
|
"text_column": "text",
|
|
"trained_samples": 3000,
|
|
"epochs": 3,
|
|
"total_steps": 2250,
|
|
"micro_batch_size": 4,
|
|
"gradient_accumulation": 2,
|
|
"learning_rate": 1e-05,
|
|
"adam_epsilon": 0.0001,
|
|
"max_grad_norm": 1.0,
|
|
"context_length": 8192,
|
|
"duration_seconds": 440.05,
|
|
"model_id": "Qwen/Qwen2.5-0.5B-Instruct",
|
|
"training_strategy": "slm",
|
|
"adapter_training": true,
|
|
"adapter_method": "peft-lora-adapter",
|
|
"lora_type": "auto",
|
|
"prepared_dataset_rows": 3000,
|
|
"external_source_budget": {},
|
|
"resource_plan": {
|
|
"memory_total_gb": 23.47,
|
|
"memory_available_gb": 20.3,
|
|
"recommended_training_memory_gb": 17.25,
|
|
"cpu_threads_available": 8,
|
|
"cuda_available": true,
|
|
"cuda_device_count": 1,
|
|
"cuda_name": "NVIDIA GeForce RTX 5090",
|
|
"vram_total_gb": 31.84,
|
|
"containerized": true,
|
|
"runtime_os": "linux",
|
|
"host_os": "win32",
|
|
"accelerators": [
|
|
{
|
|
"id": "cuda:0",
|
|
"torch_device": "cuda:0",
|
|
"backend": "cuda",
|
|
"vendor": "nvidia",
|
|
"name": "NVIDIA GeForce RTX 5090",
|
|
"memory_gb": 31.84,
|
|
"usable": true,
|
|
"reason": "Available to this PyTorch runtime."
|
|
},
|
|
{
|
|
"id": "host:amd:1",
|
|
"torch_device": "",
|
|
"backend": "host-only",
|
|
"vendor": "amd",
|
|
"name": "AMD Radeon RX 7900 XT",
|
|
"memory_gb": 0.0,
|
|
"usable": false,
|
|
"reason": "Detected by Windows, but AMD GPUs are not exposed to this Linux Docker runtime.",
|
|
"driver": "32.0.23034.4"
|
|
},
|
|
{
|
|
"id": "host:amd:2",
|
|
"torch_device": "",
|
|
"backend": "host-only",
|
|
"vendor": "amd",
|
|
"name": "AMD Radeon(TM) Graphics",
|
|
"memory_gb": 0.0,
|
|
"usable": false,
|
|
"reason": "Detected by Windows, but AMD GPUs are not exposed to this Linux Docker runtime.",
|
|
"driver": "32.0.21042.62"
|
|
}
|
|
],
|
|
"selected_accelerator": {
|
|
"id": "cuda:0",
|
|
"torch_device": "cuda:0",
|
|
"backend": "cuda",
|
|
"vendor": "nvidia",
|
|
"name": "NVIDIA GeForce RTX 5090",
|
|
"memory_gb": 31.84,
|
|
"usable": true,
|
|
"reason": "Available to this PyTorch runtime."
|
|
},
|
|
"accelerator_backend": "cuda",
|
|
"recommended_cpu_context_length": 1024,
|
|
"recommended_gpu_context_length": 4096,
|
|
"training_budget_gb": 31.84,
|
|
"context_length_options": {
|
|
"0.5B": [
|
|
512,
|
|
1024,
|
|
2048,
|
|
4096,
|
|
8192,
|
|
16384,
|
|
32768
|
|
],
|
|
"1.1B": [
|
|
512,
|
|
1024,
|
|
2048,
|
|
4096,
|
|
8192,
|
|
16384,
|
|
32768
|
|
],
|
|
"1.5B": [
|
|
512,
|
|
1024,
|
|
2048,
|
|
4096,
|
|
8192,
|
|
16384,
|
|
32768
|
|
],
|
|
"2B": [
|
|
512,
|
|
1024,
|
|
2048,
|
|
4096,
|
|
8192,
|
|
16384,
|
|
32768
|
|
],
|
|
"3.8B": [
|
|
512,
|
|
1024,
|
|
2048,
|
|
4096,
|
|
8192,
|
|
16384,
|
|
32768
|
|
],
|
|
"7B": [
|
|
512,
|
|
1024,
|
|
2048,
|
|
4096,
|
|
8192,
|
|
16384,
|
|
32768
|
|
],
|
|
"8B": [
|
|
512,
|
|
1024,
|
|
2048,
|
|
4096,
|
|
8192,
|
|
16384
|
|
],
|
|
"13B": []
|
|
},
|
|
"device": "cuda",
|
|
"requested_device": "cuda:0",
|
|
"requested_memory_gb": 10.7,
|
|
"effective_memory_gb": 10.7,
|
|
"requested_vram_gb": 24.0,
|
|
"effective_vram_gb": 24.0,
|
|
"context_length_cap": 1024,
|
|
"warnings": []
|
|
},
|
|
"model_memory_profile": {
|
|
"total_parameters": 495114112,
|
|
"trainable_parameters": 1081344,
|
|
"parameter_memory_gb": 0.92,
|
|
"estimated_minimum_training_gb": 1.25
|
|
},
|
|
"checkpoint_every_steps": 1000,
|
|
"resumed_from_checkpoint": false,
|
|
"resume_from_job_id": null,
|
|
"resume_step": 0,
|
|
"lineage_parent_job_id": 1608,
|
|
"lineage_root_job_id": 256,
|
|
"base_artifact_job_id": 1608,
|
|
"base_artifact_path": "/app/backend/data/artifacts/job_1608",
|
|
"dataset_preparation": {
|
|
"modality": "text",
|
|
"source": "hub",
|
|
"dataset": "AzharAli05/Resume-Screening-Dataset",
|
|
"max_samples": 10000,
|
|
"phase": {
|
|
"enabled": true,
|
|
"number": 3,
|
|
"total": 3,
|
|
"sample_start": 7000,
|
|
"sample_end": 10000,
|
|
"phase_samples": 3000
|
|
},
|
|
"external_sources": [],
|
|
"temporary_external_cache": true,
|
|
"text_column": "text",
|
|
"split": "train",
|
|
"conversion": "dataset rows are normalized into a text column before tokenization."
|
|
},
|
|
"external_sources": [],
|
|
"skipped_external_sources": [],
|
|
"temporary_external_cache": true,
|
|
"phased_training": true,
|
|
"phase_index": 2,
|
|
"phase_number": 3,
|
|
"phase_total": 3,
|
|
"phase_sample_start": 7000,
|
|
"phase_sample_end": 10000,
|
|
"phase_target_samples": 10000,
|
|
"phase_has_more": false,
|
|
"phase_data_exhausted": false,
|
|
"phase_requested_samples": 3000,
|
|
"phase_deferred_sources": 0,
|
|
"phase_cumulative_trained_samples": 10000,
|
|
"phase_remaining_samples": 0,
|
|
"phase_extended": false
|
|
},
|
|
"sample_report": {
|
|
"has_samples": true,
|
|
"has_training_curve": true,
|
|
"sample_outputs": [
|
|
{
|
|
"prompt": "Role: AR/VR Developer; Resume: Here's a professional resume for Mary Johnson, tailored to the AR/VR Developer role: Mary Johnson Contact",
|
|
"before": "Information:\n* Address: 123 Main St, Anytown, USA 12345\n* Phone: (555) 555-5555\n* Email: [mary.johnson@email.com](mailto:mary.johnson@email.com)\nProfessional Summary:\nHighly motivated and experienced AR/VR Developer with expertise in Unity, C#,",
|
|
"after": "Information:\n* Address: 123 Main St, Anytown, USA 12345\n* Phone: (555) 555-5555\n* Email: [mary.johnson@email.com](mailto:mary.johnson@email.com)\n* LinkedIn: linkedin.com/in/maryjohnsondeveloper\n\nSummary:\nHighly motivated and"
|
|
},
|
|
{
|
|
"prompt": "Role: product manager; Resume: here's a sample resume for brent brown applying for the role of product manager: brent brown",
|
|
"before": "Product Manager\nContact Information:\n\n* Email: [brent.brown@email.com](mailto:brent.brown@email.com)\n* Phone: (123) 456-7890\n* LinkedIn: linkedin.com/in/brentbrown\n\nSummary:\nHighly motivated and detail-oriented Product Manager with 5+ years of experience in driving successful product launches, delivering high",
|
|
"after": "product manager\n\ncontact information:\n\n* email: [brent.brown@email.com](mailto:brent.brown@email.com)\n* phone: 555-555-5555\n* linkedin: linkedin.com/in/brentbrown\n\nsummary:\nhighly motivated and detail-oriented product manager with 3+ years of experience in creating and executing successful product strategies."
|
|
},
|
|
{
|
|
"prompt": "Role: data engineer; Resume: **gina mehta** **data engineer candidate** gina mehta is a highly skilled and experienced data engineer with",
|
|
"before": "over 5 years of experience in designing, developing, and deploying scalable data solutions. She has a strong background in cloud computing, database design, and data visualization.\n\n**Key skills:**\n\n* Cloud platforms (AWS, Azure, Google Cloud)\n* Database management (MySQL, PostgreSQL, MongoDB)\n* Data modeling and querying\n* Data visualization tools (Tableau, Power BI)\n* Big data analytics",
|
|
"after": "a strong background in designing, developing, and deploying scalable data solutions. she has a proven track record of delivering high-quality data products that meet the needs of clients across various industries.\n\n**key skills and achievements:**\n\n* **data engineering:** demonstrated expertise in designing, developing, and deploying complex data pipelines using tools like apache spark, hive, and hbase.\n* **big data analytics:** proficient in"
|
|
}
|
|
]
|
|
},
|
|
"numerical_validation": {
|
|
"valid": true,
|
|
"trainable_tensors_checked": 192,
|
|
"trainable_values_checked": 1081344
|
|
},
|
|
"adapter_path": "/app/backend/data/artifacts/job_1609/adapter"
|
|
} |