From d63b8e44231d88e54f14c90a9aacb55d98ea5e6e Mon Sep 17 00:00:00 2001 From: abhishek-singh591 Date: Fri, 2 Jan 2026 08:48:42 +0000 Subject: [PATCH 01/18] Added all the changes of logger Signed-off-by: abhishek-singh591 --- QEfficient/__init__.py | 4 +- QEfficient/base/modeling_qeff.py | 6 +- QEfficient/base/pytorch_transforms.py | 4 +- QEfficient/cloud/export.py | 4 +- QEfficient/cloud/finetune.py | 4 +- QEfficient/cloud/infer.py | 4 +- QEfficient/compile/compile_helper.py | 4 +- QEfficient/compile/qnn_compiler.py | 4 +- .../models/transformers/transformer_flux.py | 4 +- .../diffusers/pipelines/flux/pipeline_flux.py | 4 +- .../diffusers/pipelines/pipeline_utils.py | 4 +- .../diffusers/pipelines/wan/pipeline_wan.py | 4 +- .../exporter/export_hf_to_cloud_ai_100.py | 4 +- QEfficient/finetune/dataset/alpaca_dataset.py | 4 +- QEfficient/finetune/dataset/custom_dataset.py | 4 +- .../finetune/dataset/grammar_dataset.py | 4 +- QEfficient/finetune/utils/config_utils.py | 4 +- QEfficient/finetune/utils/dataset_utils.py | 4 +- QEfficient/finetune/utils/plot_metrics.py | 4 +- QEfficient/generation/embedding_handler.py | 4 +- .../generation/text_generation_inference.py | 5 +- QEfficient/generation/vlm_generation.py | 4 +- QEfficient/peft/auto.py | 4 +- QEfficient/peft/lora/auto.py | 4 +- .../models/gpt_oss/modeling_gpt_oss.py | 4 +- .../models/internvl/modeling_internvl.py | 4 +- .../models/llava/modeling_llava.py | 4 +- .../models/llava_next/modeling_llava_next.py | 4 +- .../models/mistral3/modeling_mistral3.py | 4 +- .../transformers/models/modeling_auto.py | 4 +- .../models/qwen2_5_vl/modeling_qwen2_5_vl.py | 4 +- .../transformers/quantizers/quantizer_awq.py | 4 +- .../quantizer_compressed_tensors.py | 4 +- .../transformers/quantizers/quantizer_gptq.py | 4 +- .../quantizers/quantizer_mxfp4.py | 4 +- QEfficient/transformers/transform.py | 4 +- QEfficient/utils/_utils.py | 4 +- QEfficient/utils/check_ccl_specializations.py | 4 +- QEfficient/utils/device_utils.py | 4 +- QEfficient/utils/export_utils.py | 4 +- QEfficient/utils/logging_utils.py | 346 ++++++++++++++++-- QEfficient/utils/sampler_utils.py | 4 +- examples/text_generation/basic_inference.py | 2 + scripts/finetune/run_ft_model.py | 4 +- .../calculate_perplexity.py | 3 +- tests/conftest.py | 4 +- .../test_transformer_pytorch_transforms.py | 4 +- tests/utils/test_logger.py | 48 +++ 48 files changed, 497 insertions(+), 81 deletions(-) create mode 100644 tests/utils/test_logger.py diff --git a/QEfficient/__init__.py b/QEfficient/__init__.py index 3c9f68efd1..d63495f113 100644 --- a/QEfficient/__init__.py +++ b/QEfficient/__init__.py @@ -35,7 +35,9 @@ from QEfficient.peft import QEffAutoPeftModelForCausalLM from QEfficient.transformers.transform import transform from QEfficient.utils import custom_format_warning -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") # custom warning for the better logging experience warnings.formatwarning = custom_format_warning diff --git a/QEfficient/base/modeling_qeff.py b/QEfficient/base/modeling_qeff.py index b5c838a94f..1155734aac 100644 --- a/QEfficient/base/modeling_qeff.py +++ b/QEfficient/base/modeling_qeff.py @@ -7,7 +7,6 @@ import gc import inspect -import logging import shutil import subprocess import warnings @@ -35,8 +34,9 @@ load_json, ) from QEfficient.utils.export_utils import export_wrapper +from QEfficient.utils.logging_utils import QEFFLogger -logger = logging.getLogger(__name__) +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") class QEFFBaseModel(ABC): @@ -326,6 +326,7 @@ def _export( self.prefill_onnx_path = onnx_path else: self.onnx_path = onnx_path + logger.info("Model export is finished and saved at: %s", onnx_path) return onnx_path def get_onnx_path( @@ -539,4 +540,5 @@ def _compile( logger.info("Hashed parameters exported successfully.") self.qpc_path = qpc_path + logger.info("Model compilation is finished and saved at: %s", qpc_path) return qpc_path diff --git a/QEfficient/base/pytorch_transforms.py b/QEfficient/base/pytorch_transforms.py index e503a057fa..dd8d9c0b8b 100644 --- a/QEfficient/base/pytorch_transforms.py +++ b/QEfficient/base/pytorch_transforms.py @@ -9,7 +9,9 @@ from torch import nn -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") class PytorchTransform: diff --git a/QEfficient/cloud/export.py b/QEfficient/cloud/export.py index a5e0b6e195..48d10c7006 100644 --- a/QEfficient/cloud/export.py +++ b/QEfficient/cloud/export.py @@ -12,7 +12,9 @@ from QEfficient.base.common import QEFFCommonLoader from QEfficient.utils import check_and_assign_cache_dir from QEfficient.utils.custom_yaml import generate_custom_io -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") # Specifically for Docker images. ROOT_DIR = os.path.dirname(os.path.abspath("")) diff --git a/QEfficient/cloud/finetune.py b/QEfficient/cloud/finetune.py index 35ebbde326..3e67e33544 100644 --- a/QEfficient/cloud/finetune.py +++ b/QEfficient/cloud/finetune.py @@ -29,10 +29,12 @@ from QEfficient.finetune.utils.dataset_utils import get_dataloader, get_longest_seq_length from QEfficient.finetune.utils.device_map import get_device_map from QEfficient.finetune.utils.helper import Task_Mode, get_world_size -from QEfficient.finetune.utils.logging_utils import logger from QEfficient.finetune.utils.parser import get_finetune_parser from QEfficient.finetune.utils.train_utils import print_model_size, print_trainable_parameters, train from QEfficient.utils._utils import hf_download +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") # Try importing QAIC-specific module, proceed without it if unavailable try: diff --git a/QEfficient/cloud/infer.py b/QEfficient/cloud/infer.py index ef05d29abe..3a5e20ca66 100644 --- a/QEfficient/cloud/infer.py +++ b/QEfficient/cloud/infer.py @@ -17,7 +17,9 @@ from QEfficient.base.common import QEFFCommonLoader from QEfficient.utils import check_and_assign_cache_dir, load_hf_processor, load_hf_tokenizer -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") # TODO: Remove after adding support for VLM's compile and execute diff --git a/QEfficient/compile/compile_helper.py b/QEfficient/compile/compile_helper.py index 5de21f8760..bf83013f46 100644 --- a/QEfficient/compile/compile_helper.py +++ b/QEfficient/compile/compile_helper.py @@ -15,7 +15,9 @@ from QEfficient.compile.qnn_compiler import compile as qnn_compile from QEfficient.utils import constants from QEfficient.utils._utils import load_json, load_yaml -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") def create_and_dump_specializations( diff --git a/QEfficient/compile/qnn_compiler.py b/QEfficient/compile/qnn_compiler.py index e2ec203649..9fcadb6d01 100644 --- a/QEfficient/compile/qnn_compiler.py +++ b/QEfficient/compile/qnn_compiler.py @@ -18,7 +18,9 @@ generate_qnn_specialization, ) from QEfficient.utils.hash_utils import to_hashable -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") class QNN: diff --git a/QEfficient/diffusers/models/transformers/transformer_flux.py b/QEfficient/diffusers/models/transformers/transformer_flux.py index 40b7e3e7e3..7cb8475e1a 100644 --- a/QEfficient/diffusers/models/transformers/transformer_flux.py +++ b/QEfficient/diffusers/models/transformers/transformer_flux.py @@ -19,7 +19,9 @@ ) from QEfficient.diffusers.models.modeling_utils import compute_blocked_attention, get_attention_blocking_config -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") def qeff_apply_rotary_emb( diff --git a/QEfficient/diffusers/pipelines/flux/pipeline_flux.py b/QEfficient/diffusers/pipelines/flux/pipeline_flux.py index eeb260c531..52401eae9d 100644 --- a/QEfficient/diffusers/pipelines/flux/pipeline_flux.py +++ b/QEfficient/diffusers/pipelines/flux/pipeline_flux.py @@ -38,7 +38,9 @@ set_module_device_ids, ) from QEfficient.generation.cloud_infer import QAICInferenceSession -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") class QEffFluxPipeline: diff --git a/QEfficient/diffusers/pipelines/pipeline_utils.py b/QEfficient/diffusers/pipelines/pipeline_utils.py index 135a6bd07d..71ba9da4d6 100644 --- a/QEfficient/diffusers/pipelines/pipeline_utils.py +++ b/QEfficient/diffusers/pipelines/pipeline_utils.py @@ -18,7 +18,9 @@ from tqdm import tqdm from QEfficient.utils._utils import load_json -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") def calculate_compressed_latent_dimension(height: int, width: int, vae_scale_factor: int) -> int: diff --git a/QEfficient/diffusers/pipelines/wan/pipeline_wan.py b/QEfficient/diffusers/pipelines/wan/pipeline_wan.py index 888763af0b..6cd86d0407 100644 --- a/QEfficient/diffusers/pipelines/wan/pipeline_wan.py +++ b/QEfficient/diffusers/pipelines/wan/pipeline_wan.py @@ -36,7 +36,9 @@ ) from QEfficient.generation.cloud_infer import QAICInferenceSession from QEfficient.utils import constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") class QEffWanPipeline: diff --git a/QEfficient/exporter/export_hf_to_cloud_ai_100.py b/QEfficient/exporter/export_hf_to_cloud_ai_100.py index 2547d9db36..73f9d00cfe 100644 --- a/QEfficient/exporter/export_hf_to_cloud_ai_100.py +++ b/QEfficient/exporter/export_hf_to_cloud_ai_100.py @@ -20,7 +20,9 @@ from QEfficient.utils import load_hf_tokenizer from QEfficient.utils.constants import QEFF_MODELS_DIR, Constants from QEfficient.utils.generate_inputs import InputHandler -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") def convert_to_cloud_bertstyle( diff --git a/QEfficient/finetune/dataset/alpaca_dataset.py b/QEfficient/finetune/dataset/alpaca_dataset.py index ff44860eb4..f25fa9767e 100644 --- a/QEfficient/finetune/dataset/alpaca_dataset.py +++ b/QEfficient/finetune/dataset/alpaca_dataset.py @@ -11,7 +11,9 @@ import torch from torch.utils.data import Dataset -from QEfficient.finetune.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") PROMPT_DICT = { "prompt_input": ( diff --git a/QEfficient/finetune/dataset/custom_dataset.py b/QEfficient/finetune/dataset/custom_dataset.py index ef76e83ed0..2a98a3ca46 100644 --- a/QEfficient/finetune/dataset/custom_dataset.py +++ b/QEfficient/finetune/dataset/custom_dataset.py @@ -9,7 +9,9 @@ import logging from pathlib import Path -from QEfficient.finetune.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") def load_module_from_py_file(py_file: str) -> object: diff --git a/QEfficient/finetune/dataset/grammar_dataset.py b/QEfficient/finetune/dataset/grammar_dataset.py index 8fb3eb1521..4a2e4658d8 100644 --- a/QEfficient/finetune/dataset/grammar_dataset.py +++ b/QEfficient/finetune/dataset/grammar_dataset.py @@ -10,7 +10,9 @@ from datasets import load_dataset from torch.utils.data import Dataset -from QEfficient.finetune.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") class grammar(Dataset): diff --git a/QEfficient/finetune/utils/config_utils.py b/QEfficient/finetune/utils/config_utils.py index 0c8b3d8275..89d1753d9a 100644 --- a/QEfficient/finetune/utils/config_utils.py +++ b/QEfficient/finetune/utils/config_utils.py @@ -20,7 +20,9 @@ from QEfficient.finetune.configs.training import TrainConfig from QEfficient.finetune.dataset.dataset_config import DATASET_PREPROC from QEfficient.finetune.utils.helper import Peft_Method -from QEfficient.finetune.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") def update_config(config, **kwargs): diff --git a/QEfficient/finetune/utils/dataset_utils.py b/QEfficient/finetune/utils/dataset_utils.py index 01c1e32aa0..9e15a73227 100644 --- a/QEfficient/finetune/utils/dataset_utils.py +++ b/QEfficient/finetune/utils/dataset_utils.py @@ -16,7 +16,9 @@ from QEfficient.finetune.data.sampler import DistributedLengthBasedBatchSampler from QEfficient.finetune.dataset.dataset_config import DATALOADER_COLLATE_FUNC, DATASET_PREPROC from QEfficient.finetune.utils.helper import get_world_size -from QEfficient.finetune.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") def get_preprocessed_dataset( diff --git a/QEfficient/finetune/utils/plot_metrics.py b/QEfficient/finetune/utils/plot_metrics.py index 1e22bc6a83..4ea307dff7 100644 --- a/QEfficient/finetune/utils/plot_metrics.py +++ b/QEfficient/finetune/utils/plot_metrics.py @@ -11,7 +11,9 @@ import matplotlib.pyplot as plt -from QEfficient.finetune.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") def plot_metric(data, metric_name, x_label, y_label, title, colors): diff --git a/QEfficient/generation/embedding_handler.py b/QEfficient/generation/embedding_handler.py index e07b5dd046..c33906b619 100644 --- a/QEfficient/generation/embedding_handler.py +++ b/QEfficient/generation/embedding_handler.py @@ -23,7 +23,9 @@ from QEfficient.generation.cloud_infer import QAICInferenceSession from QEfficient.utils import constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") class VisionHandler: diff --git a/QEfficient/generation/text_generation_inference.py b/QEfficient/generation/text_generation_inference.py index de10c9b881..0103beec3a 100755 --- a/QEfficient/generation/text_generation_inference.py +++ b/QEfficient/generation/text_generation_inference.py @@ -19,9 +19,11 @@ from QEfficient.generation.cloud_infer import QAICInferenceSession from QEfficient.utils import padding_check_and_fix from QEfficient.utils.constants import Constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.sampler_utils import validate_sampler_inputs +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") + @dataclass class PerfMetrics: @@ -1316,4 +1318,5 @@ def generate( generated_ids=self._qaic_model.generated_ids, perf_metrics=perf_metrics, ) + logger.info("Text Generated finised") return latency_stats diff --git a/QEfficient/generation/vlm_generation.py b/QEfficient/generation/vlm_generation.py index adacc373e7..3dfd68efbd 100644 --- a/QEfficient/generation/vlm_generation.py +++ b/QEfficient/generation/vlm_generation.py @@ -37,7 +37,9 @@ ) from QEfficient.utils import LRUCache from QEfficient.utils.constants import Constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") class VisionLanguageGeneration(QEffTextGenerationBase): diff --git a/QEfficient/peft/auto.py b/QEfficient/peft/auto.py index 6c71730725..28c19601e4 100644 --- a/QEfficient/peft/auto.py +++ b/QEfficient/peft/auto.py @@ -6,7 +6,6 @@ # ---------------------------------------------------------------------------- import hashlib -import logging import warnings from typing import List, Optional, Union @@ -32,8 +31,9 @@ from QEfficient.utils import constants from QEfficient.utils._utils import get_padding_shape_from_config from QEfficient.utils.hash_utils import to_hashable +from QEfficient.utils.logging_utils import QEFFLogger -logger = logging.getLogger(__name__) +logger = QEFFLogger.get_logger("FT", loglevel="INFO") class QEffAutoPeftModelForCausalLM(QEFFBaseModel): diff --git a/QEfficient/peft/lora/auto.py b/QEfficient/peft/lora/auto.py index 8ff8335f5d..07f61c1e90 100644 --- a/QEfficient/peft/lora/auto.py +++ b/QEfficient/peft/lora/auto.py @@ -19,7 +19,9 @@ from QEfficient.peft.lora.pytorch_transforms import LoraModelInputsTransform, TargetModulesTransform from QEfficient.utils import constants, get_padding_shape_from_config from QEfficient.utils.hash_utils import to_hashable -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") class QEffAutoLoraModelForCausalLM(QEFFAutoModelForCausalLM): diff --git a/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py b/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py index 3efe890b85..48eaf6f549 100644 --- a/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py +++ b/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py @@ -33,7 +33,9 @@ from QEfficient.transformers.cache_utils import QEffHybridCacheForGPTOSS from QEfficient.transformers.modeling_attn_mask_utils import _create_causal_mask from QEfficient.utils.constants import MIN_MASKED_ATTENTION_VALUE -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") class QEffGptOssExperts(GptOssExperts): diff --git a/QEfficient/transformers/models/internvl/modeling_internvl.py b/QEfficient/transformers/models/internvl/modeling_internvl.py index b47db7edac..2b30c28225 100644 --- a/QEfficient/transformers/models/internvl/modeling_internvl.py +++ b/QEfficient/transformers/models/internvl/modeling_internvl.py @@ -13,7 +13,9 @@ from QEfficient.utils import constants from QEfficient.utils._utils import IOInfo, get_padding_shape_from_config -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") class QEffInternEncoderWrapper(nn.Module): diff --git a/QEfficient/transformers/models/llava/modeling_llava.py b/QEfficient/transformers/models/llava/modeling_llava.py index abdb77ea55..ec8c7909db 100644 --- a/QEfficient/transformers/models/llava/modeling_llava.py +++ b/QEfficient/transformers/models/llava/modeling_llava.py @@ -15,7 +15,9 @@ ) from QEfficient.utils._utils import IOInfo -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") BS = 1 FBS = 4 diff --git a/QEfficient/transformers/models/llava_next/modeling_llava_next.py b/QEfficient/transformers/models/llava_next/modeling_llava_next.py index 627f7393e2..3fe5d03bd1 100755 --- a/QEfficient/transformers/models/llava_next/modeling_llava_next.py +++ b/QEfficient/transformers/models/llava_next/modeling_llava_next.py @@ -18,7 +18,9 @@ from QEfficient.utils import constants from QEfficient.utils._utils import IOInfo -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") BS = constants.ONNX_EXPORT_EXAMPLE_BATCH_SIZE FBS = constants.ONNX_EXPORT_EXAMPLE_FBS diff --git a/QEfficient/transformers/models/mistral3/modeling_mistral3.py b/QEfficient/transformers/models/mistral3/modeling_mistral3.py index d2149b6bd4..a277ce67c8 100644 --- a/QEfficient/transformers/models/mistral3/modeling_mistral3.py +++ b/QEfficient/transformers/models/mistral3/modeling_mistral3.py @@ -21,7 +21,9 @@ from QEfficient.utils import constants from QEfficient.utils._utils import IOInfo, get_padding_shape_from_config -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") def custom_cumsum(tensor): diff --git a/QEfficient/transformers/models/modeling_auto.py b/QEfficient/transformers/models/modeling_auto.py index 236f6c9f5a..4dd37c18cc 100644 --- a/QEfficient/transformers/models/modeling_auto.py +++ b/QEfficient/transformers/models/modeling_auto.py @@ -69,9 +69,11 @@ get_padding_shape_from_config, ) from QEfficient.utils.check_ccl_specializations import process_ccl_specializations -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.sampler_utils import get_sampling_inputs_and_outputs +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") + class QEFFTransformersBase(QEFFBaseModel): """ diff --git a/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py b/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py index 21d2e026ea..6c5cd854dc 100644 --- a/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py +++ b/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py @@ -38,7 +38,9 @@ from QEfficient.utils import constants from QEfficient.utils._utils import IOInfo, get_padding_shape_from_config from QEfficient.utils.constants import MIN_MASKED_ATTENTION_VALUE -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") def qeff_apply_rotary_pos_emb(q, k, cos, sin, position_ids, mrope_section, unsqueeze_dim=1): diff --git a/QEfficient/transformers/quantizers/quantizer_awq.py b/QEfficient/transformers/quantizers/quantizer_awq.py index ef8a03521f..0968bdd271 100644 --- a/QEfficient/transformers/quantizers/quantizer_awq.py +++ b/QEfficient/transformers/quantizers/quantizer_awq.py @@ -15,7 +15,9 @@ replace_linear_layer_with_target_layer, replace_quantization_scales, ) -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") class QEffAwqConfig(AwqConfig): diff --git a/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py b/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py index e7e14166d9..86d7ae1395 100644 --- a/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py +++ b/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py @@ -14,7 +14,9 @@ from transformers.utils.quantization_config import CompressedTensorsConfig, QuantizationConfigMixin, QuantizationMethod from QEfficient.transformers.quantizers.quantizer_utils import get_keys_to_not_convert -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") FP8_DTYPE = torch.float8_e4m3fn diff --git a/QEfficient/transformers/quantizers/quantizer_gptq.py b/QEfficient/transformers/quantizers/quantizer_gptq.py index 8a0bea1a21..073f370e18 100644 --- a/QEfficient/transformers/quantizers/quantizer_gptq.py +++ b/QEfficient/transformers/quantizers/quantizer_gptq.py @@ -15,7 +15,9 @@ repack_zeros, replace_linear_layer_with_target_layer, ) -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") class QEffGPTQConfig(GPTQConfig): diff --git a/QEfficient/transformers/quantizers/quantizer_mxfp4.py b/QEfficient/transformers/quantizers/quantizer_mxfp4.py index 2ffba1beaa..bfbfe473e2 100644 --- a/QEfficient/transformers/quantizers/quantizer_mxfp4.py +++ b/QEfficient/transformers/quantizers/quantizer_mxfp4.py @@ -14,7 +14,9 @@ from transformers.utils.quantization_config import Mxfp4Config from QEfficient.transformers.quantizers.quantizer_utils import convert_moe_packed_tensors, get_keys_to_not_convert -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") class QEffMxfp4GptOssExperts(nn.Module): diff --git a/QEfficient/transformers/transform.py b/QEfficient/transformers/transform.py index 11d7c1dfd4..ecbc16a384 100644 --- a/QEfficient/transformers/transform.py +++ b/QEfficient/transformers/transform.py @@ -13,7 +13,9 @@ from QEfficient.base.modeling_qeff import QEFFBaseModel from QEfficient.transformers.cache_utils import QEffDynamicCache from QEfficient.transformers.modeling_utils import TransformersToQEffModulesDict -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") def replace_module_with_qeff_layers(model: nn.Module) -> None: diff --git a/QEfficient/utils/_utils.py b/QEfficient/utils/_utils.py index 26bae7a34b..21d60fa1c4 100644 --- a/QEfficient/utils/_utils.py +++ b/QEfficient/utils/_utils.py @@ -28,7 +28,9 @@ from QEfficient.utils.constants import KWARGS_INCLUSION_LIST, QEFF_MODELS_DIR, Constants, QnnConstants from QEfficient.utils.hash_utils import json_serializable -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") class LRUCache: diff --git a/QEfficient/utils/check_ccl_specializations.py b/QEfficient/utils/check_ccl_specializations.py index cc259ee360..0061b5ad72 100644 --- a/QEfficient/utils/check_ccl_specializations.py +++ b/QEfficient/utils/check_ccl_specializations.py @@ -8,7 +8,9 @@ from typing import List, Tuple from QEfficient.utils import constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") # Better performance when context length is multiple of 1024 → map CL to the next multiple of 1024 diff --git a/QEfficient/utils/device_utils.py b/QEfficient/utils/device_utils.py index a76dfae8af..72e5b423ee 100644 --- a/QEfficient/utils/device_utils.py +++ b/QEfficient/utils/device_utils.py @@ -10,7 +10,9 @@ import subprocess from QEfficient.utils.constants import Constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") def is_networks_loaded(stdout): diff --git a/QEfficient/utils/export_utils.py b/QEfficient/utils/export_utils.py index 33ba694cfb..140a961e23 100644 --- a/QEfficient/utils/export_utils.py +++ b/QEfficient/utils/export_utils.py @@ -17,9 +17,11 @@ from QEfficient.transformers.models.pytorch_transforms import get_decoder_layer_classes_for_export from QEfficient.utils.cache import QEFF_HOME from QEfficient.utils.hash_utils import create_export_hash -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.torch_patches import apply_torch_patches, undo_torch_patches +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") + def export_wrapper(func): """ diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index d2086d830c..f2659dff6b 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -5,54 +5,332 @@ # # ----------------------------------------------------------------------------- +import json import logging +import os +import queue +import threading +from datetime import datetime +from logging.handlers import RotatingFileHandler +from typing import Any, Dict, List, Optional +from tabulate import tabulate -class QEffFormatter(logging.Formatter): - """ - Formatter class used to set colors for printing different logging levels of messages on console. + +class JSONNamespaceFormatter(logging.Formatter): """ + Custom formatter to output log records in JSON format with metadata. + + Methods: + format(record): Formats a log record into a JSON string. - cyan: str = "\x1b[38;5;14m" - yellow: str = "\x1b[33;20m" - red: str = "\x1b[31;20m" - bold_red: str = "\x1b[31;1m" - reset: str = "\x1b[0m" - common_format: str = "%(levelname)s - %(name)s - %(message)s" # type: ignore - format_with_line_info = "%(levelname)s - %(name)s - %(message)s (%(filename)s:%(lineno)d)" # type: ignore - - FORMATS = { - logging.DEBUG: cyan + format_with_line_info + reset, - logging.INFO: cyan + common_format + reset, - logging.WARNING: yellow + common_format + reset, - logging.ERROR: red + format_with_line_info + reset, - logging.CRITICAL: bold_red + format_with_line_info + reset, - } + Parameters: + record (logging.LogRecord): The log record to format. + + Returns: + str: JSON-formatted log string. + """ def format(self, record): + log_record = { + "date": datetime.fromtimestamp(record.created).strftime("%Y-%m-%d"), + "time": datetime.fromtimestamp(record.created).strftime("%H:%M:%S"), + "level": record.levelname, + "namespace": getattr(record, "namespace", "default"), + "file": record.filename, + "line": record.lineno, + "message": record.getMessage(), + } + return json.dumps(log_record) + + +class QEFFLoggerThread(threading.Thread): + """ + Background thread to handle logging asynchronously using a queue. + + Attributes: + logger (logging.Logger): Logger instance to handle log records. + log_queue (queue.Queue): Queue from which log records are consumed. + running (bool): Flag to control thread execution. + """ + + def __init__(self, logger, log_queue): + """ + Initialize the logging thread. + + Parameters: + logger (logging.Logger): Logger instance. + log_queue (queue.Queue): Queue for log records. + """ + super().__init__(daemon=True) + self.logger = logger + self.log_queue = log_queue + self.running = True + + def run(self): + """ + Continuously process log records from the queue and pass them to the logger. + """ + while self.running: + try: + record = self.log_queue.get(timeout=1) + self.logger.handle(record) + except queue.Empty: + continue + + def stop(self): """ - Overriding the base class method to Choose format based on log level. + Stop the logging thread gracefully. """ - log_fmt = self.FORMATS.get(record.levelno) - formatter = logging.Formatter(log_fmt) - return formatter.format(record) + self.running = False -def create_logger() -> logging.Logger: +class QEFFLogger: """ - Creates a logger object with Colored QEffFormatter. + Singleton logger class for structured logging with namespace support. + + Class Attributes: + _instance (Optional[logging.Logger]): Singleton logger instance. + _logfile (Optional[str]): Path to the log file. + _log_queue (queue.Queue): Queue for asynchronous logging. + _logger_thread (Optional[QEFFLoggerThread]): Background logging thread. """ - logger = logging.getLogger("QEfficient") - # create console handler and set level to debug - ch = logging.StreamHandler() - ch.setLevel(logging.INFO) - # define formatter - ch.setFormatter(QEffFormatter()) + _instance: Optional[logging.Logger] = None + _logfile: Optional[str] = None + _log_queue: queue.Queue = queue.Queue() + _logger_thread: Optional[QEFFLoggerThread] = None + + def __init__(self, loglevel: Optional[str] = "INFO", log_path: Optional[str] = None): + """ + Initialize the logger instance with specified log level and path. + + Parameters: + loglevel (str): Logging level (e.g., "INFO", "DEBUG"). + log_path (str): Optional path to the log file. + """ + if QEFFLogger._instance is None: + self.loglevel = loglevel + self.log_path = log_path + self.logger = self._initialize_logger() + QEFFLogger._instance = self.logger + QEFFLogger._logger_thread = QEFFLoggerThread(self.logger, QEFFLogger._log_queue) + QEFFLogger._logger_thread.start() + + def _initialize_logger(self) -> logging.Logger: + """ + Set up the logger with rotating file handler and JSON formatter. + + Returns: + logging.Logger: Configured logger instance. + """ + if self.log_path is None: + log_dir = os.path.expanduser("~/.cache/qefficient_logs") + os.makedirs(log_dir, exist_ok=True) + timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") + self.log_path = os.path.join(log_dir, f"QEFF_{timestamp}.log") + + QEFFLogger._logfile = self.log_path + + numeric_level = getattr(logging, self.loglevel.upper(), None) + if not isinstance(numeric_level, int): + raise ValueError(f"Invalid log level: {self.loglevel}") + + logger = logging.getLogger("QEFF_LOGGER") + logger.setLevel(numeric_level) + + if not logger.handlers: + handler = RotatingFileHandler(self.log_path, maxBytes=5 * 1024 * 1024, backupCount=10) + handler.setFormatter(JSONNamespaceFormatter()) + logger.addHandler(handler) + + return logger + + @classmethod + def get_logger( + cls, namespace: str, loglevel: Optional[str] = "INFO", log_path: Optional[str] = None + ) -> logging.Logger: + """ + Retrieve a logger adapter with a specific namespace. + + Parameters: + namespace (str): Logical grouping for the log. + loglevel (str): Logging level. + log_path (str): Optional path to the log file. + + Returns: + logging.Logger: Logger adapter with namespace. + """ + if cls._instance is None: + cls(loglevel, log_path) + return logging.LoggerAdapter(cls._instance, {"namespace": namespace}) + + @classmethod + def log(cls, level: str, namespace: str, msg: str, fn: str = "", lno: int = 0, func: str = ""): + """ + Log a message with specified level and metadata. + + Parameters: + level (str): Logging level (e.g., "INFO", "ERROR"). + namespace (str): Logical grouping for the log. + msg (str): Log message. + fn (str): Filename where the log is generated. + lno (int): Line number in the file. + func (str): Function name. + """ + if cls._instance is None: + raise RuntimeError("Logger has not been initialized. Call get_logger() first.") + + level_num = getattr(logging, level.upper(), None) + if not isinstance(level_num, int): + raise ValueError(f"Invalid log level: {level}") + + record = cls._instance.makeRecord( + name="QEFF_LOGGER", + level=level_num, + fn=fn, + lno=lno, + msg=msg, + args=(), + exc_info=None, + func=func, + extra={"namespace": namespace}, + ) + cls._log_queue.put(record) + + @classmethod + def set_loglevel(cls, loglevel: Optional[str] = "INFO"): + """ + Update the log level of the logger. + + Parameters: + loglevel (str): New log level to set. + """ + if cls._instance is None: + raise RuntimeError("Logger has not been initialized yet. Call get_logger() first.") + + numeric_level = getattr(logging, loglevel.upper(), None) + if not isinstance(numeric_level, int): + raise ValueError(f"Invalid log level: {loglevel}") + + cls._instance.setLevel(numeric_level) + + @classmethod + def close_logger(cls): + """ + Gracefully shut down the logger and its thread. + """ + if cls._logger_thread: + cls._logger_thread.stop() + cls._logger_thread.join() + cls._logger_thread = None + + if cls._instance: + handlers = cls._instance.handlers[:] + for handler in handlers: + handler.close() + cls._instance.removeHandler(handler) + cls._instance = None + cls._logfile = None + + @classmethod + def _parse_dt(cls, date_str: str, time_str: str) -> datetime: + """Parse 'YYYY-MM-DD' and 'HH:MM:SS' into a datetime.""" + return datetime.strptime(f"{date_str} {time_str}", "%Y-%m-%d %H:%M:%S") + + @classmethod + def print_table(cls) -> None: + """ + Parse the line-delimited JSON log in cls._logfile and print timing table with t1 as baseline (0.0s): + - Model Loading : t2 - t1 + - Model Exporting : t3 - t2 + - Model Compilation : t4 - t3 + - Text Generation : t5 - t4 + - Total Time : t5 - t1 + + Milestones (matched to your log sample): + t1: first log line timestamp (baseline) + t2: "PyTorch export successful" + t3: "Transformed ONNX saved" + t4: "Model compilation is finished and saved" + t5: "Text Generated finised" + If t5 is missing, we fall back to "specialization_file_path" as readiness marker. + """ + path = cls._logfile + if not path: + raise FileNotFoundError("Log file path is not set (cls._logfile is None).") + if not os.path.exists(path): + raise FileNotFoundError(f"Log file does not exist: {path}") + + t_start: Optional[datetime] = None + t_export_done: Optional[datetime] = None + t_onnx_saved: Optional[datetime] = None + t_compile_done: Optional[datetime] = None + t_text_done: Optional[datetime] = None + t_text_ready: Optional[datetime] = None + + with open(path, "r", encoding="utf-8") as f: + for line in f: + line = line.strip() + if not line: + continue + try: + rec: Dict[str, Any] = json.loads(line) + except json.JSONDecodeError: + continue + + date_str = rec.get("date") + time_str = rec.get("time") + msg = rec.get("message", "") + if not date_str or not time_str: + continue + + ts = cls._parse_dt(date_str, time_str) + + if t_start is None: + t_start = ts + + if ("PyTorch export successful" in msg) and (t_export_done is None): + t_export_done = ts + + if ("Transformed ONNX saved" in msg) and (t_onnx_saved is None): + t_onnx_saved = ts + + if ("Model compilation is finished and saved" in msg) and (t_compile_done is None): + t_compile_done = ts + + if ("Text Generated finised" in msg) and (t_text_done is None): + t_text_done = ts + + if ("specialization_file_path" in msg) and (t_text_ready is None): + t_text_ready = ts + + if t_start is None: + raise ValueError("Could not determine start time (no valid log lines with date/time).") + + if t_text_done is None: + t_text_done = t_text_ready + + t_export_done = t_export_done or t_start + t_onnx_saved = t_onnx_saved or t_export_done + t_compile_done = t_compile_done or t_onnx_saved + t_text_done = t_text_done or t_compile_done + + def to_offset_seconds(t: datetime) -> float: + return (t - t_start).total_seconds() - logger.addHandler(ch) - return logger + o1 = 0.0 + o2 = to_offset_seconds(t_export_done) + o3 = to_offset_seconds(t_onnx_saved) + o4 = to_offset_seconds(t_compile_done) + o5 = to_offset_seconds(t_text_done) + timing_data: List[List[Any]] = [ + ["Model Loading", max(0.0, o2 - o1)], + ["Model Exporting", max(0.0, o3 - o2)], + ["Model Compilation", max(0.0, o4 - o3)], + ["Text Generation", max(0.0, o5 - o4)], + ["Total Time", max(0.0, o5 - o1)], + ] -# Define the logger object that can be used for logging purposes throughout the module. -logger = create_logger() + print(tabulate(timing_data, headers=["Step", "Time (s)"], tablefmt="github", floatfmt=".3f")) diff --git a/QEfficient/utils/sampler_utils.py b/QEfficient/utils/sampler_utils.py index 82a0843bc5..847266ae80 100644 --- a/QEfficient/utils/sampler_utils.py +++ b/QEfficient/utils/sampler_utils.py @@ -11,7 +11,9 @@ from QEfficient.utils import constants from QEfficient.utils.constants import Constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") def validate_sampler_inputs( diff --git a/examples/text_generation/basic_inference.py b/examples/text_generation/basic_inference.py index 6340ec7256..a4041d514e 100644 --- a/examples/text_generation/basic_inference.py +++ b/examples/text_generation/basic_inference.py @@ -10,6 +10,7 @@ from transformers import AutoTokenizer from QEfficient import QEFFAutoModelForCausalLM +from QEfficient.utils.logging_utils import QEFFLogger def main(): @@ -51,6 +52,7 @@ def main(): print(f"\nPrompt: {args.prompt}") print(f"Generated: {exec_info.generated_texts[0]}") + QEFFLogger.print_table() if __name__ == "__main__": diff --git a/scripts/finetune/run_ft_model.py b/scripts/finetune/run_ft_model.py index f5b64e717b..97ffb80cb7 100644 --- a/scripts/finetune/run_ft_model.py +++ b/scripts/finetune/run_ft_model.py @@ -14,7 +14,9 @@ from transformers import AutoModelForCausalLM, AutoTokenizer from QEfficient.finetune.configs.training import TrainConfig -from QEfficient.finetune.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("FT", loglevel="INFO") # Suppress all warnings warnings.filterwarnings("ignore") diff --git a/scripts/perplexity_computation/calculate_perplexity.py b/scripts/perplexity_computation/calculate_perplexity.py index e2988a0ae2..04ce624ec6 100644 --- a/scripts/perplexity_computation/calculate_perplexity.py +++ b/scripts/perplexity_computation/calculate_perplexity.py @@ -18,8 +18,9 @@ from transformers import AutoModelForCausalLM, AutoTokenizer from QEfficient.generation.cloud_infer import QAICInferenceSession +from QEfficient.utils.logging_utils import QEFFLogger -logger = logging.getLogger(__name__) +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") # 1. Data Loading diff --git a/tests/conftest.py b/tests/conftest.py index ba0f341fec..b5037b7d07 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -13,9 +13,11 @@ from transformers import AutoConfig from QEfficient.utils.constants import QEFF_MODELS_DIR -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.test_utils import ModelConfig +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") + def get_custom_model_config_dict(configs): """ diff --git a/tests/transformers/test_transformer_pytorch_transforms.py b/tests/transformers/test_transformer_pytorch_transforms.py index eb05b3f95e..279d49bdf4 100644 --- a/tests/transformers/test_transformer_pytorch_transforms.py +++ b/tests/transformers/test_transformer_pytorch_transforms.py @@ -20,7 +20,9 @@ from QEfficient.transformers.quantizers.quant_transforms import AwqToMatmulNbitsTransform, GPTQToMatmulNbitsTransform from QEfficient.transformers.spd.turbo import ResBlock from QEfficient.utils._utils import get_padding_shape_from_config -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") KVCacheTransformTestConfigs = [ ("llama", 3, 32, 128, {"num_key_value_heads": 8, "intermediate_size": 512}, 0.8), diff --git a/tests/utils/test_logger.py b/tests/utils/test_logger.py new file mode 100644 index 0000000000..e6bd51c5cf --- /dev/null +++ b/tests/utils/test_logger.py @@ -0,0 +1,48 @@ +# ----------------------------------------------------------------------------- +# +# Copyright (c) Qualcomm Technologies, Inc. and/or its subsidiaries. +# SPDX-License-Identifier: BSD-3-Clause +# +# ----------------------------------------------------------------------------- + +import threading +import time + +from QEfficient.utils.logging_utils import QEFFLogger + +# ------------------------------- +# Define namespace once +# ------------------------------- +NAMESPACE = "model" + +# ------------------------------- +# Initialize logger +# ------------------------------- +logger = QEFFLogger.get_logger(NAMESPACE, "DEBUG") + + +# ------------------------------- +# Worker function for threads +# ------------------------------- +def log_worker(thread_id): + for i in range(5): + logger.info(f"Thread-{thread_id} logging message {i}") + time.sleep(0.1) + + +# ------------------------------- +# Create and start threads +# ------------------------------- +threads = [] +for t_id in range(3): + t = threading.Thread(target=log_worker, args=(t_id,)) + threads.append(t) + t.start() + +for t in threads: + t.join() + +# ------------------------------- +# Graceful shutdown +# ------------------------------- +QEFFLogger.close_logger() From fbf3dbb218dca1477304766cc4b8d3f7b2d5f39f Mon Sep 17 00:00:00 2001 From: abhishek-singh591 Date: Fri, 2 Jan 2026 09:03:06 +0000 Subject: [PATCH 02/18] Minor fixes Signed-off-by: abhishek-singh591 --- pyproject.toml | 1 + 1 file changed, 1 insertion(+) diff --git a/pyproject.toml b/pyproject.toml index 9da98f71dc..7367a6c6b2 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -44,6 +44,7 @@ dependencies = [ "imageio==2.37.2", "imageio-ffmpeg==0.6.0", "torch==2.7.0; platform_machine=='aarch64'", + "tabulate", # Specifying torch cpu package URL per python version, update the list once pytorch releases whl for python>3.11 "torch@https://download.pytorch.org/whl/cpu/torch-2.4.1%2Bcpu-cp38-cp38-linux_x86_64.whl ; python_version=='3.8' and platform_machine=='x86_64'", "torch@https://download.pytorch.org/whl/cpu/torch-2.7.0%2Bcpu-cp39-cp39-manylinux_2_28_x86_64.whl ; python_version=='3.9' and platform_machine=='x86_64'", From 0ae92049c6b7b96a7552681e8cf56c7d99996773 Mon Sep 17 00:00:00 2001 From: abhishek-singh591 Date: Mon, 5 Jan 2026 07:24:30 +0000 Subject: [PATCH 03/18] Fixed finetunning logger Signed-off-by: abhishek-singh591 --- QEfficient/cloud/finetune.py | 4 +--- QEfficient/finetune/dataset/custom_dataset.py | 4 +--- QEfficient/finetune/dataset/grammar_dataset.py | 4 +--- QEfficient/finetune/utils/dataset_utils.py | 4 +--- scripts/finetune/run_ft_model.py | 4 +--- 5 files changed, 5 insertions(+), 15 deletions(-) diff --git a/QEfficient/cloud/finetune.py b/QEfficient/cloud/finetune.py index 3e67e33544..35ebbde326 100644 --- a/QEfficient/cloud/finetune.py +++ b/QEfficient/cloud/finetune.py @@ -29,12 +29,10 @@ from QEfficient.finetune.utils.dataset_utils import get_dataloader, get_longest_seq_length from QEfficient.finetune.utils.device_map import get_device_map from QEfficient.finetune.utils.helper import Task_Mode, get_world_size +from QEfficient.finetune.utils.logging_utils import logger from QEfficient.finetune.utils.parser import get_finetune_parser from QEfficient.finetune.utils.train_utils import print_model_size, print_trainable_parameters, train from QEfficient.utils._utils import hf_download -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT", loglevel="INFO") # Try importing QAIC-specific module, proceed without it if unavailable try: diff --git a/QEfficient/finetune/dataset/custom_dataset.py b/QEfficient/finetune/dataset/custom_dataset.py index 2a98a3ca46..ef76e83ed0 100644 --- a/QEfficient/finetune/dataset/custom_dataset.py +++ b/QEfficient/finetune/dataset/custom_dataset.py @@ -9,9 +9,7 @@ import logging from pathlib import Path -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +from QEfficient.finetune.utils.logging_utils import logger def load_module_from_py_file(py_file: str) -> object: diff --git a/QEfficient/finetune/dataset/grammar_dataset.py b/QEfficient/finetune/dataset/grammar_dataset.py index 4a2e4658d8..8fb3eb1521 100644 --- a/QEfficient/finetune/dataset/grammar_dataset.py +++ b/QEfficient/finetune/dataset/grammar_dataset.py @@ -10,9 +10,7 @@ from datasets import load_dataset from torch.utils.data import Dataset -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +from QEfficient.finetune.utils.logging_utils import logger class grammar(Dataset): diff --git a/QEfficient/finetune/utils/dataset_utils.py b/QEfficient/finetune/utils/dataset_utils.py index 9e15a73227..01c1e32aa0 100644 --- a/QEfficient/finetune/utils/dataset_utils.py +++ b/QEfficient/finetune/utils/dataset_utils.py @@ -16,9 +16,7 @@ from QEfficient.finetune.data.sampler import DistributedLengthBasedBatchSampler from QEfficient.finetune.dataset.dataset_config import DATALOADER_COLLATE_FUNC, DATASET_PREPROC from QEfficient.finetune.utils.helper import get_world_size -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +from QEfficient.finetune.utils.logging_utils import logger def get_preprocessed_dataset( diff --git a/scripts/finetune/run_ft_model.py b/scripts/finetune/run_ft_model.py index 97ffb80cb7..f5b64e717b 100644 --- a/scripts/finetune/run_ft_model.py +++ b/scripts/finetune/run_ft_model.py @@ -14,9 +14,7 @@ from transformers import AutoModelForCausalLM, AutoTokenizer from QEfficient.finetune.configs.training import TrainConfig -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +from QEfficient.finetune.utils.logging_utils import logger # Suppress all warnings warnings.filterwarnings("ignore") From f0fb8a759d4e848922b322af849faafe41f59b0e Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Tue, 10 Feb 2026 06:09:44 +0000 Subject: [PATCH 04/18] Made Minnor fix Signed-off-by: Abhishek Kumar Singh --- QEfficient/utils/torch_patches.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/QEfficient/utils/torch_patches.py b/QEfficient/utils/torch_patches.py index b0fbcc45e4..3f7f83ad7f 100644 --- a/QEfficient/utils/torch_patches.py +++ b/QEfficient/utils/torch_patches.py @@ -11,7 +11,9 @@ import torch.onnx.utils as onnx_utils from torch import _C -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") # Store original references before patching _original_setup_trace_module_map = onnx_utils._setup_trace_module_map From 276c850d0eb6c13b926b5e25a1f849217ceff781 Mon Sep 17 00:00:00 2001 From: abhishek-singh591 Date: Thu, 26 Feb 2026 04:26:13 +0000 Subject: [PATCH 05/18] Made minor fix Signed-off-by: abhishek-singh591 --- .../generation/text_generation_inference.py | 3 +- .../transformers/models/modeling_auto.py | 13 +- QEfficient/utils/logging_utils.py | 121 ++++++++++-------- examples/text_generation/basic_inference.py | 2 - 4 files changed, 78 insertions(+), 61 deletions(-) diff --git a/QEfficient/generation/text_generation_inference.py b/QEfficient/generation/text_generation_inference.py index 8f70648beb..b453498bbb 100755 --- a/QEfficient/generation/text_generation_inference.py +++ b/QEfficient/generation/text_generation_inference.py @@ -1318,5 +1318,6 @@ def generate( generated_ids=self._qaic_model.generated_ids, perf_metrics=perf_metrics, ) - logger.info("Text Generated finised") + logger.info("Text generation finised") + QEFFLogger.print_table() return latency_stats diff --git a/QEfficient/transformers/models/modeling_auto.py b/QEfficient/transformers/models/modeling_auto.py index 20fa6cc3e1..bfd1a360ad 100644 --- a/QEfficient/transformers/models/modeling_auto.py +++ b/QEfficient/transformers/models/modeling_auto.py @@ -133,7 +133,7 @@ def from_pretrained(cls, pretrained_model_name_or_path: str, *args, **kwargs): logger.warning("Updating low_cpu_mem_usage=False") kwargs.update({"attn_implementation": "eager", "low_cpu_mem_usage": False}) - + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, *args, **kwargs) return cls(model, pretrained_model_name_or_path=pretrained_model_name_or_path) @@ -290,7 +290,7 @@ def from_pretrained(cls, pretrained_model_name_or_path, pooling=None, *args, **k logger.warning("Updating low_cpu_mem_usage=False") kwargs.update({"attn_implementation": "eager", "low_cpu_mem_usage": False}) - + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, *args, **kwargs) # This is support models that should be classified to in a different auto class but transformers load them via this class @@ -644,7 +644,7 @@ def from_pretrained(cls, pretrained_model_name_or_path, *args, **kwargs): logger.warning("Updating low_cpu_mem_usage=False") kwargs.update({"attn_implementation": "eager", "low_cpu_mem_usage": False}) - + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, *args, **kwargs) return cls(model, pretrained_model_name_or_path=pretrained_model_name_or_path, **kwargs) @@ -1178,7 +1178,7 @@ def from_pretrained(cls, pretrained_model_name_or_path: str, qaic_config: Option logger.warning("Updating low_cpu_mem_usage=False") kwargs.update({"attn_implementation": "eager", "low_cpu_mem_usage": False}) - + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, **kwargs) return cls( model, @@ -1928,6 +1928,7 @@ def from_pretrained( config = AutoConfig.from_pretrained(pretrained_model_name_or_path, trust_remote_code=True) config._attn_implementation = "eager" config.vision_config.use_flash_attn = "false" + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, config, *args, **kwargs) return cls( @@ -2512,6 +2513,7 @@ def from_pretrained( logger.warning("Updating low_cpu_mem_usage=False") kwargs.update({"attn_implementation": "eager", "low_cpu_mem_usage": False}) + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, **kwargs) return cls( model, @@ -2736,6 +2738,7 @@ def from_pretrained( kv_offload = kwargs.pop("kv_offload", None) kwargs.update({"attn_implementation": "eager", "low_cpu_mem_usage": False}) + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, *args, **kwargs) if qaic_config is not None: qaic_config["pretrained_model_name_or_path"] = pretrained_model_name_or_path @@ -3890,7 +3893,7 @@ def from_pretrained(cls, pretrained_model_name_or_path, pooling=None, *args, **k logger.warning("Updating low_cpu_mem_usage=False") kwargs.update({"attn_implementation": "eager", "low_cpu_mem_usage": False}) - + logger.info("Initiating the model weight loading.") model = cls._hf_auto_class.from_pretrained(pretrained_model_name_or_path, *args, **kwargs) # This is support models that should be classified to in a different auto class but transformers load them via this class diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index f2659dff6b..cb1372a2f8 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -242,19 +242,26 @@ def _parse_dt(cls, date_str: str, time_str: str) -> datetime: def print_table(cls) -> None: """ Parse the line-delimited JSON log in cls._logfile and print timing table with t1 as baseline (0.0s): - - Model Loading : t2 - t1 - - Model Exporting : t3 - t2 - - Model Compilation : t4 - t3 - - Text Generation : t5 - t4 - - Total Time : t5 - t1 - - Milestones (matched to your log sample): - t1: first log line timestamp (baseline) - t2: "PyTorch export successful" - t3: "Transformed ONNX saved" - t4: "Model compilation is finished and saved" - t5: "Text Generated finised" - If t5 is missing, we fall back to "specialization_file_path" as readiness marker. + - Model Loading : t2 - t1 + - Model Exporting : t3 - t2 + - Model Compilation : t4 - t3 + - Text Generation : t5 - t4 + - Total Time : t5 - t1 + + Milestones (case-insensitive): + START_LOAD : "Initiating the model weight loading." -> t1 (baseline) + LOAD_DONE : "Pytorch transforms applied to model" -> t2 (end of loading) + ONNX_SAVED : "Transformed ONNX saved" -> t3 + COMPILE_DONE: "Model compilation is finished and saved" -> t4 + TEXT_DONE : "Text Generated finised" -> t5 candidate + TEXT_READY : "specialization_file_path" -> t5 fallback + + Notes: + - t2 is always LOAD_DONE (no conditional fallback). If missing, falls back to t1. + - t3 falls back to t2 if ONNX_SAVED is absent. + - t4 falls back to t3 if COMPILE_DONE is absent. + - t5 prefers TEXT_DONE; else TEXT_READY; else t4. + - Timestamps are enforced to be strictly increasing by +1 ms when equal/non-monotonic. """ path = cls._logfile if not path: @@ -262,16 +269,33 @@ def print_table(cls) -> None: if not os.path.exists(path): raise FileNotFoundError(f"Log file does not exist: {path}") + # Milestone map (case-insensitive) + SUBSTR_TO_KEY: Dict[str, str] = { + "initiating the model weight loading.": "START_LOAD", + "pytorch transforms applied to model": "LOAD_DONE", + "transformed onnx saved": "ONNX_SAVED", + "model compilation is finished and saved": "COMPILE_DONE", + "text generated finised": "TEXT_DONE", + "specialization_file_path": "TEXT_READY", + } + + def classify(msg: str) -> Optional[str]: + m = msg.lower() + for needle, key in SUBSTR_TO_KEY.items(): + if needle in m: + return key + return None + + from datetime import timedelta + t_start: Optional[datetime] = None - t_export_done: Optional[datetime] = None - t_onnx_saved: Optional[datetime] = None - t_compile_done: Optional[datetime] = None - t_text_done: Optional[datetime] = None - t_text_ready: Optional[datetime] = None + last_ts: Optional[datetime] = None + times: Dict[str, datetime] = {} + # Scan the log; enforce strictly increasing timestamps with open(path, "r", encoding="utf-8") as f: - for line in f: - line = line.strip() + for raw in f: + line = raw.strip() if not line: continue try: @@ -287,50 +311,41 @@ def print_table(cls) -> None: ts = cls._parse_dt(date_str, time_str) - if t_start is None: - t_start = ts + if last_ts is not None and ts <= last_ts: + ts = last_ts + timedelta(milliseconds=1) - if ("PyTorch export successful" in msg) and (t_export_done is None): - t_export_done = ts + key = classify(msg) + if key and key not in times: + times[key] = ts + if key == "START_LOAD" and t_start is None: + t_start = ts - if ("Transformed ONNX saved" in msg) and (t_onnx_saved is None): - t_onnx_saved = ts - - if ("Model compilation is finished and saved" in msg) and (t_compile_done is None): - t_compile_done = ts - - if ("Text Generated finised" in msg) and (t_text_done is None): - t_text_done = ts - - if ("specialization_file_path" in msg) and (t_text_ready is None): - t_text_ready = ts + last_ts = ts if t_start is None: - raise ValueError("Could not determine start time (no valid log lines with date/time).") - - if t_text_done is None: - t_text_done = t_text_ready + raise ValueError("Missing required milestone: 'Initiating the model weight loading.'") - t_export_done = t_export_done or t_start - t_onnx_saved = t_onnx_saved or t_export_done - t_compile_done = t_compile_done or t_onnx_saved - t_text_done = t_text_done or t_compile_done + # Resolve chain + t2 = times.get("LOAD_DONE", t_start) # end of loading + t3 = times.get("ONNX_SAVED") or t2 # export end + t4 = times.get("COMPILE_DONE") or t3 # compile end + t5 = (times.get("TEXT_DONE") or times.get("TEXT_READY") or t4) # text gen end - def to_offset_seconds(t: datetime) -> float: + def offset_seconds(t: datetime) -> float: return (t - t_start).total_seconds() o1 = 0.0 - o2 = to_offset_seconds(t_export_done) - o3 = to_offset_seconds(t_onnx_saved) - o4 = to_offset_seconds(t_compile_done) - o5 = to_offset_seconds(t_text_done) + o2 = offset_seconds(t2) + o3 = offset_seconds(t3) + o4 = offset_seconds(t4) + o5 = offset_seconds(t5) timing_data: List[List[Any]] = [ - ["Model Loading", max(0.0, o2 - o1)], - ["Model Exporting", max(0.0, o3 - o2)], + ["Model Loading", max(0.0, o2 - o1)], + ["Model Exporting", max(0.0, o3 - o2)], ["Model Compilation", max(0.0, o4 - o3)], - ["Text Generation", max(0.0, o5 - o4)], - ["Total Time", max(0.0, o5 - o1)], + ["Text Generation", max(0.0, o5 - o4)], + ["Total Time", max(0.0, o5 - o1)], ] - print(tabulate(timing_data, headers=["Step", "Time (s)"], tablefmt="github", floatfmt=".3f")) + print(tabulate(timing_data, headers=["Step", "Time (s)"], tablefmt="github", floatfmt=".3f")) \ No newline at end of file diff --git a/examples/text_generation/basic_inference.py b/examples/text_generation/basic_inference.py index a4041d514e..6340ec7256 100644 --- a/examples/text_generation/basic_inference.py +++ b/examples/text_generation/basic_inference.py @@ -10,7 +10,6 @@ from transformers import AutoTokenizer from QEfficient import QEFFAutoModelForCausalLM -from QEfficient.utils.logging_utils import QEFFLogger def main(): @@ -52,7 +51,6 @@ def main(): print(f"\nPrompt: {args.prompt}") print(f"Generated: {exec_info.generated_texts[0]}") - QEFFLogger.print_table() if __name__ == "__main__": From 9a5bde4fea0c091069652a340f1d82cedd8ba4c7 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Thu, 26 Feb 2026 04:31:23 +0000 Subject: [PATCH 06/18] fixed lint error Signed-off-by: Abhishek Kumar Singh --- QEfficient/utils/logging_utils.py | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index cb1372a2f8..c28c9718a1 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -327,9 +327,9 @@ def classify(msg: str) -> Optional[str]: # Resolve chain t2 = times.get("LOAD_DONE", t_start) # end of loading - t3 = times.get("ONNX_SAVED") or t2 # export end + t3 = times.get("ONNX_SAVED") or t2 # export end t4 = times.get("COMPILE_DONE") or t3 # compile end - t5 = (times.get("TEXT_DONE") or times.get("TEXT_READY") or t4) # text gen end + t5 = times.get("TEXT_DONE") or times.get("TEXT_READY") or t4 # text gen end def offset_seconds(t: datetime) -> float: return (t - t_start).total_seconds() @@ -341,11 +341,11 @@ def offset_seconds(t: datetime) -> float: o5 = offset_seconds(t5) timing_data: List[List[Any]] = [ - ["Model Loading", max(0.0, o2 - o1)], - ["Model Exporting", max(0.0, o3 - o2)], + ["Model Loading", max(0.0, o2 - o1)], + ["Model Exporting", max(0.0, o3 - o2)], ["Model Compilation", max(0.0, o4 - o3)], - ["Text Generation", max(0.0, o5 - o4)], - ["Total Time", max(0.0, o5 - o1)], + ["Text Generation", max(0.0, o5 - o4)], + ["Total Time", max(0.0, o5 - o1)], ] - print(tabulate(timing_data, headers=["Step", "Time (s)"], tablefmt="github", floatfmt=".3f")) \ No newline at end of file + print(tabulate(timing_data, headers=["Step", "Time (s)"], tablefmt="github", floatfmt=".3f")) From 9c01ff742fd6dd6c37ea1aeb99cc8729ce9e4f12 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Mon, 16 Mar 2026 10:38:26 +0000 Subject: [PATCH 07/18] Addressed few of the comments Signed-off-by: Abhishek Kumar Singh --- QEfficient/base/modeling_qeff.py | 1 - QEfficient/generation/text_generation_inference.py | 2 +- QEfficient/utils/logging_utils.py | 2 +- pyproject.toml | 2 +- 4 files changed, 3 insertions(+), 4 deletions(-) diff --git a/QEfficient/base/modeling_qeff.py b/QEfficient/base/modeling_qeff.py index 0bb1176927..ed09a3951c 100644 --- a/QEfficient/base/modeling_qeff.py +++ b/QEfficient/base/modeling_qeff.py @@ -320,7 +320,6 @@ def _export( self.onnx_path = onnx_path logger.info("Model export is finished and saved at: %s", onnx_path) - self.onnx_path = onnx_path return onnx_path def get_onnx_path( diff --git a/QEfficient/generation/text_generation_inference.py b/QEfficient/generation/text_generation_inference.py index b453498bbb..45666c49ed 100755 --- a/QEfficient/generation/text_generation_inference.py +++ b/QEfficient/generation/text_generation_inference.py @@ -1318,6 +1318,6 @@ def generate( generated_ids=self._qaic_model.generated_ids, perf_metrics=perf_metrics, ) - logger.info("Text generation finised") + logger.info("Text generation finished") QEFFLogger.print_table() return latency_stats diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index c28c9718a1..8ff5c52f97 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -347,5 +347,5 @@ def offset_seconds(t: datetime) -> float: ["Text Generation", max(0.0, o5 - o4)], ["Total Time", max(0.0, o5 - o1)], ] - + print("\n") print(tabulate(timing_data, headers=["Step", "Time (s)"], tablefmt="github", floatfmt=".3f")) diff --git a/pyproject.toml b/pyproject.toml index 9378303002..ca96a777af 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -44,7 +44,7 @@ dependencies = [ "imageio==2.37.2", "imageio-ffmpeg==0.6.0", "torch==2.7.0; platform_machine=='aarch64'", - "tabulate", + "tabulate==0.9.0", # Specifying torch cpu package URL per python version, update the list once pytorch releases whl for python>3.11 "torch@https://download.pytorch.org/whl/cpu/torch-2.4.1%2Bcpu-cp38-cp38-linux_x86_64.whl ; python_version=='3.8' and platform_machine=='x86_64'", "torch@https://download.pytorch.org/whl/cpu/torch-2.7.0%2Bcpu-cp39-cp39-manylinux_2_28_x86_64.whl ; python_version=='3.9' and platform_machine=='x86_64'", From eb442f8bf9f60a42c709fd75838ea9a637fdff59 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Mon, 16 Mar 2026 11:31:41 +0000 Subject: [PATCH 08/18] Addressed a few more comments Signed-off-by: Abhishek Kumar Singh --- QEfficient/__init__.py | 2 +- QEfficient/base/modeling_qeff.py | 2 +- QEfficient/base/pytorch_transforms.py | 2 +- QEfficient/cloud/export.py | 2 +- QEfficient/cloud/infer.py | 2 +- QEfficient/compile/compile_helper.py | 2 +- QEfficient/compile/qnn_compiler.py | 2 +- .../models/transformers/transformer_flux.py | 2 +- .../diffusers/pipelines/flux/pipeline_flux.py | 2 +- .../diffusers/pipelines/pipeline_utils.py | 2 +- .../diffusers/pipelines/wan/pipeline_wan.py | 2 +- .../exporter/export_hf_to_cloud_ai_100.py | 2 +- QEfficient/finetune/dataset/alpaca_dataset.py | 2 +- QEfficient/finetune/utils/config_utils.py | 2 +- QEfficient/finetune/utils/plot_metrics.py | 2 +- QEfficient/generation/embedding_handler.py | 2 +- .../generation/text_generation_inference.py | 2 +- QEfficient/generation/vlm_generation.py | 2 +- QEfficient/peft/auto.py | 2 +- QEfficient/peft/lora/auto.py | 2 +- .../models/gpt_oss/modeling_gpt_oss.py | 2 +- .../models/internvl/modeling_internvl.py | 2 +- .../models/llava/modeling_llava.py | 2 +- .../models/llava_next/modeling_llava_next.py | 2 +- .../models/mistral3/modeling_mistral3.py | 2 +- .../transformers/models/modeling_auto.py | 2 +- .../models/qwen2_5_vl/modeling_qwen2_5_vl.py | 2 +- .../transformers/quantizers/quantizer_awq.py | 2 +- .../quantizer_compressed_tensors.py | 2 +- .../transformers/quantizers/quantizer_gptq.py | 2 +- .../quantizers/quantizer_mxfp4.py | 2 +- QEfficient/transformers/transform.py | 2 +- QEfficient/utils/_utils.py | 2 +- QEfficient/utils/check_ccl_specializations.py | 2 +- QEfficient/utils/constants.py | 21 +++ QEfficient/utils/device_utils.py | 2 +- QEfficient/utils/export_utils.py | 2 +- QEfficient/utils/logging_utils.py | 162 +++++++----------- QEfficient/utils/sampler_utils.py | 2 +- QEfficient/utils/torch_patches.py | 2 +- .../calculate_perplexity.py | 2 +- tests/conftest.py | 2 +- .../test_transformer_pytorch_transforms.py | 2 +- 43 files changed, 121 insertions(+), 144 deletions(-) diff --git a/QEfficient/__init__.py b/QEfficient/__init__.py index 4a4aefa86b..1f109fab89 100644 --- a/QEfficient/__init__.py +++ b/QEfficient/__init__.py @@ -38,7 +38,7 @@ from QEfficient.utils import custom_format_warning from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") # custom warning for the better logging experience warnings.formatwarning = custom_format_warning diff --git a/QEfficient/base/modeling_qeff.py b/QEfficient/base/modeling_qeff.py index 9a34d12cf5..c8af3d284e 100644 --- a/QEfficient/base/modeling_qeff.py +++ b/QEfficient/base/modeling_qeff.py @@ -36,7 +36,7 @@ from QEfficient.utils.export_utils import export_wrapper from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") class QEFFBaseModel(ABC): diff --git a/QEfficient/base/pytorch_transforms.py b/QEfficient/base/pytorch_transforms.py index 2f4085ed71..6a5d64e96b 100644 --- a/QEfficient/base/pytorch_transforms.py +++ b/QEfficient/base/pytorch_transforms.py @@ -11,7 +11,7 @@ from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") class PytorchTransform: diff --git a/QEfficient/cloud/export.py b/QEfficient/cloud/export.py index 48d10c7006..f50b78c5d0 100644 --- a/QEfficient/cloud/export.py +++ b/QEfficient/cloud/export.py @@ -14,7 +14,7 @@ from QEfficient.utils.custom_yaml import generate_custom_io from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") # Specifically for Docker images. ROOT_DIR = os.path.dirname(os.path.abspath("")) diff --git a/QEfficient/cloud/infer.py b/QEfficient/cloud/infer.py index 4e85845ff6..ab5dc194c7 100644 --- a/QEfficient/cloud/infer.py +++ b/QEfficient/cloud/infer.py @@ -19,7 +19,7 @@ from QEfficient.utils import check_and_assign_cache_dir, load_hf_processor, load_hf_tokenizer from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") # TODO: Remove after adding support for VLM's compile and execute diff --git a/QEfficient/compile/compile_helper.py b/QEfficient/compile/compile_helper.py index bc7133503f..687fee33c2 100644 --- a/QEfficient/compile/compile_helper.py +++ b/QEfficient/compile/compile_helper.py @@ -17,7 +17,7 @@ from QEfficient.utils._utils import load_json, load_yaml from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") def create_and_dump_specializations( diff --git a/QEfficient/compile/qnn_compiler.py b/QEfficient/compile/qnn_compiler.py index 9fcadb6d01..16120cf343 100644 --- a/QEfficient/compile/qnn_compiler.py +++ b/QEfficient/compile/qnn_compiler.py @@ -20,7 +20,7 @@ from QEfficient.utils.hash_utils import to_hashable from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") class QNN: diff --git a/QEfficient/diffusers/models/transformers/transformer_flux.py b/QEfficient/diffusers/models/transformers/transformer_flux.py index 8c74deab1e..9ce7432230 100644 --- a/QEfficient/diffusers/models/transformers/transformer_flux.py +++ b/QEfficient/diffusers/models/transformers/transformer_flux.py @@ -22,7 +22,7 @@ from QEfficient.diffusers.models.modeling_utils import compute_blocked_attention, get_attention_blocking_config from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") def qeff_apply_rotary_emb( diff --git a/QEfficient/diffusers/pipelines/flux/pipeline_flux.py b/QEfficient/diffusers/pipelines/flux/pipeline_flux.py index 28d11f2613..407afc071a 100644 --- a/QEfficient/diffusers/pipelines/flux/pipeline_flux.py +++ b/QEfficient/diffusers/pipelines/flux/pipeline_flux.py @@ -40,7 +40,7 @@ from QEfficient.generation.cloud_infer import QAICInferenceSession from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEffFluxPipeline: diff --git a/QEfficient/diffusers/pipelines/pipeline_utils.py b/QEfficient/diffusers/pipelines/pipeline_utils.py index e42040feaa..dbc2d2b705 100644 --- a/QEfficient/diffusers/pipelines/pipeline_utils.py +++ b/QEfficient/diffusers/pipelines/pipeline_utils.py @@ -18,7 +18,7 @@ from QEfficient.utils._utils import load_json from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") def calculate_compressed_latent_dimension(height: int, width: int, vae_scale_factor: int) -> int: diff --git a/QEfficient/diffusers/pipelines/wan/pipeline_wan.py b/QEfficient/diffusers/pipelines/wan/pipeline_wan.py index fc0cc410bc..fe587ca177 100644 --- a/QEfficient/diffusers/pipelines/wan/pipeline_wan.py +++ b/QEfficient/diffusers/pipelines/wan/pipeline_wan.py @@ -39,7 +39,7 @@ from QEfficient.utils import constants from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEffWanPipeline: diff --git a/QEfficient/exporter/export_hf_to_cloud_ai_100.py b/QEfficient/exporter/export_hf_to_cloud_ai_100.py index 73f9d00cfe..7b753b374f 100644 --- a/QEfficient/exporter/export_hf_to_cloud_ai_100.py +++ b/QEfficient/exporter/export_hf_to_cloud_ai_100.py @@ -22,7 +22,7 @@ from QEfficient.utils.generate_inputs import InputHandler from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") def convert_to_cloud_bertstyle( diff --git a/QEfficient/finetune/dataset/alpaca_dataset.py b/QEfficient/finetune/dataset/alpaca_dataset.py index f5fd5637ea..f790efa368 100644 --- a/QEfficient/finetune/dataset/alpaca_dataset.py +++ b/QEfficient/finetune/dataset/alpaca_dataset.py @@ -13,7 +13,7 @@ from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +logger = QEFFLogger.get_logger("FT") PROMPT_DICT = { "prompt_input": ( diff --git a/QEfficient/finetune/utils/config_utils.py b/QEfficient/finetune/utils/config_utils.py index 89d1753d9a..9757ba5038 100644 --- a/QEfficient/finetune/utils/config_utils.py +++ b/QEfficient/finetune/utils/config_utils.py @@ -22,7 +22,7 @@ from QEfficient.finetune.utils.helper import Peft_Method from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +logger = QEFFLogger.get_logger("FT") def update_config(config, **kwargs): diff --git a/QEfficient/finetune/utils/plot_metrics.py b/QEfficient/finetune/utils/plot_metrics.py index 4ea307dff7..545894244c 100644 --- a/QEfficient/finetune/utils/plot_metrics.py +++ b/QEfficient/finetune/utils/plot_metrics.py @@ -13,7 +13,7 @@ from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +logger = QEFFLogger.get_logger("FT") def plot_metric(data, metric_name, x_label, y_label, title, colors): diff --git a/QEfficient/generation/embedding_handler.py b/QEfficient/generation/embedding_handler.py index c33906b619..51f023b4d5 100644 --- a/QEfficient/generation/embedding_handler.py +++ b/QEfficient/generation/embedding_handler.py @@ -25,7 +25,7 @@ from QEfficient.utils import constants from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") class VisionHandler: diff --git a/QEfficient/generation/text_generation_inference.py b/QEfficient/generation/text_generation_inference.py index 45666c49ed..2ae182e747 100755 --- a/QEfficient/generation/text_generation_inference.py +++ b/QEfficient/generation/text_generation_inference.py @@ -22,7 +22,7 @@ from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.sampler_utils import validate_sampler_inputs -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") @dataclass diff --git a/QEfficient/generation/vlm_generation.py b/QEfficient/generation/vlm_generation.py index 3dfd68efbd..3a1b1bde15 100644 --- a/QEfficient/generation/vlm_generation.py +++ b/QEfficient/generation/vlm_generation.py @@ -39,7 +39,7 @@ from QEfficient.utils.constants import Constants from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") class VisionLanguageGeneration(QEffTextGenerationBase): diff --git a/QEfficient/peft/auto.py b/QEfficient/peft/auto.py index f151473244..324c207a1b 100644 --- a/QEfficient/peft/auto.py +++ b/QEfficient/peft/auto.py @@ -33,7 +33,7 @@ from QEfficient.utils.hash_utils import to_hashable from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +logger = QEFFLogger.get_logger("FT") class QEffAutoPeftModelForCausalLM(QEFFBaseModel): diff --git a/QEfficient/peft/lora/auto.py b/QEfficient/peft/lora/auto.py index 7b98b8e4ee..86c61184af 100644 --- a/QEfficient/peft/lora/auto.py +++ b/QEfficient/peft/lora/auto.py @@ -21,7 +21,7 @@ from QEfficient.utils.hash_utils import to_hashable from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("FT", loglevel="INFO") +logger = QEFFLogger.get_logger("FT") class QEffAutoLoraModelForCausalLM(QEFFAutoModelForCausalLM): diff --git a/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py b/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py index 9fc4e9d823..a40e225525 100644 --- a/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py +++ b/QEfficient/transformers/models/gpt_oss/modeling_gpt_oss.py @@ -35,7 +35,7 @@ from QEfficient.utils.constants import MIN_MASKED_ATTENTION_VALUE from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEffGptOssExperts(GptOssExperts): diff --git a/QEfficient/transformers/models/internvl/modeling_internvl.py b/QEfficient/transformers/models/internvl/modeling_internvl.py index 15ccdc7299..6e601fedbe 100644 --- a/QEfficient/transformers/models/internvl/modeling_internvl.py +++ b/QEfficient/transformers/models/internvl/modeling_internvl.py @@ -15,7 +15,7 @@ from QEfficient.utils._utils import IOInfo, get_padding_shape_from_config from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEffInternEncoderWrapper(nn.Module): diff --git a/QEfficient/transformers/models/llava/modeling_llava.py b/QEfficient/transformers/models/llava/modeling_llava.py index 8a83ce184f..5211d04aff 100644 --- a/QEfficient/transformers/models/llava/modeling_llava.py +++ b/QEfficient/transformers/models/llava/modeling_llava.py @@ -17,7 +17,7 @@ from QEfficient.utils._utils import IOInfo from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") BS = 1 FBS = 4 diff --git a/QEfficient/transformers/models/llava_next/modeling_llava_next.py b/QEfficient/transformers/models/llava_next/modeling_llava_next.py index fa7bb1e8b8..ef32739f90 100755 --- a/QEfficient/transformers/models/llava_next/modeling_llava_next.py +++ b/QEfficient/transformers/models/llava_next/modeling_llava_next.py @@ -20,7 +20,7 @@ from QEfficient.utils._utils import IOInfo from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") BS = constants.ONNX_EXPORT_EXAMPLE_BATCH_SIZE FBS = constants.ONNX_EXPORT_EXAMPLE_FBS diff --git a/QEfficient/transformers/models/mistral3/modeling_mistral3.py b/QEfficient/transformers/models/mistral3/modeling_mistral3.py index 91c0f2506d..61ff1c04fc 100644 --- a/QEfficient/transformers/models/mistral3/modeling_mistral3.py +++ b/QEfficient/transformers/models/mistral3/modeling_mistral3.py @@ -23,7 +23,7 @@ from QEfficient.utils._utils import IOInfo, get_padding_shape_from_config from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") def custom_cumsum(tensor): diff --git a/QEfficient/transformers/models/modeling_auto.py b/QEfficient/transformers/models/modeling_auto.py index 98d1a3190c..504910c367 100644 --- a/QEfficient/transformers/models/modeling_auto.py +++ b/QEfficient/transformers/models/modeling_auto.py @@ -76,7 +76,7 @@ from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.sampler_utils import get_sampling_inputs_and_outputs -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEFFTransformersBase(QEFFBaseModel): diff --git a/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py b/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py index cfb2382d16..0731e54736 100644 --- a/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py +++ b/QEfficient/transformers/models/qwen2_5_vl/modeling_qwen2_5_vl.py @@ -40,7 +40,7 @@ from QEfficient.utils.constants import MIN_MASKED_ATTENTION_VALUE from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") def qeff_apply_rotary_pos_emb(q, k, cos, sin, position_ids, mrope_section, unsqueeze_dim=1): diff --git a/QEfficient/transformers/quantizers/quantizer_awq.py b/QEfficient/transformers/quantizers/quantizer_awq.py index 0968bdd271..c8f7648b42 100644 --- a/QEfficient/transformers/quantizers/quantizer_awq.py +++ b/QEfficient/transformers/quantizers/quantizer_awq.py @@ -17,7 +17,7 @@ ) from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEffAwqConfig(AwqConfig): diff --git a/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py b/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py index 86d7ae1395..c1c69c27b0 100644 --- a/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py +++ b/QEfficient/transformers/quantizers/quantizer_compressed_tensors.py @@ -16,7 +16,7 @@ from QEfficient.transformers.quantizers.quantizer_utils import get_keys_to_not_convert from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") FP8_DTYPE = torch.float8_e4m3fn diff --git a/QEfficient/transformers/quantizers/quantizer_gptq.py b/QEfficient/transformers/quantizers/quantizer_gptq.py index 073f370e18..4117359f6f 100644 --- a/QEfficient/transformers/quantizers/quantizer_gptq.py +++ b/QEfficient/transformers/quantizers/quantizer_gptq.py @@ -17,7 +17,7 @@ ) from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEffGPTQConfig(GPTQConfig): diff --git a/QEfficient/transformers/quantizers/quantizer_mxfp4.py b/QEfficient/transformers/quantizers/quantizer_mxfp4.py index bfbfe473e2..3a5a92a3db 100644 --- a/QEfficient/transformers/quantizers/quantizer_mxfp4.py +++ b/QEfficient/transformers/quantizers/quantizer_mxfp4.py @@ -16,7 +16,7 @@ from QEfficient.transformers.quantizers.quantizer_utils import convert_moe_packed_tensors, get_keys_to_not_convert from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("MODEL", loglevel="INFO") +logger = QEFFLogger.get_logger("MODEL") class QEffMxfp4GptOssExperts(nn.Module): diff --git a/QEfficient/transformers/transform.py b/QEfficient/transformers/transform.py index ecbc16a384..4fde89039f 100644 --- a/QEfficient/transformers/transform.py +++ b/QEfficient/transformers/transform.py @@ -15,7 +15,7 @@ from QEfficient.transformers.modeling_utils import TransformersToQEffModulesDict from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") def replace_module_with_qeff_layers(model: nn.Module) -> None: diff --git a/QEfficient/utils/_utils.py b/QEfficient/utils/_utils.py index 21d60fa1c4..d41fc79b19 100644 --- a/QEfficient/utils/_utils.py +++ b/QEfficient/utils/_utils.py @@ -30,7 +30,7 @@ from QEfficient.utils.hash_utils import json_serializable from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") class LRUCache: diff --git a/QEfficient/utils/check_ccl_specializations.py b/QEfficient/utils/check_ccl_specializations.py index 794e88eca6..6b3d4e3f64 100644 --- a/QEfficient/utils/check_ccl_specializations.py +++ b/QEfficient/utils/check_ccl_specializations.py @@ -10,7 +10,7 @@ from QEfficient.utils import constants from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") # Better performance when context length is multiple of 1024 → map CL to the next multiple of 1024 diff --git a/QEfficient/utils/constants.py b/QEfficient/utils/constants.py index 7e6dd1cbbe..53a36456e8 100644 --- a/QEfficient/utils/constants.py +++ b/QEfficient/utils/constants.py @@ -302,3 +302,24 @@ class QnnConstants: }, "SKIP_QNN_CONVERTER_STEP": False, } + + +@dataclass +class LoggerConfig: + """ + Centralized logger configuration for the project. + - Keep all defaults here. + - Environment variable names are centralized to avoid magic strings. + """ + + # Environment variables + log_path_env: str = "QEFF_LOG_PATH" # optional: set log file path + log_level_env: str = "QEFF_LOG_LEVEL" # project-wide log level (e.g., DEBUG/INFO/WARN/ERROR) + + # Defaults + default_log_dir: str = os.path.expanduser("~/.cache/qefficient_logs") + default_level: str = "INFO" # default when env is not set + + # Rotating file behavior + max_bytes: int = 5 * 1024 * 1024 # 5 MB + backup_count: int = 10 # keep last 10 files diff --git a/QEfficient/utils/device_utils.py b/QEfficient/utils/device_utils.py index 72e5b423ee..2e889f79a8 100644 --- a/QEfficient/utils/device_utils.py +++ b/QEfficient/utils/device_utils.py @@ -12,7 +12,7 @@ from QEfficient.utils.constants import Constants from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") def is_networks_loaded(stdout): diff --git a/QEfficient/utils/export_utils.py b/QEfficient/utils/export_utils.py index 46de7a7249..c3f63a0139 100644 --- a/QEfficient/utils/export_utils.py +++ b/QEfficient/utils/export_utils.py @@ -19,7 +19,7 @@ from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.torch_patches import apply_torch_patches, undo_torch_patches -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") def export_wrapper(func): diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index 8ff5c52f97..e144971fb9 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -16,19 +16,13 @@ from tabulate import tabulate +# Import centralized config +from QEfficient.utils.constants import LoggerConfig + class JSONNamespaceFormatter(logging.Formatter): """ Custom formatter to output log records in JSON format with metadata. - - Methods: - format(record): Formats a log record into a JSON string. - - Parameters: - record (logging.LogRecord): The log record to format. - - Returns: - str: JSON-formatted log string. """ def format(self, record): @@ -46,31 +40,25 @@ def format(self, record): class QEFFLoggerThread(threading.Thread): """ - Background thread to handle logging asynchronously using a queue. + Custom formatter to output log records in JSON format with metadata. + + Methods: + format(record): Formats a log record into a JSON string. - Attributes: - logger (logging.Logger): Logger instance to handle log records. - log_queue (queue.Queue): Queue from which log records are consumed. - running (bool): Flag to control thread execution. + Parameters: + record (logging.LogRecord): The log record to format. + + Returns: + str: JSON-formatted log string. """ def __init__(self, logger, log_queue): - """ - Initialize the logging thread. - - Parameters: - logger (logging.Logger): Logger instance. - log_queue (queue.Queue): Queue for log records. - """ super().__init__(daemon=True) self.logger = logger self.log_queue = log_queue self.running = True def run(self): - """ - Continuously process log records from the queue and pass them to the logger. - """ while self.running: try: record = self.log_queue.get(timeout=1) @@ -79,9 +67,6 @@ def run(self): continue def stop(self): - """ - Stop the logging thread gracefully. - """ self.running = False @@ -89,11 +74,9 @@ class QEFFLogger: """ Singleton logger class for structured logging with namespace support. - Class Attributes: - _instance (Optional[logging.Logger]): Singleton logger instance. - _logfile (Optional[str]): Path to the log file. - _log_queue (queue.Queue): Queue for asynchronous logging. - _logger_thread (Optional[QEFFLoggerThread]): Background logging thread. + Project-wide behavior: + - A single log level is enforced using env `QEFF_LOG_LEVEL` (default = INFO). + - Log path resolved with priority: explicit arg > env `QEFF_LOG_PATH` > default dir + timestamp. """ _instance: Optional[logging.Logger] = None @@ -101,17 +84,34 @@ class QEFFLogger: _log_queue: queue.Queue = queue.Queue() _logger_thread: Optional[QEFFLoggerThread] = None - def __init__(self, loglevel: Optional[str] = "INFO", log_path: Optional[str] = None): + def __init__(self, loglevel: Optional[str] = None, log_path: Optional[str] = None): """ - Initialize the logger instance with specified log level and path. - - Parameters: - loglevel (str): Logging level (e.g., "INFO", "DEBUG"). - log_path (str): Optional path to the log file. + Initialize the logger instance with specified path. Level is globally controlled by env. + Args: + loglevel: kept for backward compatibility, but env `QEFF_LOG_LEVEL` takes precedence. + log_path: optional path to the log file (highest priority). """ if QEFFLogger._instance is None: - self.loglevel = loglevel - self.log_path = log_path + # Determine effective log level: + # Priority: ENV(QEFF_LOG_LEVEL) -> arg(loglevel) -> LoggerConfig.default_level + env_level = os.environ.get(LoggerConfig.log_level_env) + effective_level_name = (env_level or loglevel or LoggerConfig.default_level).upper() + numeric_level = getattr(logging, effective_level_name, None) + if not isinstance(numeric_level, int): + raise ValueError(f"Invalid log level: {effective_level_name}") + self.loglevel = effective_level_name + + # Resolve log path (arg > env > default dir + timestamp) + env_path = os.environ.get(LoggerConfig.log_path_env) + self.log_path = log_path or env_path + if not self.log_path: + os.makedirs(LoggerConfig.default_log_dir, exist_ok=True) + timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") + self.log_path = os.path.join(LoggerConfig.default_log_dir, f"QEFF_{timestamp}.log") + else: + os.makedirs(os.path.dirname(os.path.abspath(self.log_path)), exist_ok=True) + + # Initialize the base logger and start background thread self.logger = self._initialize_logger() QEFFLogger._instance = self.logger QEFFLogger._logger_thread = QEFFLoggerThread(self.logger, QEFFLogger._log_queue) @@ -120,27 +120,19 @@ def __init__(self, loglevel: Optional[str] = "INFO", log_path: Optional[str] = N def _initialize_logger(self) -> logging.Logger: """ Set up the logger with rotating file handler and JSON formatter. - - Returns: - logging.Logger: Configured logger instance. """ - if self.log_path is None: - log_dir = os.path.expanduser("~/.cache/qefficient_logs") - os.makedirs(log_dir, exist_ok=True) - timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") - self.log_path = os.path.join(log_dir, f"QEFF_{timestamp}.log") - QEFFLogger._logfile = self.log_path - numeric_level = getattr(logging, self.loglevel.upper(), None) - if not isinstance(numeric_level, int): - raise ValueError(f"Invalid log level: {self.loglevel}") - logger = logging.getLogger("QEFF_LOGGER") - logger.setLevel(numeric_level) + logger.setLevel(getattr(logging, self.loglevel)) + # Avoid duplicate handlers if reinitialized in same process if not logger.handlers: - handler = RotatingFileHandler(self.log_path, maxBytes=5 * 1024 * 1024, backupCount=10) + handler = RotatingFileHandler( + self.log_path, + maxBytes=LoggerConfig.max_bytes, + backupCount=LoggerConfig.backup_count, + ) handler.setFormatter(JSONNamespaceFormatter()) logger.addHandler(handler) @@ -148,18 +140,11 @@ def _initialize_logger(self) -> logging.Logger: @classmethod def get_logger( - cls, namespace: str, loglevel: Optional[str] = "INFO", log_path: Optional[str] = None + cls, namespace: str, loglevel: Optional[str] = None, log_path: Optional[str] = None ) -> logging.Logger: """ Retrieve a logger adapter with a specific namespace. - - Parameters: - namespace (str): Logical grouping for the log. - loglevel (str): Logging level. - log_path (str): Optional path to the log file. - - Returns: - logging.Logger: Logger adapter with namespace. + Note: project-wide level comes from env `QEFF_LOG_LEVEL` (default INFO). """ if cls._instance is None: cls(loglevel, log_path) @@ -169,14 +154,6 @@ def get_logger( def log(cls, level: str, namespace: str, msg: str, fn: str = "", lno: int = 0, func: str = ""): """ Log a message with specified level and metadata. - - Parameters: - level (str): Logging level (e.g., "INFO", "ERROR"). - namespace (str): Logical grouping for the log. - msg (str): Log message. - fn (str): Filename where the log is generated. - lno (int): Line number in the file. - func (str): Function name. """ if cls._instance is None: raise RuntimeError("Logger has not been initialized. Call get_logger() first.") @@ -199,19 +176,20 @@ def log(cls, level: str, namespace: str, msg: str, fn: str = "", lno: int = 0, f cls._log_queue.put(record) @classmethod - def set_loglevel(cls, loglevel: Optional[str] = "INFO"): + def set_loglevel(cls, loglevel: Optional[str] = None): """ - Update the log level of the logger. - - Parameters: - loglevel (str): New log level to set. + Update the log level of the logger at runtime. + Priority remains ENV > arg > default. + If ENV is set, it will continue to override; otherwise arg/default apply. """ if cls._instance is None: raise RuntimeError("Logger has not been initialized yet. Call get_logger() first.") - numeric_level = getattr(logging, loglevel.upper(), None) + env_level = os.environ.get(LoggerConfig.log_level_env) + effective_level_name = (env_level or loglevel or LoggerConfig.default_level).upper() + numeric_level = getattr(logging, effective_level_name, None) if not isinstance(numeric_level, int): - raise ValueError(f"Invalid log level: {loglevel}") + raise ValueError(f"Invalid log level: {effective_level_name}") cls._instance.setLevel(numeric_level) @@ -241,27 +219,7 @@ def _parse_dt(cls, date_str: str, time_str: str) -> datetime: @classmethod def print_table(cls) -> None: """ - Parse the line-delimited JSON log in cls._logfile and print timing table with t1 as baseline (0.0s): - - Model Loading : t2 - t1 - - Model Exporting : t3 - t2 - - Model Compilation : t4 - t3 - - Text Generation : t5 - t4 - - Total Time : t5 - t1 - - Milestones (case-insensitive): - START_LOAD : "Initiating the model weight loading." -> t1 (baseline) - LOAD_DONE : "Pytorch transforms applied to model" -> t2 (end of loading) - ONNX_SAVED : "Transformed ONNX saved" -> t3 - COMPILE_DONE: "Model compilation is finished and saved" -> t4 - TEXT_DONE : "Text Generated finised" -> t5 candidate - TEXT_READY : "specialization_file_path" -> t5 fallback - - Notes: - - t2 is always LOAD_DONE (no conditional fallback). If missing, falls back to t1. - - t3 falls back to t2 if ONNX_SAVED is absent. - - t4 falls back to t3 if COMPILE_DONE is absent. - - t5 prefers TEXT_DONE; else TEXT_READY; else t4. - - Timestamps are enforced to be strictly increasing by +1 ms when equal/non-monotonic. + Parse the line-delimited JSON log in cls._logfile and print timing table with t1 as baseline (0.0s). """ path = cls._logfile if not path: @@ -269,7 +227,6 @@ def print_table(cls) -> None: if not os.path.exists(path): raise FileNotFoundError(f"Log file does not exist: {path}") - # Milestone map (case-insensitive) SUBSTR_TO_KEY: Dict[str, str] = { "initiating the model weight loading.": "START_LOAD", "pytorch transforms applied to model": "LOAD_DONE", @@ -292,7 +249,6 @@ def classify(msg: str) -> Optional[str]: last_ts: Optional[datetime] = None times: Dict[str, datetime] = {} - # Scan the log; enforce strictly increasing timestamps with open(path, "r", encoding="utf-8") as f: for raw in f: line = raw.strip() @@ -311,6 +267,7 @@ def classify(msg: str) -> Optional[str]: ts = cls._parse_dt(date_str, time_str) + # Enforce strictly increasing timestamps if last_ts is not None and ts <= last_ts: ts = last_ts + timedelta(milliseconds=1) @@ -325,7 +282,6 @@ def classify(msg: str) -> Optional[str]: if t_start is None: raise ValueError("Missing required milestone: 'Initiating the model weight loading.'") - # Resolve chain t2 = times.get("LOAD_DONE", t_start) # end of loading t3 = times.get("ONNX_SAVED") or t2 # export end t4 = times.get("COMPILE_DONE") or t3 # compile end diff --git a/QEfficient/utils/sampler_utils.py b/QEfficient/utils/sampler_utils.py index 847266ae80..aa0adb7a0e 100644 --- a/QEfficient/utils/sampler_utils.py +++ b/QEfficient/utils/sampler_utils.py @@ -13,7 +13,7 @@ from QEfficient.utils.constants import Constants from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") def validate_sampler_inputs( diff --git a/QEfficient/utils/torch_patches.py b/QEfficient/utils/torch_patches.py index 7e69f7b24d..3fff6fdf38 100644 --- a/QEfficient/utils/torch_patches.py +++ b/QEfficient/utils/torch_patches.py @@ -13,7 +13,7 @@ from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") # Store original references before patching _original_setup_trace_module_map = onnx_utils._setup_trace_module_map diff --git a/scripts/perplexity_computation/calculate_perplexity.py b/scripts/perplexity_computation/calculate_perplexity.py index 04ce624ec6..d13f171f93 100644 --- a/scripts/perplexity_computation/calculate_perplexity.py +++ b/scripts/perplexity_computation/calculate_perplexity.py @@ -20,7 +20,7 @@ from QEfficient.generation.cloud_infer import QAICInferenceSession from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") # 1. Data Loading diff --git a/tests/conftest.py b/tests/conftest.py index 5efd73d1f8..4f6eb90d0c 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -13,7 +13,7 @@ from QEfficient.utils.constants import QEFF_MODELS_DIR from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") def qeff_models_clean_up(): diff --git a/tests/transformers/test_transformer_pytorch_transforms.py b/tests/transformers/test_transformer_pytorch_transforms.py index 279d49bdf4..7bc890ea7d 100644 --- a/tests/transformers/test_transformer_pytorch_transforms.py +++ b/tests/transformers/test_transformer_pytorch_transforms.py @@ -22,7 +22,7 @@ from QEfficient.utils._utils import get_padding_shape_from_config from QEfficient.utils.logging_utils import QEFFLogger -logger = QEFFLogger.get_logger("INFRA", loglevel="INFO") +logger = QEFFLogger.get_logger("INFRA") KVCacheTransformTestConfigs = [ ("llama", 3, 32, 128, {"num_key_value_heads": 8, "intermediate_size": 512}, 0.8), From 5de01b460314fe4eb80dead63ee85c58ef1d6ed2 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Tue, 17 Mar 2026 15:58:34 +0000 Subject: [PATCH 09/18] Made minnor fix Signed-off-by: Abhishek Kumar Singh --- QEfficient/utils/logging_utils.py | 8 ++++---- scripts/Jenkinsfile | 14 ++++++++++++++ 2 files changed, 18 insertions(+), 4 deletions(-) diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index e144971fb9..e332baaf73 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -104,13 +104,13 @@ def __init__(self, loglevel: Optional[str] = None, log_path: Optional[str] = Non # Resolve log path (arg > env > default dir + timestamp) env_path = os.environ.get(LoggerConfig.log_path_env) self.log_path = log_path or env_path + timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") if not self.log_path: os.makedirs(LoggerConfig.default_log_dir, exist_ok=True) - timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") self.log_path = os.path.join(LoggerConfig.default_log_dir, f"QEFF_{timestamp}.log") else: - os.makedirs(os.path.dirname(os.path.abspath(self.log_path)), exist_ok=True) - + os.makedirs(self.log_path, exist_ok=True) + self.log_path = os.path.join(self.log_path, f"QEFF_{timestamp}.log") # Initialize the base logger and start background thread self.logger = self._initialize_logger() QEFFLogger._instance = self.logger @@ -223,7 +223,7 @@ def print_table(cls) -> None: """ path = cls._logfile if not path: - raise FileNotFoundError("Log file path is not set (cls._logfile is None).") + raise FileNotFoundError(f"Log file path is not set ({cls._logfile} is None).") if not os.path.exists(path): raise FileNotFoundError(f"Log file does not exist: {path}") diff --git a/scripts/Jenkinsfile b/scripts/Jenkinsfile index b791f3a318..0d200d523d 100644 --- a/scripts/Jenkinsfile +++ b/scripts/Jenkinsfile @@ -39,6 +39,8 @@ pipeline { cd /efficient-transformers && . preflight_qeff/bin/activate && mkdir -p $PWD/Non_cli_qaic && + mkdir -p $PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Qeff_logs && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/Non_cli_qaic && pytest tests -m '(not cli) and (not on_qaic) and (not finetune)' --ignore tests/vllm --ignore tests/transformers/models/image_text_to_text -n 4 --junitxml=tests/tests_log1.xml --durations=10 && @@ -56,6 +58,8 @@ pipeline { cd /efficient-transformers && . preflight_qeff/bin/activate && mkdir -p $PWD/Non_qaic_llm && + mkdir -p $PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Qeff_logs && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/Non_qaic_llm && pytest tests -m '(not cli) and (on_qaic) and (llm_model) and (not nightly) and (not multimodal) and (not qnn) and (not finetune) and (not diffusion_models)' --ignore tests/vllm --junitxml=tests/tests_log2.xml --durations=10 && @@ -73,6 +77,8 @@ pipeline { cd /efficient-transformers && . preflight_qeff/bin/activate && mkdir -p $PWD/Non_qaic_feature && + mkdir -p $PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Qeff_logs && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/Non_qaic_feature && pytest tests -m '(not cli) and (on_qaic) and (feature) and (not nightly) and (not multimodal) and (not qnn) and (not finetune) and (not diffusion_models)' --ignore tests/vllm --junitxml=tests/tests_log2_feature.xml --durations=10 && @@ -92,6 +98,8 @@ pipeline { cd /efficient-transformers && . preflight_qeff/bin/activate && mkdir -p $PWD/Non_cli_qaic_multimodal && + mkdir -p $PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Qeff_logs && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/Non_cli_qaic_multimodal && pytest tests -m '(not cli) and (on_qaic) and (multimodal) and (not qnn) and (not finetune) and (not diffusion_models)' --ignore tests/vllm --junitxml=tests/tests_log6.xml --durations=10 && @@ -109,6 +117,8 @@ pipeline { cd /efficient-transformers && . preflight_qeff/bin/activate && mkdir -p $PWD/Non_cli_qaic_diffusion && + mkdir -p $PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Qeff_logs && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/Non_cli_qaic_diffusion && export HF_HUB_CACHE=/huggingface_hub && @@ -129,6 +139,8 @@ pipeline { cd /efficient-transformers && . preflight_qeff/bin/activate && mkdir -p $PWD/cli && + mkdir -p $PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Qeff_logs && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/cli && pytest tests -m '(cli and not qnn) and (not finetune)' --ignore tests/vllm --junitxml=tests/tests_log3.xml --durations=10 && @@ -207,6 +219,8 @@ pipeline { # pip install /opt/qti-aic/integrations/torch_qaic/py310/torch_qaic-0.1.0-cp310-cp310-linux_x86_64.whl && pip install torch==2.9.0 torchvision==0.24.0 torchaudio==2.9.0 --index-url https://download.pytorch.org/whl/cpu && mkdir -p $PWD/cli_qaic_finetuning && + mkdir -p $PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Qeff_logs && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/cli_qaic_finetuning && pytest tests -m '(cli) and (on_qaic) and (not qnn) and (not multimodal) and (finetune)' --ignore tests/vllm --junitxml=tests/tests_log_finetune.xml --durations=10 && From 92149913d507f9c048d33721c247f0206a70ddc1 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Sun, 19 Apr 2026 08:50:42 +0000 Subject: [PATCH 10/18] pushed lint fix Signed-off-by: Abhishek Kumar Singh --- QEfficient/transformers/models/modeling_auto.py | 3 --- .../models/audio_models/test_audio_embedding_models.py | 3 --- .../models/audio_models/test_speech_seq2seq_models.py | 2 -- .../models/causal_lm_models/check_causal_models.py | 1 - .../models/causal_lm_models/test_causal_lm_blocking_hqkv.py | 6 ------ .../models/causal_lm_models/test_causal_lm_models.py | 4 ---- .../models/causal_lm_models/test_causal_lm_pl1.py | 6 ------ .../models/causal_lm_models/test_causal_tlm_models.py | 6 ------ .../models/causal_lm_models/test_fp16_causal_lm.py | 3 --- .../models/image_text_to_text/test_custom_dtype.py | 2 -- .../subfunction/test_causal_lm_blocking_subfunction.py | 3 --- tests/transformers/subfunction/test_subfunction_vlm.py | 4 ---- 12 files changed, 43 deletions(-) diff --git a/QEfficient/transformers/models/modeling_auto.py b/QEfficient/transformers/models/modeling_auto.py index 0e634ed88d..17a81992cc 100644 --- a/QEfficient/transformers/models/modeling_auto.py +++ b/QEfficient/transformers/models/modeling_auto.py @@ -77,9 +77,7 @@ from QEfficient.utils.logging_utils import QEFFLogger from QEfficient.utils.sampler_utils import get_sampling_inputs_and_outputs -<<<<<<< HEAD logger = QEFFLogger.get_logger("MODEL") -======= CUSTOM_IO_DTYPE_MAP = { torch.float16: "float16", torch.bfloat16: "bfloat16", @@ -92,7 +90,6 @@ torch.bfloat16: np.float16, # Since numpy doesn't support bfloat16 torch.float32: np.float32, } ->>>>>>> main class QEFFTransformersBase(QEFFBaseModel): diff --git a/tests/transformers/models/audio_models/test_audio_embedding_models.py b/tests/transformers/models/audio_models/test_audio_embedding_models.py index 64dc06a595..82c613e557 100644 --- a/tests/transformers/models/audio_models/test_audio_embedding_models.py +++ b/tests/transformers/models/audio_models/test_audio_embedding_models.py @@ -139,7 +139,6 @@ def check_ctc_pytorch_vs_kv_vs_ort_vs_ai100( qnn_config: Optional[str] = None, compare_results: Optional[bool] = False, ): - replace_transformers_quantizers() model_config = {"model_name": model_name} model_config["n_layer"] = n_layer @@ -200,7 +199,6 @@ def check_ctc_pytorch_vs_kv_vs_ort_vs_ai100( @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models) def test_full_ctc_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - torch.manual_seed(42) check_ctc_pytorch_vs_kv_vs_ort_vs_ai100( model_name=model_name, compare_results=True, manual_cleanup=manual_cleanup, num_devices=4 @@ -211,7 +209,6 @@ def test_full_ctc_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models) def test_few_ctc_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - torch.manual_seed(42) check_ctc_pytorch_vs_kv_vs_ort_vs_ai100(model_name=model_name, n_layer=4, manual_cleanup=manual_cleanup) diff --git a/tests/transformers/models/audio_models/test_speech_seq2seq_models.py b/tests/transformers/models/audio_models/test_speech_seq2seq_models.py index 6509d02fe7..0c6fb29087 100644 --- a/tests/transformers/models/audio_models/test_speech_seq2seq_models.py +++ b/tests/transformers/models/audio_models/test_speech_seq2seq_models.py @@ -374,7 +374,6 @@ def check_seq2seq_pytorch_vs_kv_vs_ort_vs_ai100( @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models) def test_full_seq2seq_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - torch.manual_seed(42) check_seq2seq_pytorch_vs_kv_vs_ort_vs_ai100( model_name=model_name, compare_results=True, manual_cleanup=manual_cleanup, num_devices=4 @@ -385,7 +384,6 @@ def test_full_seq2seq_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models) def test_few_seq2seq_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - torch.manual_seed(42) check_seq2seq_pytorch_vs_kv_vs_ort_vs_ai100(model_name=model_name, n_layer=4, manual_cleanup=manual_cleanup) diff --git a/tests/transformers/models/causal_lm_models/check_causal_models.py b/tests/transformers/models/causal_lm_models/check_causal_models.py index cc2d074a08..f878acbe73 100644 --- a/tests/transformers/models/causal_lm_models/check_causal_models.py +++ b/tests/transformers/models/causal_lm_models/check_causal_models.py @@ -57,7 +57,6 @@ def check_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100( retain_full_kv: Optional[bool] = None, compare_results: bool = False, ): - torch.manual_seed(42) replace_transformers_quantizers() model_hf = load_hf_causal_lm_model(model_name, num_hidden_layers=n_layer, config=config) diff --git a/tests/transformers/models/causal_lm_models/test_causal_lm_blocking_hqkv.py b/tests/transformers/models/causal_lm_models/test_causal_lm_blocking_hqkv.py index 4bf067e7c4..0568939cd2 100644 --- a/tests/transformers/models/causal_lm_models/test_causal_lm_blocking_hqkv.py +++ b/tests/transformers/models/causal_lm_models/test_causal_lm_blocking_hqkv.py @@ -31,7 +31,6 @@ @pytest.mark.on_qaic @pytest.mark.parametrize("model_name", test_models_blockedKV[:1]) def test_full_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - HEAD_BLOCK_SIZE = 8 NUM_KV_BLOCKS = 2 NUM_Q_BLOCKS = 2 @@ -77,7 +76,6 @@ def test_full_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manu @pytest.mark.on_qaic @pytest.mark.parametrize("model_name", test_models_blockedKV[:1]) def test_few_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - HEAD_BLOCK_SIZE = 8 NUM_KV_BLOCKS = 2 NUM_Q_BLOCKS = 2 @@ -123,7 +121,6 @@ def test_few_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manua @pytest.mark.on_qaic @pytest.mark.parametrize("model_name", test_models_blockedKV[:1]) def test_dummy_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - HEAD_BLOCK_SIZE = 8 NUM_KV_BLOCKS = 2 NUM_Q_BLOCKS = 2 @@ -178,7 +175,6 @@ def test_dummy_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100(model_name, man @pytest.mark.on_qaic @pytest.mark.parametrize("model_name", test_models_blockedKV[:1]) def test_full_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_cleanup): - HEAD_BLOCK_SIZE = 8 NUM_KV_BLOCKS = 2 NUM_Q_BLOCKS = 2 @@ -244,7 +240,6 @@ def test_full_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, m @pytest.mark.on_qaic @pytest.mark.parametrize("model_name", test_models_blockedKV[:1]) def test_few_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_cleanup): - HEAD_BLOCK_SIZE = 8 NUM_KV_BLOCKS = 2 NUM_Q_BLOCKS = 2 @@ -310,7 +305,6 @@ def test_few_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, ma @pytest.mark.on_qaic @pytest.mark.parametrize("model_name", test_models_blockedKV[:1]) def test_dummy_causal_all_blocking_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_cleanup): - HEAD_BLOCK_SIZE = 8 NUM_KV_BLOCKS = 2 NUM_Q_BLOCKS = 2 diff --git a/tests/transformers/models/causal_lm_models/test_causal_lm_models.py b/tests/transformers/models/causal_lm_models/test_causal_lm_models.py index 8dbb0915b8..8c61cdc98d 100644 --- a/tests/transformers/models/causal_lm_models/test_causal_lm_models.py +++ b/tests/transformers/models/causal_lm_models/test_causal_lm_models.py @@ -33,7 +33,6 @@ @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_causal) def test_full_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - if model_name in ModelConfig.FULL_MODEL_TESTS_TO_SKIP: pytest.skip(f"Skipping full model test for {model_name} due to resource constraints.") check_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100( @@ -55,7 +54,6 @@ def test_few_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup) @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_causal) def test_dummy_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - custom_config = model_config_dict[model_name] hf_config = AutoConfig.from_pretrained( model_name, @@ -89,7 +87,6 @@ def test_full_causal_lm_pytorch_vs_ort_vs_ai100_cb(model_name, manual_cleanup): @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_causal) def test_few_causal_lm_pytorch_vs_ort_vs_ai100_cb(model_name, manual_cleanup): - n_layer = get_custom_n_layers(model_name) check_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100( model_name=model_name, @@ -104,7 +101,6 @@ def test_few_causal_lm_pytorch_vs_ort_vs_ai100_cb(model_name, manual_cleanup): @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_causal) def test_dummy_causal_lm_pytorch_vs_ort_vs_ai100_cb(model_name, manual_cleanup): - custom_config = model_config_dict[model_name] hf_config = AutoConfig.from_pretrained( model_name, diff --git a/tests/transformers/models/causal_lm_models/test_causal_lm_pl1.py b/tests/transformers/models/causal_lm_models/test_causal_lm_pl1.py index b6641d7951..f5f2384e67 100644 --- a/tests/transformers/models/causal_lm_models/test_causal_lm_pl1.py +++ b/tests/transformers/models/causal_lm_models/test_causal_lm_pl1.py @@ -32,7 +32,6 @@ @pytest.mark.parametrize("model_name", test_models_pl1) @pytest.mark.parametrize("retain_full_kv", [True, False]) def test_full_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1(model_name, retain_full_kv, manual_cleanup): - if model_name == "gpt2" and retain_full_kv: pytest.skip("Skipping test for gpt2 with retain_full_kv=True as it is not supported.") @@ -52,7 +51,6 @@ def test_full_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1(model_name, retain_ful @pytest.mark.parametrize("model_name", test_models_pl1) @pytest.mark.parametrize("retain_full_kv", [True, False]) def test_few_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1(model_name, retain_full_kv, manual_cleanup): - if model_name == "gpt2" and retain_full_kv: pytest.skip("Skipping test for gpt2 with retain_full_kv=True as it is not supported.") torch.manual_seed(42) @@ -71,7 +69,6 @@ def test_few_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1(model_name, retain_full @pytest.mark.parametrize("model_name", test_models_pl1) @pytest.mark.parametrize("retain_full_kv", [True, False]) def test_dummy_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1(model_name, retain_full_kv, manual_cleanup): - if model_name == "gpt2" and retain_full_kv: pytest.skip("Skipping test for gpt2 with retain_full_kv=True as it is not supported.") @@ -97,7 +94,6 @@ def test_dummy_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1(model_name, retain_fu @pytest.mark.parametrize("model_name", test_models_pl1) @pytest.mark.parametrize("retain_full_kv", [True, False]) def test_full_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1_CB(model_name, retain_full_kv, manual_cleanup): - if model_name == "gpt2" and retain_full_kv: pytest.skip("Skipping test for gpt2 with retain_full_kv=True as it is not supported.") torch.manual_seed(42) @@ -117,7 +113,6 @@ def test_full_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1_CB(model_name, retain_ @pytest.mark.parametrize("model_name", test_models_pl1) @pytest.mark.parametrize("retain_full_kv", [True, False]) def test_few_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1_CB(model_name, retain_full_kv, manual_cleanup): - if model_name == "gpt2" and retain_full_kv: pytest.skip("Skipping test for gpt2 with retain_full_kv=True as it is not supported.") torch.manual_seed(42) @@ -137,7 +132,6 @@ def test_few_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1_CB(model_name, retain_f @pytest.mark.parametrize("model_name", test_models_pl1) @pytest.mark.parametrize("retain_full_kv", [True, False]) def test_dummy_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100_pl1_CB(model_name, retain_full_kv, manual_cleanup): - if model_name == "gpt2" and retain_full_kv: pytest.skip("Skipping test for gpt2 with retain_full_kv=True as it is not supported.") diff --git a/tests/transformers/models/causal_lm_models/test_causal_tlm_models.py b/tests/transformers/models/causal_lm_models/test_causal_tlm_models.py index 0b488a5037..9d02acbd29 100644 --- a/tests/transformers/models/causal_lm_models/test_causal_tlm_models.py +++ b/tests/transformers/models/causal_lm_models/test_causal_tlm_models.py @@ -32,7 +32,6 @@ @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_spd) def test_full_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - check_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100( model_name=model_name, num_speculative_tokens=Constants.NUM_SPECULATIVE_TOKENS, @@ -46,7 +45,6 @@ def test_full_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanu @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_spd) def test_few_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - n_layer = get_custom_n_layers(model_name) check_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100( model_name=model_name, @@ -61,7 +59,6 @@ def test_few_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_spd) def test_dummy_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_cleanup): - custom_config = model_config_dict[model_name] hf_config = AutoConfig.from_pretrained( model_name, @@ -81,7 +78,6 @@ def test_dummy_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100(model_name, manual_clean @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_spd) def test_full_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_cleanup): - check_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100( model_name=model_name, num_speculative_tokens=Constants.NUM_SPECULATIVE_TOKENS, @@ -96,7 +92,6 @@ def test_full_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_cle @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_spd) def test_few_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_cleanup): - n_layer = get_custom_n_layers(model_name) check_causal_lm_pytorch_vs_kv_vs_ort_vs_ai100( model_name=model_name, @@ -112,7 +107,6 @@ def test_few_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_clea @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models_spd) def test_dummy_causal_tlm_pytorch_vs_kv_vs_ort_vs_ai100_CB(model_name, manual_cleanup): - custom_config = model_config_dict[model_name] hf_config = AutoConfig.from_pretrained( model_name, diff --git a/tests/transformers/models/causal_lm_models/test_fp16_causal_lm.py b/tests/transformers/models/causal_lm_models/test_fp16_causal_lm.py index 2ff366ece2..af8c3b70f0 100644 --- a/tests/transformers/models/causal_lm_models/test_fp16_causal_lm.py +++ b/tests/transformers/models/causal_lm_models/test_fp16_causal_lm.py @@ -127,7 +127,6 @@ def check_causal_lm_pytorch_vs_kv_vs_ai100( @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models) def test_full_fp16_causal_lm_pytorch_vs_kv_vs_ai100(model_name, manual_cleanup): - torch.manual_seed(42) check_causal_lm_pytorch_vs_kv_vs_ai100( model_name=model_name, torch_dtype=torch.float16, manual_cleanup=manual_cleanup @@ -139,7 +138,6 @@ def test_full_fp16_causal_lm_pytorch_vs_kv_vs_ai100(model_name, manual_cleanup): @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models) def test_few_fp16_causal_lm_pytorch_vs_kv_vs_ai100(model_name, manual_cleanup): - torch.manual_seed(42) n_layer = get_custom_n_layers(model_name) check_causal_lm_pytorch_vs_kv_vs_ai100( @@ -152,7 +150,6 @@ def test_few_fp16_causal_lm_pytorch_vs_kv_vs_ai100(model_name, manual_cleanup): @pytest.mark.llm_model @pytest.mark.parametrize("model_name", test_models) def test_dummy_fp16_causal_lm_pytorch_vs_kv_vs_ai100(model_name, manual_cleanup): - torch.manual_seed(42) custom_config = model_config_dict[model_name] hf_config = AutoConfig.from_pretrained( diff --git a/tests/transformers/models/image_text_to_text/test_custom_dtype.py b/tests/transformers/models/image_text_to_text/test_custom_dtype.py index 95f62f1ac9..f291c5d12c 100644 --- a/tests/transformers/models/image_text_to_text/test_custom_dtype.py +++ b/tests/transformers/models/image_text_to_text/test_custom_dtype.py @@ -41,7 +41,6 @@ def test_full_image_text_to_text_pytorch_vs_kv_vs_ort_vs_ai100_custom_dtype( model_name, kv_offload, torch_dtype, manual_cleanup ): - if model_name in ModelConfig.SKIPPED_MODELS: pytest.skip("Test skipped for this model due to some issues.") if model_name in ModelConfig.DUAL_QPC_MODELS and not kv_offload: @@ -65,7 +64,6 @@ def test_full_image_text_to_text_pytorch_vs_kv_vs_ort_vs_ai100_custom_dtype( def test_few_image_text_to_text_pytorch_vs_kv_vs_ort_vs_ai100_custom_dtype( model_name, kv_offload, torch_dtype, manual_cleanup ): - if model_name in ModelConfig.SKIPPED_MODELS: pytest.skip("Test skipped for this model due to some issues.") if model_name in ModelConfig.DUAL_QPC_MODELS and not kv_offload: diff --git a/tests/transformers/subfunction/test_causal_lm_blocking_subfunction.py b/tests/transformers/subfunction/test_causal_lm_blocking_subfunction.py index 5c58508385..b3f42e1b0c 100644 --- a/tests/transformers/subfunction/test_causal_lm_blocking_subfunction.py +++ b/tests/transformers/subfunction/test_causal_lm_blocking_subfunction.py @@ -64,7 +64,6 @@ def check_blockedKV_onnx_function_count_with_subfunction( @pytest.mark.feature @pytest.mark.parametrize("model_name", test_models_blockedKV) def test_full_blockedKV_onnx_function_count_with_subfunction(model_name, manual_cleanup): - # Keep model small for test runtime, and avoid CB path (not needed for function count). check_blockedKV_onnx_function_count_with_subfunction(model_name, manual_cleanup=manual_cleanup) @@ -73,7 +72,6 @@ def test_full_blockedKV_onnx_function_count_with_subfunction(model_name, manual_ @pytest.mark.feature @pytest.mark.parametrize("model_name", test_models_blockedKV) def test_few_blockedKV_onnx_function_count_with_subfunction(model_name, manual_cleanup): - # Keep model small for test runtime, and avoid CB path (not needed for function count). n_layer = get_custom_n_layers(model_name) @@ -84,7 +82,6 @@ def test_few_blockedKV_onnx_function_count_with_subfunction(model_name, manual_c @pytest.mark.feature @pytest.mark.parametrize("model_name", test_models_blockedKV) def test_dummy_blockedKV_onnx_function_count_with_subfunction(model_name, manual_cleanup): - # Keep model small for test runtime, and avoid CB path (not needed for function count). hf_config = AutoConfig.from_pretrained( model_name, diff --git a/tests/transformers/subfunction/test_subfunction_vlm.py b/tests/transformers/subfunction/test_subfunction_vlm.py index baf690e638..39e2c6d0ac 100644 --- a/tests/transformers/subfunction/test_subfunction_vlm.py +++ b/tests/transformers/subfunction/test_subfunction_vlm.py @@ -50,7 +50,6 @@ def check_image_text_to_text_subfunction_core( num_devices: int = 1, config: Optional[AutoConfig] = None, ): - img_size = model_config_dict[model_name]["img_size"] img_url = model_config_dict[model_name]["img_url"] query = model_config_dict[model_name]["text_prompt"] @@ -117,7 +116,6 @@ def check_image_text_to_text_subfunction_core( @pytest.mark.parametrize("model_name", test_mm_models) @pytest.mark.parametrize("kv_offload", [True]) def test_full_image_text_to_text_subfunction(model_name, kv_offload, manual_cleanup): - torch.manual_seed(42) check_image_text_to_text_subfunction_core(model_name, kv_offload=kv_offload, manual_cleanup=manual_cleanup) @@ -127,7 +125,6 @@ def test_full_image_text_to_text_subfunction(model_name, kv_offload, manual_clea @pytest.mark.parametrize("model_name", test_mm_models) @pytest.mark.parametrize("kv_offload", [True]) def test_few_image_text_to_text_subfunction(model_name, kv_offload, manual_cleanup): - torch.manual_seed(42) check_image_text_to_text_subfunction_core( model_name, @@ -142,7 +139,6 @@ def test_few_image_text_to_text_subfunction(model_name, kv_offload, manual_clean @pytest.mark.parametrize("model_name", test_mm_models) @pytest.mark.parametrize("kv_offload", [True]) def test_dummy_image_text_to_text_subfunction(model_name, kv_offload, manual_cleanup): - torch.manual_seed(42) hf_config = AutoConfig.from_pretrained( model_name, trust_remote_code=True, **model_config_dict[model_name].get("additional_params", {}) From db84bd2538edb79f62ecd749ab46dadf02a05282 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Sun, 19 Apr 2026 09:15:36 +0000 Subject: [PATCH 11/18] Made minnor fix Signed-off-by: Abhishek Kumar Singh --- QEfficient/transformers/modeling_utils.py | 4 +++- 1 file changed, 3 insertions(+), 1 deletion(-) diff --git a/QEfficient/transformers/modeling_utils.py b/QEfficient/transformers/modeling_utils.py index a29d0e0966..2fb7bd24d4 100644 --- a/QEfficient/transformers/modeling_utils.py +++ b/QEfficient/transformers/modeling_utils.py @@ -92,7 +92,7 @@ from QEfficient.customop import CustomRMSNormAIC from QEfficient.proxy.pytorch_transform import QeffProxyModuleTransform from QEfficient.utils.constants import MIN_MASKED_ATTENTION_VALUE -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger if TYPE_CHECKING: from QEfficient.base.modeling_qeff import QEFFBaseModel @@ -164,6 +164,8 @@ QEffWhisperPositionalEmbedding, ) +logger = QEFFLogger.get_logger("MODEL") + # Define a named tuple for ModelArchitectures # Required for the Automation tool ModelArchitectures = namedtuple("ModelArchitectures", ["architectures"]) From 55eceb0434fe308069d1a8f91829b4f0a85802e2 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Sun, 19 Apr 2026 09:28:23 +0000 Subject: [PATCH 12/18] Added logger support to new models Signed-off-by: Abhishek Kumar Singh --- QEfficient/diffusers/pipelines/wan/pipeline_wan_i2v.py | 4 +++- QEfficient/transformers/models/qwen3_vl/modeling_qwen3_vl.py | 4 +++- .../transformers/models/qwen3_vl_moe/modeling_qwen3_vl_moe.py | 4 +++- 3 files changed, 9 insertions(+), 3 deletions(-) diff --git a/QEfficient/diffusers/pipelines/wan/pipeline_wan_i2v.py b/QEfficient/diffusers/pipelines/wan/pipeline_wan_i2v.py index 0c302ca4b2..c4a8d0ccfa 100644 --- a/QEfficient/diffusers/pipelines/wan/pipeline_wan_i2v.py +++ b/QEfficient/diffusers/pipelines/wan/pipeline_wan_i2v.py @@ -41,7 +41,9 @@ ) from QEfficient.generation.cloud_infer import QAICInferenceSession from QEfficient.utils import constants -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL") class QEffWanImageToVideoPipeline: diff --git a/QEfficient/transformers/models/qwen3_vl/modeling_qwen3_vl.py b/QEfficient/transformers/models/qwen3_vl/modeling_qwen3_vl.py index 6d6c6b42d6..1881ed2de4 100644 --- a/QEfficient/transformers/models/qwen3_vl/modeling_qwen3_vl.py +++ b/QEfficient/transformers/models/qwen3_vl/modeling_qwen3_vl.py @@ -41,7 +41,9 @@ from QEfficient.utils import constants from QEfficient.utils._utils import IOInfo, get_padding_shape_from_config from QEfficient.utils.constants import MIN_MASKED_ATTENTION_VALUE -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL") def qeff_apply_interleaved_mrope(freqs, mrope_section): diff --git a/QEfficient/transformers/models/qwen3_vl_moe/modeling_qwen3_vl_moe.py b/QEfficient/transformers/models/qwen3_vl_moe/modeling_qwen3_vl_moe.py index 114bcb3bdd..49c279d021 100644 --- a/QEfficient/transformers/models/qwen3_vl_moe/modeling_qwen3_vl_moe.py +++ b/QEfficient/transformers/models/qwen3_vl_moe/modeling_qwen3_vl_moe.py @@ -42,7 +42,9 @@ from QEfficient.utils import constants from QEfficient.utils._utils import IOInfo, get_padding_shape_from_config from QEfficient.utils.constants import MIN_MASKED_ATTENTION_VALUE -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL") def qeff_apply_interleaved_mrope(freqs, mrope_section): From 519f2aff4f8f3d8bb51c4439ff12c888e888139e Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Sun, 19 Apr 2026 20:10:53 +0530 Subject: [PATCH 13/18] Update logging_utils.py Signed-off-by: Abhishek Kumar Singh --- QEfficient/utils/logging_utils.py | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index e332baaf73..8858aca13f 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -280,7 +280,17 @@ def classify(msg: str) -> Optional[str]: last_ts = ts if t_start is None: - raise ValueError("Missing required milestone: 'Initiating the model weight loading.'") + logging.warning( + "Missing required milestone: 'Initiating the model weight loading.' " + "Defaulting t_start to first available timestamp (0.0 baseline)." + ) + + if times: + # Use earliest recorded milestone as baseline + t_start = min(times.values()) + else: + # Absolute fallback: zero baseline + t_start = datetime.min t2 = times.get("LOAD_DONE", t_start) # end of loading t3 = times.get("ONNX_SAVED") or t2 # export end From 9b6711832cb942509807c46ad8f1b8fe97a6e1e2 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Wed, 29 Apr 2026 15:01:36 +0530 Subject: [PATCH 14/18] Removed qeff logger from fine tuning Signed-off-by: Abhishek Kumar Singh --- QEfficient/finetune/dataset/alpaca_dataset.py | 4 +--- QEfficient/finetune/utils/config_utils.py | 4 +--- QEfficient/finetune/utils/plot_metrics.py | 4 +--- run.py | 13 +++++++++++++ scripts/Jenkinsfile | 3 +-- 5 files changed, 17 insertions(+), 11 deletions(-) create mode 100644 run.py diff --git a/QEfficient/finetune/dataset/alpaca_dataset.py b/QEfficient/finetune/dataset/alpaca_dataset.py index f790efa368..5d24819e0b 100644 --- a/QEfficient/finetune/dataset/alpaca_dataset.py +++ b/QEfficient/finetune/dataset/alpaca_dataset.py @@ -11,9 +11,7 @@ import torch from torch.utils.data import Dataset -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT") +from QEfficient.finetune.utils.logging_utils import logger PROMPT_DICT = { "prompt_input": ( diff --git a/QEfficient/finetune/utils/config_utils.py b/QEfficient/finetune/utils/config_utils.py index 9757ba5038..0c8b3d8275 100644 --- a/QEfficient/finetune/utils/config_utils.py +++ b/QEfficient/finetune/utils/config_utils.py @@ -20,9 +20,7 @@ from QEfficient.finetune.configs.training import TrainConfig from QEfficient.finetune.dataset.dataset_config import DATASET_PREPROC from QEfficient.finetune.utils.helper import Peft_Method -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT") +from QEfficient.finetune.utils.logging_utils import logger def update_config(config, **kwargs): diff --git a/QEfficient/finetune/utils/plot_metrics.py b/QEfficient/finetune/utils/plot_metrics.py index 545894244c..1e22bc6a83 100644 --- a/QEfficient/finetune/utils/plot_metrics.py +++ b/QEfficient/finetune/utils/plot_metrics.py @@ -11,9 +11,7 @@ import matplotlib.pyplot as plt -from QEfficient.utils.logging_utils import QEFFLogger - -logger = QEFFLogger.get_logger("FT") +from QEfficient.finetune.utils.logging_utils import logger def plot_metric(data, metric_name, x_label, y_label, title, colors): diff --git a/run.py b/run.py new file mode 100644 index 0000000000..028124c7cd --- /dev/null +++ b/run.py @@ -0,0 +1,13 @@ +from transformers import AutoTokenizer + +from QEfficient import QEFFAutoModelForCausalLM + +model_name = "meta-llama/Llama-3.2-1B" +model1 = QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers=2) +# model2=QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers = 2) +# with_sub_func_onnx = model1.export(use_onnx_subfunctions=True) +model1.compile(num_devices=1, num_cores=16, use_onnx_subfunctions=True) +hash_0_1 = model1.export_hash +inputs = "Help me with this" +tokenizer = AutoTokenizer.from_pretrained(model_name) +generation_00 = model1.generate(prompts=["Help me with this"], tokenizer=tokenizer) diff --git a/scripts/Jenkinsfile b/scripts/Jenkinsfile index 1bde75df8f..50af13cffd 100644 --- a/scripts/Jenkinsfile +++ b/scripts/Jenkinsfile @@ -101,8 +101,7 @@ pipeline { cd /efficient-transformers && . preflight_qeff/bin/activate && mkdir -p $PWD/Non_qaic_llm && - mkdir -p $PWD/Qeff_logs && - export QEFF_LOG_PATH=$PWD/Qeff_logs && + export QEFF_LOG_PATH=$PWD/Non_qaic_llm && export TOKENIZERS_PARALLELISM=false && export QEFF_HOME=$PWD/Non_qaic_llm && pytest tests -m '(llm_model) and (not qnn) and ${TEST_FILTER}' --ignore tests/vllm --ignore tests/unit_test --junitxml=tests/tests_log2.xml --durations=10 && From 7d2e4021262f93519e4c3f2011d93ae157cf7af3 Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Wed, 29 Apr 2026 15:01:58 +0530 Subject: [PATCH 15/18] Removed qeff logger from fine tuning Signed-off-by: Abhishek Kumar Singh --- run.py | 13 ------------- 1 file changed, 13 deletions(-) delete mode 100644 run.py diff --git a/run.py b/run.py deleted file mode 100644 index 028124c7cd..0000000000 --- a/run.py +++ /dev/null @@ -1,13 +0,0 @@ -from transformers import AutoTokenizer - -from QEfficient import QEFFAutoModelForCausalLM - -model_name = "meta-llama/Llama-3.2-1B" -model1 = QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers=2) -# model2=QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers = 2) -# with_sub_func_onnx = model1.export(use_onnx_subfunctions=True) -model1.compile(num_devices=1, num_cores=16, use_onnx_subfunctions=True) -hash_0_1 = model1.export_hash -inputs = "Help me with this" -tokenizer = AutoTokenizer.from_pretrained(model_name) -generation_00 = model1.generate(prompts=["Help me with this"], tokenizer=tokenizer) From 6e79a460575b6d6a5cb7a256bee39b3088f9fe1f Mon Sep 17 00:00:00 2001 From: Abhishek Kumar Singh Date: Mon, 4 May 2026 12:23:09 +0530 Subject: [PATCH 16/18] Added few changes Signed-off-by: Abhishek Kumar Singh --- QEfficient/utils/logging_utils.py | 260 ++++++++++++++---------------- tests/utils/test_logger.py | 93 ++++++----- 2 files changed, 178 insertions(+), 175 deletions(-) diff --git a/QEfficient/utils/logging_utils.py b/QEfficient/utils/logging_utils.py index 8858aca13f..451d9a5449 100644 --- a/QEfficient/utils/logging_utils.py +++ b/QEfficient/utils/logging_utils.py @@ -8,11 +8,11 @@ import json import logging import os -import queue import threading from datetime import datetime from logging.handlers import RotatingFileHandler -from typing import Any, Dict, List, Optional +from pathlib import Path +from typing import Any, Dict, Iterable, List, Optional, Tuple from tabulate import tabulate @@ -27,6 +27,7 @@ class JSONNamespaceFormatter(logging.Formatter): def format(self, record): log_record = { + "created": record.created, "date": datetime.fromtimestamp(record.created).strftime("%Y-%m-%d"), "time": datetime.fromtimestamp(record.created).strftime("%H:%M:%S"), "level": record.levelname, @@ -38,38 +39,6 @@ def format(self, record): return json.dumps(log_record) -class QEFFLoggerThread(threading.Thread): - """ - Custom formatter to output log records in JSON format with metadata. - - Methods: - format(record): Formats a log record into a JSON string. - - Parameters: - record (logging.LogRecord): The log record to format. - - Returns: - str: JSON-formatted log string. - """ - - def __init__(self, logger, log_queue): - super().__init__(daemon=True) - self.logger = logger - self.log_queue = log_queue - self.running = True - - def run(self): - while self.running: - try: - record = self.log_queue.get(timeout=1) - self.logger.handle(record) - except queue.Empty: - continue - - def stop(self): - self.running = False - - class QEFFLogger: """ Singleton logger class for structured logging with namespace support. @@ -81,8 +50,7 @@ class QEFFLogger: _instance: Optional[logging.Logger] = None _logfile: Optional[str] = None - _log_queue: queue.Queue = queue.Queue() - _logger_thread: Optional[QEFFLoggerThread] = None + _init_lock = threading.Lock() def __init__(self, loglevel: Optional[str] = None, log_path: Optional[str] = None): """ @@ -91,7 +59,10 @@ def __init__(self, loglevel: Optional[str] = None, log_path: Optional[str] = Non loglevel: kept for backward compatibility, but env `QEFF_LOG_LEVEL` takes precedence. log_path: optional path to the log file (highest priority). """ - if QEFFLogger._instance is None: + with QEFFLogger._init_lock: + if QEFFLogger._instance is not None: + return + # Determine effective log level: # Priority: ENV(QEFF_LOG_LEVEL) -> arg(loglevel) -> LoggerConfig.default_level env_level = os.environ.get(LoggerConfig.log_level_env) @@ -103,19 +74,28 @@ def __init__(self, loglevel: Optional[str] = None, log_path: Optional[str] = Non # Resolve log path (arg > env > default dir + timestamp) env_path = os.environ.get(LoggerConfig.log_path_env) - self.log_path = log_path or env_path - timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") - if not self.log_path: - os.makedirs(LoggerConfig.default_log_dir, exist_ok=True) - self.log_path = os.path.join(LoggerConfig.default_log_dir, f"QEFF_{timestamp}.log") - else: - os.makedirs(self.log_path, exist_ok=True) - self.log_path = os.path.join(self.log_path, f"QEFF_{timestamp}.log") - # Initialize the base logger and start background thread + self.log_path = self._resolve_log_path(log_path or env_path) + + # Initialize the base logger self.logger = self._initialize_logger() QEFFLogger._instance = self.logger - QEFFLogger._logger_thread = QEFFLoggerThread(self.logger, QEFFLogger._log_queue) - QEFFLogger._logger_thread.start() + + @classmethod + def _resolve_log_path(cls, requested_path: Optional[str]) -> str: + """Resolve the final log file path from a user path or defaults.""" + timestamp = datetime.now().strftime("%Y%m%d_%H%M%S") + default_file = os.path.join(LoggerConfig.default_log_dir, f"QEFF_{timestamp}.log") + if not requested_path: + os.makedirs(LoggerConfig.default_log_dir, exist_ok=True) + return default_file + + path = Path(requested_path).expanduser() + if path.suffix.lower() == ".log": + path.parent.mkdir(parents=True, exist_ok=True) + return str(path) + + path.mkdir(parents=True, exist_ok=True) + return str(path / f"QEFF_{timestamp}.log") def _initialize_logger(self) -> logging.Logger: """ @@ -125,16 +105,20 @@ def _initialize_logger(self) -> logging.Logger: logger = logging.getLogger("QEFF_LOGGER") logger.setLevel(getattr(logging, self.loglevel)) + logger.propagate = False # Avoid duplicate handlers if reinitialized in same process - if not logger.handlers: - handler = RotatingFileHandler( - self.log_path, - maxBytes=LoggerConfig.max_bytes, - backupCount=LoggerConfig.backup_count, - ) - handler.setFormatter(JSONNamespaceFormatter()) - logger.addHandler(handler) + for handler in logger.handlers[:]: + handler.close() + logger.removeHandler(handler) + + handler = RotatingFileHandler( + self.log_path, + maxBytes=LoggerConfig.max_bytes, + backupCount=LoggerConfig.backup_count, + ) + handler.setFormatter(JSONNamespaceFormatter()) + logger.addHandler(handler) return logger @@ -162,18 +146,8 @@ def log(cls, level: str, namespace: str, msg: str, fn: str = "", lno: int = 0, f if not isinstance(level_num, int): raise ValueError(f"Invalid log level: {level}") - record = cls._instance.makeRecord( - name="QEFF_LOGGER", - level=level_num, - fn=fn, - lno=lno, - msg=msg, - args=(), - exc_info=None, - func=func, - extra={"namespace": namespace}, - ) - cls._log_queue.put(record) + logger = logging.LoggerAdapter(cls._instance, {"namespace": namespace}) + logger.log(level_num, msg, stacklevel=2) @classmethod def set_loglevel(cls, loglevel: Optional[str] = None): @@ -196,16 +170,12 @@ def set_loglevel(cls, loglevel: Optional[str] = None): @classmethod def close_logger(cls): """ - Gracefully shut down the logger and its thread. + Gracefully shut down the logger. """ - if cls._logger_thread: - cls._logger_thread.stop() - cls._logger_thread.join() - cls._logger_thread = None - if cls._instance: handlers = cls._instance.handlers[:] for handler in handlers: + handler.flush() handler.close() cls._instance.removeHandler(handler) cls._instance = None @@ -217,85 +187,98 @@ def _parse_dt(cls, date_str: str, time_str: str) -> datetime: return datetime.strptime(f"{date_str} {time_str}", "%Y-%m-%d %H:%M:%S") @classmethod - def print_table(cls) -> None: - """ - Parse the line-delimited JSON log in cls._logfile and print timing table with t1 as baseline (0.0s). - """ - path = cls._logfile - if not path: - raise FileNotFoundError(f"Log file path is not set ({cls._logfile} is None).") - if not os.path.exists(path): - raise FileNotFoundError(f"Log file does not exist: {path}") - - SUBSTR_TO_KEY: Dict[str, str] = { - "initiating the model weight loading.": "START_LOAD", - "pytorch transforms applied to model": "LOAD_DONE", - "transformed onnx saved": "ONNX_SAVED", - "model compilation is finished and saved": "COMPILE_DONE", - "text generated finised": "TEXT_DONE", - "specialization_file_path": "TEXT_READY", - } + def get_logfile_path(cls) -> Optional[str]: + """Return active log file path, if logger is initialized.""" + return cls._logfile - def classify(msg: str) -> Optional[str]: - m = msg.lower() - for needle, key in SUBSTR_TO_KEY.items(): - if needle in m: - return key - return None - - from datetime import timedelta - - t_start: Optional[datetime] = None - last_ts: Optional[datetime] = None - times: Dict[str, datetime] = {} - - with open(path, "r", encoding="utf-8") as f: - for raw in f: + @classmethod + def _iter_log_records(cls, path: str) -> Iterable[Dict[str, Any]]: + with open(path, "r", encoding="utf-8") as handle: + for raw in handle: line = raw.strip() if not line: continue try: - rec: Dict[str, Any] = json.loads(line) + record = json.loads(line) except json.JSONDecodeError: continue + if isinstance(record, dict): + yield record - date_str = rec.get("date") - time_str = rec.get("time") - msg = rec.get("message", "") - if not date_str or not time_str: - continue - - ts = cls._parse_dt(date_str, time_str) + @classmethod + def _get_record_timestamp(cls, record: Dict[str, Any]) -> Optional[datetime]: + created = record.get("created") + if isinstance(created, (float, int)): + return datetime.fromtimestamp(float(created)) + + date_str = record.get("date") + time_str = record.get("time") + if not date_str or not time_str: + return None + try: + return cls._parse_dt(str(date_str), str(time_str)) + except ValueError: + return None - # Enforce strictly increasing timestamps - if last_ts is not None and ts <= last_ts: - ts = last_ts + timedelta(milliseconds=1) + @classmethod + def _extract_milestone_times(cls, path: str) -> Dict[str, datetime]: + """ + Extract first occurrence timestamp for each milestone key from JSON log lines. + """ + milestone_patterns: Dict[str, Tuple[str, ...]] = { + "START_LOAD": ("initiating the model weight loading",), + "LOAD_DONE": ("pytorch transforms applied to model",), + "ONNX_SAVED": ("model export is finished and saved", "transformed onnx saved"), + "COMPILE_DONE": ("model compilation is finished and saved",), + "TEXT_DONE": ("text generation finished", "text generated finised"), + "TEXT_READY": ("specialization_file_path",), + } - key = classify(msg) - if key and key not in times: - times[key] = ts - if key == "START_LOAD" and t_start is None: - t_start = ts + times: Dict[str, datetime] = {} + for record in cls._iter_log_records(path): + message = str(record.get("message", "")).lower() + timestamp = cls._get_record_timestamp(record) + if not timestamp: + continue - last_ts = ts + for key, patterns in milestone_patterns.items(): + if key in times: + continue + if any(pattern in message for pattern in patterns): + times[key] = timestamp + break + return times - if t_start is None: - logging.warning( - "Missing required milestone: 'Initiating the model weight loading.' " - "Defaulting t_start to first available timestamp (0.0 baseline)." - ) + @classmethod + def print_table(cls) -> bool: + """ + Parse the line-delimited JSON log in cls._logfile and print timing table with t1 as baseline (0.0s). + """ + path = cls._logfile + if not path: + return False + if not os.path.exists(path): + return False - if times: - # Use earliest recorded milestone as baseline - t_start = min(times.values()) - else: - # Absolute fallback: zero baseline - t_start = datetime.min + times = cls._extract_milestone_times(path) + if not times: + return False + t_start = times.get("START_LOAD", min(times.values())) t2 = times.get("LOAD_DONE", t_start) # end of loading - t3 = times.get("ONNX_SAVED") or t2 # export end - t4 = times.get("COMPILE_DONE") or t3 # compile end - t5 = times.get("TEXT_DONE") or times.get("TEXT_READY") or t4 # text gen end + t3 = times.get("ONNX_SAVED", t2) # export end + t4 = times.get("COMPILE_DONE", t3) # compile end + t5 = times.get("TEXT_DONE", times.get("TEXT_READY", t4)) # text gen end + + # Keep boundaries monotonic for stable table output. + if t2 < t_start: + t2 = t_start + if t3 < t2: + t3 = t2 + if t4 < t3: + t4 = t3 + if t5 < t4: + t5 = t4 def offset_seconds(t: datetime) -> float: return (t - t_start).total_seconds() @@ -315,3 +298,4 @@ def offset_seconds(t: datetime) -> float: ] print("\n") print(tabulate(timing_data, headers=["Step", "Time (s)"], tablefmt="github", floatfmt=".3f")) + return True diff --git a/tests/utils/test_logger.py b/tests/utils/test_logger.py index 20f45e7a23..6e27e44809 100644 --- a/tests/utils/test_logger.py +++ b/tests/utils/test_logger.py @@ -5,44 +5,63 @@ # # ----------------------------------------------------------------------------- -import threading +import json import time +import pytest + from QEfficient.utils.logging_utils import QEFFLogger -# ------------------------------- -# Define namespace once -# ------------------------------- -NAMESPACE = "model" - -# ------------------------------- -# Initialize logger -# ------------------------------- -logger = QEFFLogger.get_logger(NAMESPACE, "DEBUG") - - -# ------------------------------- -# Worker function for threads -# ------------------------------- -def log_worker(thread_id): - for i in range(5): - logger.info(f"Thread-{thread_id} logging message {i}") - time.sleep(0.1) - - -# ------------------------------- -# Create and start threads -# ------------------------------- -threads = [] -for t_id in range(3): - t = threading.Thread(target=log_worker, args=(t_id,)) - threads.append(t) - t.start() - -for t in threads: - t.join() - -# ------------------------------- -# Graceful shutdown -# ------------------------------- -# QEFFLogger.close_logger() + +@pytest.fixture(autouse=True) +def reset_logger_state(): + QEFFLogger.close_logger() + yield + QEFFLogger.close_logger() + # Keep process-global logger initialized for tests importing module-level adapters. + QEFFLogger.get_logger("INFRA") + + +def test_logger_writes_json_records(tmp_path): + logger = QEFFLogger.get_logger("model", "DEBUG", str(tmp_path)) + logger.info("hello logger") + logger.warning("warning logger") + + log_path = QEFFLogger.get_logfile_path() + assert log_path is not None + + QEFFLogger.close_logger() + with open(log_path, "r", encoding="utf-8") as handle: + rows = [json.loads(line) for line in handle if line.strip()] + + assert len(rows) >= 2 + assert rows[-1]["namespace"] == "model" + assert rows[-1]["level"] == "WARNING" + assert rows[-1]["message"] == "warning logger" + assert isinstance(rows[-1]["created"], float) + + +def test_print_table_from_logged_milestones(tmp_path, capsys): + logger = QEFFLogger.get_logger("infra", "INFO", str(tmp_path)) + logger.info("Initiating the model weight loading.") + time.sleep(0.01) + logger.info("Pytorch transforms applied to model: test") + time.sleep(0.01) + logger.info("Model export is finished and saved at: /tmp/model.onnx") + time.sleep(0.01) + logger.info("Model compilation is finished and saved at: /tmp/model.qpc") + time.sleep(0.01) + logger.info("Text generation finished") + + assert QEFFLogger.print_table() is True + output = capsys.readouterr().out + assert "Model Loading" in output + assert "Model Exporting" in output + assert "Model Compilation" in output + assert "Text Generation" in output + assert "Total Time" in output + + +def test_print_table_without_log_file_returns_false(): + QEFFLogger.close_logger() + assert QEFFLogger.print_table() is False From 6e5c843ef3378a78664e2fcbf165b0319b689b00 Mon Sep 17 00:00:00 2001 From: abhishek-singh591 Date: Mon, 4 May 2026 07:40:29 +0000 Subject: [PATCH 17/18] Lint fix Signed-off-by: abhishek-singh591 --- QEfficient/compile/compile_helper.py | 2 +- .../causallm/example_pytorch_transforms.py | 12 ++++++------ run.py | 13 +++++++++++++ 3 files changed, 20 insertions(+), 7 deletions(-) create mode 100644 run.py diff --git a/QEfficient/compile/compile_helper.py b/QEfficient/compile/compile_helper.py index 2b64e550d1..0abaa7ae9e 100644 --- a/QEfficient/compile/compile_helper.py +++ b/QEfficient/compile/compile_helper.py @@ -14,7 +14,7 @@ from QEfficient.compile.qnn_compiler import compile as qnn_compile from QEfficient.utils import constants -from QEfficient.utils._utils import load_json, load_yaml +from QEfficient.utils._utils import load_json, load_yaml, to_named_specializations from QEfficient.utils.logging_utils import QEFFLogger logger = QEFFLogger.get_logger("INFRA") diff --git a/examples/onboarding_guide/causallm/example_pytorch_transforms.py b/examples/onboarding_guide/causallm/example_pytorch_transforms.py index ff62588f9c..503efc12dc 100644 --- a/examples/onboarding_guide/causallm/example_pytorch_transforms.py +++ b/examples/onboarding_guide/causallm/example_pytorch_transforms.py @@ -27,12 +27,6 @@ from types import MethodType from typing import Callable, Optional, Tuple, Union -from QEfficient.transformers.models.blueprint.modeling_blueprint import ( - QEffBlueprintAttention, - QEffBlueprintDecoderLayer, - QEffBlueprintForCausalLM, - QEffBlueprintModel, -) from torch import nn # Example imports for three representative models @@ -62,6 +56,12 @@ from QEfficient.base.pytorch_transforms import ExternalModuleMapperTransform, ModuleMappingTransform from QEfficient.customop import CustomRMSNormAIC from QEfficient.transformers.embeddings.embedding_utils import POOLING_MAP, PooledModel, validate_user_pooling_function +from QEfficient.transformers.models.blueprint.modeling_blueprint import ( + QEffBlueprintAttention, + QEffBlueprintDecoderLayer, + QEffBlueprintForCausalLM, + QEffBlueprintModel, +) from QEfficient.transformers.models.llama.modeling_llama import ( QEffLlamaAttention, QEffLlamaDecoderLayer, diff --git a/run.py b/run.py new file mode 100644 index 0000000000..028124c7cd --- /dev/null +++ b/run.py @@ -0,0 +1,13 @@ +from transformers import AutoTokenizer + +from QEfficient import QEFFAutoModelForCausalLM + +model_name = "meta-llama/Llama-3.2-1B" +model1 = QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers=2) +# model2=QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers = 2) +# with_sub_func_onnx = model1.export(use_onnx_subfunctions=True) +model1.compile(num_devices=1, num_cores=16, use_onnx_subfunctions=True) +hash_0_1 = model1.export_hash +inputs = "Help me with this" +tokenizer = AutoTokenizer.from_pretrained(model_name) +generation_00 = model1.generate(prompts=["Help me with this"], tokenizer=tokenizer) From 63610287668763198f6f5619d43415d63a5b1337 Mon Sep 17 00:00:00 2001 From: abhishek-singh591 Date: Mon, 4 May 2026 07:51:07 +0000 Subject: [PATCH 18/18] Made minor fixes Signed-off-by: abhishek-singh591 --- .../transformers/models/pytorch_transforms.py | 4 +++- .../causallm/example_pytorch_transforms.py | 12 ++++++------ run.py | 13 ------------- 3 files changed, 9 insertions(+), 20 deletions(-) delete mode 100644 run.py diff --git a/QEfficient/transformers/models/pytorch_transforms.py b/QEfficient/transformers/models/pytorch_transforms.py index 5ff06e6443..de9443c5fb 100644 --- a/QEfficient/transformers/models/pytorch_transforms.py +++ b/QEfficient/transformers/models/pytorch_transforms.py @@ -509,7 +509,9 @@ from QEfficient.transformers.post_processing import build_and_attach_mlp, model_type_registry from QEfficient.transformers.sampler.sampler import sampler_forward from QEfficient.transformers.spd.spd_transform_forward import tlm_forward -from QEfficient.utils.logging_utils import logger +from QEfficient.utils.logging_utils import QEFFLogger + +logger = QEFFLogger.get_logger("MODEL") SPD_TARGET = "target" diff --git a/examples/onboarding_guide/causallm/example_pytorch_transforms.py b/examples/onboarding_guide/causallm/example_pytorch_transforms.py index 503efc12dc..ff62588f9c 100644 --- a/examples/onboarding_guide/causallm/example_pytorch_transforms.py +++ b/examples/onboarding_guide/causallm/example_pytorch_transforms.py @@ -27,6 +27,12 @@ from types import MethodType from typing import Callable, Optional, Tuple, Union +from QEfficient.transformers.models.blueprint.modeling_blueprint import ( + QEffBlueprintAttention, + QEffBlueprintDecoderLayer, + QEffBlueprintForCausalLM, + QEffBlueprintModel, +) from torch import nn # Example imports for three representative models @@ -56,12 +62,6 @@ from QEfficient.base.pytorch_transforms import ExternalModuleMapperTransform, ModuleMappingTransform from QEfficient.customop import CustomRMSNormAIC from QEfficient.transformers.embeddings.embedding_utils import POOLING_MAP, PooledModel, validate_user_pooling_function -from QEfficient.transformers.models.blueprint.modeling_blueprint import ( - QEffBlueprintAttention, - QEffBlueprintDecoderLayer, - QEffBlueprintForCausalLM, - QEffBlueprintModel, -) from QEfficient.transformers.models.llama.modeling_llama import ( QEffLlamaAttention, QEffLlamaDecoderLayer, diff --git a/run.py b/run.py deleted file mode 100644 index 028124c7cd..0000000000 --- a/run.py +++ /dev/null @@ -1,13 +0,0 @@ -from transformers import AutoTokenizer - -from QEfficient import QEFFAutoModelForCausalLM - -model_name = "meta-llama/Llama-3.2-1B" -model1 = QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers=2) -# model2=QEFFAutoModelForCausalLM.from_pretrained(model_name, num_hidden_layers = 2) -# with_sub_func_onnx = model1.export(use_onnx_subfunctions=True) -model1.compile(num_devices=1, num_cores=16, use_onnx_subfunctions=True) -hash_0_1 = model1.export_hash -inputs = "Help me with this" -tokenizer = AutoTokenizer.from_pretrained(model_name) -generation_00 = model1.generate(prompts=["Help me with this"], tokenizer=tokenizer)