diff --git a/llm_benchmark.py b/llm_benchmark.py new file mode 100644 index 0000000..84fd309 --- /dev/null +++ b/llm_benchmark.py @@ -0,0 +1,133 @@ +#!/usr/bin/env python3 +import time +import requests +import json +import argparse +import sys + +def benchmark_llm(url, model, api_key, prompt, max_tokens, temperature, show_output): + headers = { + "Content-Type": "application/json", + "Authorization": f"Bearer {api_key}" + } + + payload = { + "model": model, + "messages": [{"role": "user", "content": prompt}], + "max_tokens": max_tokens, + "temperature": temperature, + "stream": True, + # Asks the server to include exact token usage in the final stream chunk + "stream_options": {"include_usage": True} + } + + print(f"Benchmarking model: '{model}' at '{url}'...") + print(f"Prompt: {prompt[:50]}...\n") + print("-" * 50) + + start_time = time.perf_counter() + first_token_time = None + last_token_time = None + + content = "" + chunk_count = 0 + exact_completion_tokens = None + + try: + response = requests.post(url, headers=headers, json=payload, stream=True) + response.raise_for_status() + + for line in response.iter_lines(): + if line: + line = line.decode('utf-8') + if line.startswith("data: "): + data_str = line[6:] + + if data_str == "[DONE]": + break + + try: + data = json.loads(data_str) + + # Look for exact token usage reported by the server + if "usage" in data and data["usage"] is not None: + exact_completion_tokens = data["usage"].get("completion_tokens") + + choices = data.get("choices", []) + if choices: + delta = choices[0].get("delta", {}) + if "content" in delta: + chunk = delta["content"] + if chunk: + if first_token_time is None: + first_token_time = time.perf_counter() + + content += chunk + chunk_count += 1 + last_token_time = time.perf_counter() + + if show_output: + sys.stdout.write(chunk) + sys.stdout.flush() + + except json.JSONDecodeError: + continue + + except requests.exceptions.RequestException as e: + print(f"\nError during API request: {e}") + return + + if show_output: + print("\n") + print("-" * 50) + + if first_token_time is None: + print("Error: No tokens were generated. Check the model name and API key.") + return + + # Calculate Metrics + ttft = first_token_time - start_time + generation_time = last_token_time - first_token_time + total_time = last_token_time - start_time + + # Use the exact server token count if provided, otherwise assume 1 chunk = 1 token + final_token_count = exact_completion_tokens if exact_completion_tokens is not None else chunk_count + + # Tokens per second (excluding the first token's time since that's prompt processing) + if final_token_count > 1 and generation_time > 0: + tps = (final_token_count - 1) / generation_time + elif final_token_count == 1 and total_time > 0: + tps = 1 / total_time + else: + tps = 0 + + # Print Results + print("🎯 BENCHMARK RESULTS") + print("-" * 50) + print(f"Total Time: {total_time:.3f} s") + print(f"Time to First Token: {ttft:.3f} s") + print(f"Generation Time: {generation_time:.3f} s") + print(f"Tokens Generated: {final_token_count} {'(Exact)' if exact_completion_tokens else '(Approx chunk count)'}") + print(f"Tokens Per Second (TPS): {tps:.2f} tokens/sec") + +if __name__ == "__main__": + parser = argparse.ArgumentParser(description="Benchmark an OpenAI-compatible LLM endpoint.") + + parser.add_argument("--url", type=str, default="http://192.168.221.15/v1/chat/completions", + help="API endpoint URL (default: http://192.168.221.15/v1/chat/completions)") + parser.add_argument("--model", type=str, default="qwen3.8-27b-nvfp4", + help="The name of the model to benchmark (default: qwen3.8-27b-nvfp4)") + parser.add_argument("--api-key", type=str, default="EMPTY", + help="API Key (default: EMPTY for local engines)") + parser.add_argument("--prompt", type=str, default="Explain the theory of relativity and quantum mechanics in great detail. Write at least 4 paragraphs.", + help="The prompt to send to the model.") + parser.add_argument("--max-tokens", type=int, default=512, + help="Maximum number of tokens to generate.") + parser.add_argument("--temperature", type=float, default=0.0, + help="Sampling temperature (default 0.0 for deterministic output).") + parser.add_argument("--show-output", action="store_true", + help="Print the model's text response as it generates.") + + args = parser.parse_args() + + benchmark_llm(args.url, args.model, args.api_key, args.prompt, args.max_tokens, args.temperature, args.show_output) diff --git a/point_system/students/views.py b/point_system/students/views.py index 97d3b54..ecf5796 100644 --- a/point_system/students/views.py +++ b/point_system/students/views.py @@ -22,7 +22,7 @@ def get_month_mondays(year, month): c = calendar.Calendar(firstweekday=calendar.MONDAY) return [d for d in c.itermonthdates(year, month) if d.month == month and d.weekday() == 0] -from django.utils import translation +from django.utils import translation, timezone from django.utils.translation import gettext_lazy as _ def set_language(request, lang_code): @@ -443,7 +443,7 @@ def record_attendance(request): point_records = [] if request.method == 'POST': student_ids = request.POST.getlist('student_id') - current_datetime = datetime.now() + current_datetime = timezone.now() four_hours_ago = current_datetime - timedelta(hours=4) for student_id in student_ids: