"""
Can this hosting account run Volt? Run it in the app's folder with the app's Python (cPanel Terminal):

    python check.py             report CPU, memory and limits, then time the model on Solara-sized prompts
    python check.py --download  first download the model into models/model.gguf (1.28 GB, checked by SHA-256)

Uses the same settings as the app (SOLARA_THREADS, SOLARA_CTX, SOLARA_MODEL), so try SOLARA_THREADS=1 / 2 / 4
to see which is fastest on your plan.
"""
import hashlib
import json
import os
import platform
import resource
import secrets
import sys
import time
import urllib.request

HERE = os.path.dirname(os.path.abspath(__file__))
MODEL_PATH = os.environ.get('SOLARA_MODEL') or os.path.join(HERE, 'models', 'model.gguf')
MODEL_URL = 'https://huggingface.co/ggml-org/Qwen3-1.7B-GGUF/resolve/main/Qwen3-1.7B-Q4_K_M.gguf'
MODEL_SHA256 = 'd2387ca2dbfee2ffabce7120d3770dadca0b293052bc2f0e138fdc940d9bc7b5'
THREADS = max(1, int(os.environ.get('SOLARA_THREADS') or 2))
CTX = max(2048, int(os.environ.get('SOLARA_CTX') or 3072))


def line(label, value):
    print('  %-22s %s' % (label, value))


def read(path):
    try:
        with open(path) as f:
            return f.read()
    except OSError:
        return ''


def sha256(path):
    h = hashlib.sha256()
    with open(path, 'rb') as f:
        for chunk in iter(lambda: f.read(1 << 20), b''):
            h.update(chunk)
    return h.hexdigest()


def download():
    os.makedirs(os.path.dirname(MODEL_PATH), exist_ok=True)
    if os.path.exists(MODEL_PATH) and sha256(MODEL_PATH) == MODEL_SHA256:
        print('Model already downloaded and verified.')
        return
    part = MODEL_PATH + '.part'
    print('Downloading the model (1.28 GB) from Hugging Face...')
    with urllib.request.urlopen(MODEL_URL) as r, open(part, 'wb') as out:
        total = int(r.headers.get('Content-Length') or 0)
        done, last = 0, 0
        while True:
            chunk = r.read(1 << 20)
            if not chunk:
                break
            out.write(chunk)
            done += len(chunk)
            if total and done - last > total // 20:
                last = done
                print('  %d%%' % (done * 100 // total))
    print('Checking the file...')
    if sha256(part) != MODEL_SHA256:
        os.remove(part)
        sys.exit('The download is damaged (SHA-256 mismatch). Run again.')
    os.replace(part, MODEL_PATH)
    print('Model saved to ' + MODEL_PATH)


def key_check():
    key = (os.environ.get('SOLARA_AI_KEY') or '').strip()
    print('\nAccess key')
    if len(key) >= 20:
        line('SOLARA_AI_KEY', 'set (ends ...%s)' % key[-4:])
    else:
        print('  Not set. Add this as the environment variable SOLARA_AI_KEY, then restart the app:')
        print('  ' + secrets.token_urlsafe(32))


def report():
    print('\nThis account')
    line('Python', sys.version.split()[0] + ' (' + platform.machine() + ')')
    cpu = read('/proc/cpuinfo')
    model = next((l.split(':', 1)[1].strip() for l in cpu.splitlines() if l.startswith('model name')), 'unknown')
    flags = next((l.split(':', 1)[1].split() for l in cpu.splitlines() if l.startswith('flags')), [])
    line('CPU', model)
    line('CPU features', ', '.join(f for f in ('avx', 'avx2', 'fma', 'f16c', 'avx512f') if f in flags) or 'none found')
    try:
        line('Cores usable', len(os.sched_getaffinity(0)))
    except AttributeError:
        line('Cores usable', os.cpu_count())
    mem = dict(l.split(':', 1) for l in read('/proc/meminfo').splitlines() if ':' in l)
    line('Memory (total)', mem.get('MemTotal', '?').strip())
    line('Memory (free)', mem.get('MemAvailable', '?').strip())
    for path in ('/sys/fs/cgroup/memory.max', '/sys/fs/cgroup/memory/memory.limit_in_bytes'):
        v = read(path).strip()
        if v and v != 'max' and int(v) < 1 << 50:
            line('Memory limit', '%.1f GB' % (int(v) / 1e9))
    soft, _ = resource.getrlimit(resource.RLIMIT_AS)
    line('Address-space limit', 'none' if soft == resource.RLIM_INFINITY else '%.1f GB' % (soft / 1e9))
    if 'avx2' not in flags:
        print('\n  ! No AVX2: the model will be very slow on this CPU.')
    if soft != resource.RLIM_INFINITY and soft < 3e9:
        print('\n  ! The address-space limit is under 3 GB: the model may fail to load. Ask your host to raise it.')


def bench():
    print('\nThe model')
    try:
        import llama_cpp
        from llama_cpp import Llama, LlamaGrammar
    except ImportError:
        sys.exit('  llama-cpp-python is not installed in this Python. See README (pip install step).')
    line('llama-cpp-python', getattr(llama_cpp, '__version__', '?'))
    if not os.path.exists(MODEL_PATH):
        sys.exit('  No model at %s. Run: python check.py --download' % MODEL_PATH)
    line('Threads / context', '%d / %d tokens' % (THREADS, CTX))
    t = time.time()
    llm = Llama(model_path=MODEL_PATH, n_ctx=CTX, n_threads=THREADS, n_threads_batch=THREADS, n_batch=512,
                use_mmap=True, verbose=False)
    line('Load time', '%.1f s' % (time.time() - t))

    schema = {'type': 'object', 'additionalProperties': False, 'required': ['intent', 'request'],
              'properties': {'intent': {'type': 'string', 'enum': ['availability', 'book', 'room_service', 'other']},
                             'request': {'type': 'string', 'maxLength': 80}}}
    grammar = LlamaGrammar.from_json_schema(json.dumps(schema), verbose=False)
    filler = ('Apartment N%d is a two-bedroom unit with a kitchen, Wi-Fi, parking and 24-hour power. '
              'Check-in is from 2pm and check-out by 12 noon. ')
    results = []
    for target, label in ((750, 'Typical message'), (1560, 'Largest message'), (750, 'Repeat (cached)')):
        text, i = '', 0
        while len(llm.tokenize(text.encode(), add_bos=False)) < target - 60:
            i += 1
            text += filler % i
        prompt = ('<|im_start|>system\nYou read hotel guest messages and reply in JSON.\n' + text + '<|im_end|>\n'
                  '<|im_start|>user\nmy AC no dey cool at all<|im_end|>\n<|im_start|>assistant\n<think>\n\n</think>\n\n')
        if label.startswith('Typical'):
            llm.reset()
        t = time.time()
        out = llm.create_completion(prompt=prompt, max_tokens=40, temperature=0, grammar=grammar, stop=['<|im_end|>'])
        secs = time.time() - t
        results.append(secs)
        line(label, '%.1f s for %d tokens -> %s' % (secs, out['usage']['prompt_tokens'], out['choices'][0]['text'].strip()))
    peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss / 1024
    line('Peak memory', '%.0f MB' % peak)

    # A guest message takes one call, sometimes two (understand, then pick an answer).
    worst = results[0] + results[1]
    print('\nVerdict')
    if worst <= 20:
        print('  Good: Volt will answer in about %.0f-%.0f s. Keep the plugin timeout at 25-40 s.' % (results[0], worst))
    elif worst <= 45:
        print('  Usable but slow (up to ~%.0f s). Set the timeout in Solara to 60 and try SOLARA_THREADS=1/4.' % worst)
    else:
        print('  Too slow for chat (~%.0f s). Use a small VPS instead (see ai-server/README.md).' % worst)


if __name__ == '__main__':
    if '--download' in sys.argv:
        download()
    key_check()
    report()
    bench()
