{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":84795,"databundleVersionId":10462807,"sourceType":"competition"},{"sourceId":162798,"sourceType":"modelInstanceVersion","modelInstanceId":138435,"modelId":161088}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true},"papermill":{"default_parameters":{},"duration":22.341371,"end_time":"2024-12-11T03:22:13.479076","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-12-11T03:21:51.137705","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"import io\nimport os\nimport shutil\nimport json\nimport pandas as pd\nimport base64\n\nimport kaggle_evaluation.konwinski_prize_inference_server\n\n# Initialize global variables\ninstance_count = None\nfirst_prediction = True\nrepo_dataframe = None  # Global variable to store the DataFrame\n\ndef get_number_of_instances(num_instances: int) -> None:\n    \"\"\"\n    The very first message from the gateway will be the total number of instances to be served.\n    You don't need to edit this function.\n    \"\"\"\n    global instance_count\n    instance_count = num_instances\n\ndef predict(problem_statement: str, repo_archive: io.BytesIO) -> str:\n    \"\"\"\n    Inference function to read the repository archive into a pandas DataFrame.\n\n    Args:\n        problem_statement: The text of the git issue.\n        repo_archive: A BytesIO buffer containing a .tar archive of the codebase.\n\n    Returns:\n        A JSON string representing the DataFrame with repository contents.\n    \"\"\"\n    global first_prediction, repo_dataframe\n\n    if not first_prediction:\n        return None  # Skip processing if it's not the first prediction.\n\n    try:\n        # Step 1: Write the uploaded repository archive to a file\n        archive_filename = 'repo_archive.tar'\n        with open(archive_filename, 'wb') as f:\n            f.write(repo_archive.read())\n        print(f\"Successfully wrote the archive to '{archive_filename}'.\")\n\n        # Step 2: Define the extraction directory\n        repo_path = 'repo'\n\n        # Step 3: Remove the extraction directory if it already exists to ensure a clean state\n        if os.path.exists(repo_path):\n            shutil.rmtree(repo_path)\n            print(f\"Removed existing directory '{repo_path}/'.\")\n\n        # Step 4: Extract the repository archive to the specified directory\n        try:\n            shutil.unpack_archive(archive_filename, extract_dir=repo_path)\n            print(f\"Successfully extracted the archive to '{repo_path}/'.\")\n        except shutil.ReadError as e:\n            error_message = f\"Error unpacking archive: {e}\"\n            print(error_message)\n            return json.dumps({\"error\": error_message})\n\n        # Step 5: Remove the archive file after extraction to save space\n        os.remove(archive_filename)\n        print(f\"Removed the archive file '{archive_filename}'.\")\n\n        # Step 6: Initialize a list to hold repository data\n        data = []\n\n        # Step 7: Walk through the repository directory to read files\n        for root, dirs, files in os.walk(repo_path):\n            for file in files:\n                file_full_path = os.path.join(root, file)  # Get the full file path\n                relative_path = os.path.relpath(file_full_path, repo_path)\n                try:\n                    # Attempt to read the file content as text\n                    with open(file_full_path, 'r', encoding='utf-8') as f:\n                        content = f.read()\n                    # Append the file path, content, and is_binary flag to the data list\n                    data.append({\n                        'file_path': relative_path,\n                        'content': content,\n                        'is_binary': False\n                    })\n                except UnicodeDecodeError:\n                    # If a UnicodeDecodeError occurs, treat the file as binary\n                    try:\n                        with open(file_full_path, 'rb') as f:\n                            binary_content = f.read()\n                        # Encode the binary content using Base64 to include in JSON\n                        encoded_content = base64.b64encode(binary_content).decode('utf-8')\n                        data.append({\n                            'file_path': relative_path,\n                            'content': encoded_content,\n                            'is_binary': True\n                        })\n                        print(f\"Encoded binary file '{file_full_path}'.\")\n                    except Exception as e:\n                        # If reading as binary also fails, note the error\n                        data.append({\n                            'file_path': relative_path,\n                            'content': f\"Could not read file: {e}\",\n                            'is_binary': None\n                        })\n                        print(f\"Could not read file '{file_full_path}': {e}\")\n                except Exception as e:\n                    # Handle other exceptions\n                    data.append({\n                        'file_path': relative_path,\n                        'content': f\"Could not read file: {e}\",\n                        'is_binary': None\n                    })\n                    print(f\"Could not read file '{file_full_path}': {e}\")\n\n        # Step 8: Create a pandas DataFrame from the collected data\n        repo_df = pd.DataFrame(data)\n        print(\"Successfully created the DataFrame from repository contents.\")\n\n        # Step 9: Convert the DataFrame to a JSON string\n        repo_json = repo_df.to_json(orient='records', indent=2)\n        print(\"Converted the DataFrame to JSON.\")\n\n        # Store the DataFrame in the global variable\n        repo_dataframe = repo_df\n        print(\"Stored the DataFrame in the global variable 'repo_dataframe'.\")\n\n        # Update the prediction flag\n        first_prediction = False\n\n        # Return the JSON string\n        return repo_json\n\n    except Exception as e:\n        # Handle unexpected exceptions and return as JSON error\n        error_response = {\n            \"error\": str(e)\n        }\n        print(f\"An unexpected error occurred: {e}\")\n        first_prediction = False\n        return json.dumps(error_response)\n\n# Initialize the inference server\ninference_server = kaggle_evaluation.konwinski_prize_inference_server.KPrizeInferenceServer(\n    get_number_of_instances,   \n    predict\n)\n\nif os.getenv('KAGGLE_IS_COMPETITION_RERUN'):\n    inference_server.serve()\nelse:\n    inference_server.run_local_gateway(\n        data_paths=(\n            '/kaggle/input/konwinski-prize/',  # Path to the entire competition dataset\n            '/kaggle/tmp/konwinski-prize/',   # Path to a scratch directory for unpacking data.a_zip.\n        )\n    )\n\n# After the inference server has processed the predictions\nif repo_dataframe is not None:\n    # Perform operations on the DataFrame\n    print(repo_dataframe.head())\n    # You can also save it to a file if needed\n    repo_dataframe.to_csv('repository_contents.csv', index=False)\nelse:\n    print(\"The DataFrame 'repo_dataframe' is not available.\")\n","metadata":{}},{"cell_type":"code","source":"!pip install networkx plotly ray\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os \nimport subprocess\n\nsplits = {'dev': 'data/dev-00000-of-00001.parquet', 'test': 'data/test-00000-of-00001.parquet', 'train': 'data/train-00000-of-00001.parquet'}\ndf = pd.read_parquet(\"hf://datasets/princeton-nlp/SWE-bench/\" + splits[\"dev\"])\ndf\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport subprocess\nimport pandas as pd\nimport tempfile\n\n\ndef is_binary_file(filepath):\n    \"\"\"Utility function to detect if a file is binary.\"\"\"\n    with open(filepath, 'rb') as file:\n        chunk = file.read(1024)\n        return b'\\0' in chunk\n\n\ndef analyze_repo_contents(df, index):\n    \"\"\"\n    Analyze repository contents before and after a patch and load them into separate DataFrames,\n    including an 'is_binary' field.\n    \n    Args:\n        df (pd.DataFrame): DataFrame containing SWE-bench data.\n        index (int): Index of the row to analyze.\n\n    Returns:\n        pd.DataFrame, pd.DataFrame: DataFrames for repository contents before and after the patch.\n    \"\"\"\n    if index >= len(df):\n        raise IndexError(\"The provided index is out of range.\")\n\n    # Extract relevant row\n    row = df.iloc[index]\n\n    # Extract repository information\n    repo_url = f\"https://github.com/{row['repo']}.git\"\n    repo_name = row['repo'].split('/')[-1]\n    repo_path = f\"./{repo_name}\"\n\n    # Clone the repository if not already cloned\n    if not os.path.exists(repo_path):\n        print(f\"Cloning repository {repo_url}...\")\n        subprocess.run([\"git\", \"clone\", repo_url, repo_path], check=True)\n    else:\n        print(f\"Repository {repo_name} already cloned.\")\n\n    # Checkout the base commit (before the patch)\n    base_commit = row['base_commit']\n    subprocess.run([\"git\", \"checkout\", base_commit], cwd=repo_path, check=True)\n\n    # Load repository contents before the patch\n    print(f\"Loading repository contents at base commit {base_commit}...\")\n    pre_patch_files = []\n    for root, dirs, files in os.walk(repo_path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            is_binary = is_binary_file(file_path)\n            if is_binary:\n                with open(file_path, \"rb\") as f:  # Binary mode\n                    content = f.read()\n            else:\n                with open(file_path, \"r\", encoding=\"utf-8\", errors=\"ignore\") as f:  # Text mode\n                    content = f.read()\n            pre_patch_files.append({\"file_path\": file_path, \"content\": content, \"is_binary\": is_binary})\n\n    pre_patch_df = pd.DataFrame(pre_patch_files)\n\n    # Save and attempt to apply the patch\n    patch = row['patch']\n    with tempfile.NamedTemporaryFile(delete=False, suffix=\".diff\") as temp_patch_file:\n        patch_file_path = temp_patch_file.name\n        temp_patch_file.write(patch.encode(\"utf-8\"))\n\n    try:\n        subprocess.run([\"git\", \"apply\", patch_file_path], cwd=repo_path, check=True)\n        print(\"Patch applied successfully.\")\n    except subprocess.CalledProcessError as e:\n        print(\"Patch application failed.\")\n        print(f\"Error: {e}\")\n        print(\"Proceeding without applying the patch.\")\n    finally:\n        os.remove(patch_file_path)\n\n    # Load repository contents after attempting to apply the patch\n    print(\"Loading repository contents after patch attempt...\")\n    post_patch_files = []\n    for root, dirs, files in os.walk(repo_path):\n        for file in files:\n            file_path = os.path.join(root, file)\n            is_binary = is_binary_file(file_path)\n            if is_binary:\n                with open(file_path, \"rb\") as f:  # Binary mode\n                    content = f.read()\n            else:\n                with open(file_path, \"r\", encoding=\"utf-8\", errors=\"ignore\") as f:  # Text mode\n                    content = f.read()\n            post_patch_files.append({\"file_path\": file_path, \"content\": content, \"is_binary\": is_binary})\n\n    post_patch_df = pd.DataFrame(post_patch_files)\n\n    return pre_patch_df, post_patch_df\n\n\n# Analyze the first row\npre_patch_df, post_patch_df = analyze_repo_contents(df, index=0)\n\n# View DataFrames\nprint(pre_patch_df.head())\nprint(post_patch_df.head())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define mapping dictionaries\nextension_category_map = {\n    # Source Code\n    '.py': 'Source Code Files',\n    '.js': 'Source Code Files',\n    '.java': 'Source Code Files',\n    '.c': 'Source Code Files',\n    '.cpp': 'Source Code Files',\n    '.cs': 'Source Code Files',\n    '.rb': 'Source Code Files',\n    '.go': 'Source Code Files',\n    '.ts': 'Source Code Files',\n    '.php': 'Source Code Files',\n    '.swift': 'Source Code Files',\n    '.kt': 'Source Code Files',\n\n    # Configuration\n    '.cfg': 'Configuration Files',\n    '.toml': 'Configuration Files',\n    '.yaml': 'Configuration Files',\n    '.yml': 'Configuration Files',\n    '.json': 'Configuration Files',\n    '.ini': 'Configuration Files',\n    '.env': 'Configuration Files',\n    '.editorconfig': 'Configuration Files',\n    '.git-blame-ignore-revs': 'Version Control Configuration Files',\n    'pyproject.toml': 'Configuration Files',\n\n    # Documentation\n    '.md': 'Documentation Files',\n    '.rst': 'Documentation Files',\n    '.txt': 'Documentation Files',\n    '.adoc': 'Documentation Files',\n\n    # License and Legal\n    'LICENSE': 'License and Legal Files',\n    'LICENSE.txt': 'License and Legal Files',\n    'LICENSE.md': 'License and Legal Files',\n    'NOTICE': 'License and Legal Files',\n\n    # Scripts\n    '.sh': 'Scripts and Utilities',\n    '.bat': 'Scripts and Utilities',\n    '.ps1': 'Scripts and Utilities',\n    '.pyw': 'Scripts and Utilities',\n\n    # Testing\n    '.test': 'Testing Files',\n    '.spec': 'Testing Files',\n    '.pytest': 'Testing Files',\n    'pytest.ini': 'Testing Files',\n    'tox.ini': 'Testing Files',\n\n    # Build and Deployment\n    'Dockerfile': 'Build and Deployment Files',\n    '.dockerignore': 'Build and Deployment Files',\n    'Makefile': 'Build and Deployment Files',\n    'docker-compose.yml': 'Build and Deployment Files',\n    'Jenkinsfile': 'Build and Deployment Files',\n    'build.gradle': 'Build and Deployment Files',\n    'pom.xml': 'Build and Deployment Files',\n\n    # Version Control Configuration\n    '.gitignore': 'Version Control Configuration Files',\n    '.gitattributes': 'Version Control Configuration Files',\n    '.gitmodules': 'Version Control Configuration Files',\n\n    # Workflow and CI\n    '.travis.yml': 'Workflow and CI Files',\n    '.circleci/config.yml': 'Workflow and CI Files',\n    '.github/workflows/ci.yaml': 'Workflow and CI Files',\n    '.github/workflows/release-tests.yml': 'Workflow and CI Files',\n    '.github/workflows/release.yml': 'Workflow and CI Files',\n    '.github/workflows/codeql-analysis.yml': 'Workflow and CI Files',\n    '.github/workflows/backport.yml': 'Workflow and CI Files',\n    # Add more special files as needed\n}\n\n# Directory patterns mapped to categories\ndirectory_category_map = {\n    'docs/': 'Documentation Files',\n    'docs': 'Documentation Files',\n    'test/': 'Testing Files',\n    'tests/': 'Testing Files',\n    'assets/': 'Binary and Asset Files',\n    'static/': 'Binary and Asset Files',\n    'scripts/': 'Scripts and Utilities',\n    'bin/': 'Scripts and Utilities',\n    'config/': 'Configuration Files',\n    'src/': 'Source Code Files',\n    'lib/': 'Source Code Files',\n    'include/': 'Source Code Files',\n    'examples/': 'Examples and Demos',\n    'public/': 'Public Assets',\n    # Add more directory patterns as needed\n}\n\n# Binary file extensions\nbinary_extensions = {\n    '.png', '.jpg', '.jpeg', '.gif', '.svg', '.exe', '.dll', '.so',\n    '.bin', '.ico', '.pdf', '.zip', '.tar', '.gz', '.7z', '.rar',\n    '.mp3', '.mp4', '.avi', '.mov', '.wmv', '.flv', '.mkv', '.bmp',\n    '.tiff', '.woff', '.woff2', '.ttf', '.eot', '.otf', '.dmg',\n    '.iso', '.jar', '.war', '.ear'\n}\n\n# Function to classify files\ndef classify_file(row):\n    file_path = row['file_path']\n    is_binary = row['is_binary']\n    \n    # Normalize file_path for consistent matching\n    normalized_path = file_path.lower()\n    \n    # Check for exact matches first (e.g., Dockerfile, LICENSE)\n    if file_path in extension_category_map:\n        return extension_category_map[file_path]\n    \n    # Check directory patterns\n    for dir_pattern, category in directory_category_map.items():\n        if normalized_path.startswith(dir_pattern):\n            return category\n    \n    # Extract the file extension or specific filename\n    basename = os.path.basename(file_path)\n    _, ext = os.path.splitext(basename)\n    ext = ext.lower()\n    \n    # Check exact filename matches if not already matched\n    if basename in extension_category_map:\n        return extension_category_map[basename]\n    \n    # Check extension-based category\n    if ext in extension_category_map:\n        return extension_category_map[ext]\n    \n    # Check for binary files based on extension\n    if is_binary or ext in binary_extensions:\n        return 'Binary and Asset Files'\n    \n    # Default category\n    return 'Miscellaneous Files'\n\n# Apply classification\npre_patch_df['Category'] = pre_patch_df.apply(classify_file, axis=1)\npost_patch_df['Category'] = post_patch_df.apply(classify_file, axis=1)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pre_patch_df.columns","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pre_patch_df.iloc[:1, :]","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pre_patch_df['Category'].value_counts()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"post_patch_df['Category'].value_counts()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Codebase Deep Dive**","metadata":{}},{"cell_type":"code","source":"import torch\n\nprint(\"GPUs available:\", torch.cuda.device_count())\nfor i in range(torch.cuda.device_count()):\n    print(f\"GPU {i}: {torch.cuda.get_device_name(i)}\")\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import ray\n\nray.init(num_gpus=torch.cuda.device_count())\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport logging\nfrom transformers import AutoTokenizer, AutoModelForCausalLM\nimport torch\nimport pandas as pd\nfrom tqdm import tqdm\nimport ray\nfrom typing import List, Dict\nimport numpy as np\n\n# Configuration\nMODEL_DIR = \"/kaggle/input/qwen2.5-coder/transformers/1.5b-instruct/1\"\nHF_TOKEN = \"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\"\nOUTPUT_FILE = \"complete_codebase_documentation.md\"\nLOG_FILE = \"repo_inferencer.log\"\nMAX_TOKENS = 4096\nCHUNK_SIZE = 2000\n\nPROMPT_TEMPLATES = {\n    \"Source Code\": {\n        \"summary\": \"\"\"You are an expert software developer tasked with analyzing and documenting source code. Follow these steps to provide a comprehensive summary:\n\nStep 1: Code Understanding\n- Carefully read through the provided source code\n- Identify the main purpose and functionality\n- Note any important dependencies or imports\n\nStep 2: Architecture Analysis\n- Identify key classes, functions, and modules\n- Understand the code structure and organization\n- Note any design patterns or architectural choices\n\nStep 3: Generate Summary\nProvide a clear summary addressing:\n1. Main purpose and functionality\n2. Key components and their roles\n3. Important dependencies\n4. Notable design patterns or architectural decisions\n5. Any performance considerations\n\nFile: {file_path}\n\nCode:\n{content}\n\nSummary:\"\"\",\n\n        \"key_points\": \"\"\"As an expert developer, extract key technical points from this code by following these steps:\n\nStep 1: Technical Analysis\n- Review the code's technical implementation\n- Identify core algorithms and data structures\n- Note any important configurations or settings\n\nStep 2: Implementation Details\n- List critical functions and their purposes\n- Document important class relationships\n- Note any complex logic or algorithms\n\nStep 3: Identify Key Technical Points\nFocus on:\n1. Critical functions and their roles\n2. Important class hierarchies\n3. Key algorithms and data structures\n4. Configuration parameters\n5. Error handling mechanisms\n6. Performance optimizations\n\nFile: {file_path}\n\nCode:\n{content}\n\nKey Technical Points:\"\"\"\n    },\n\n    \"Configuration\": {\n        \"summary\": \"\"\"As a configuration specialist, analyze this configuration file following these steps:\n\nStep 1: Configuration Analysis\n- Identify the configuration type and format\n- Understand the scope and purpose\n- Note any environment-specific settings\n\nStep 2: Settings Review\n- Review all configuration parameters\n- Identify critical settings\n- Note any security-related configurations\n\nStep 3: Generate Summary\nAddress:\n1. Configuration file purpose\n2. Scope and environment context\n3. Critical settings and their impacts\n4. Security considerations\n5. Integration points\n\nFile: {file_path}\n\nConfiguration:\n{content}\n\nSummary:\"\"\",\n\n        \"key_points\": \"\"\"As a configuration expert, extract key configuration points following these steps:\n\nStep 1: Parameter Analysis\n- Identify all important parameters\n- Understand their purposes and impacts\n- Note any dependencies between settings\n\nStep 2: Critical Settings\n- List mission-critical configurations\n- Document default values and their implications\n- Note any security-sensitive settings\n\nStep 3: Generate Key Points\nFocus on:\n1. Essential configuration parameters\n2. Environment-specific settings\n3. Security configurations\n4. Integration parameters\n5. Performance-related settings\n\nFile: {file_path}\n\nConfiguration:\n{content}\n\nKey Configuration Points:\"\"\"\n    },\n\n    \"Documentation\": {\n        \"summary\": \"\"\"As a technical documentation expert, analyze this documentation following these steps:\n\nStep 1: Content Analysis\n- Understand the documentation scope\n- Identify main topics covered\n- Note any important guidelines or requirements\n\nStep 2: Documentation Review\n- Evaluate completeness and clarity\n- Identify key information sections\n- Note any technical specifications\n\nStep 3: Generate Summary\nAddress:\n1. Documentation purpose and scope\n2. Main topics covered\n3. Key guidelines or requirements\n4. Technical specifications\n5. Important usage examples\n\nFile: {file_path}\n\nDocumentation:\n{content}\n\nSummary:\"\"\",\n\n        \"key_points\": \"\"\"As a documentation specialist, extract key documentation points following these steps:\n\nStep 1: Content Review\n- Identify critical information\n- Note important guidelines\n- List key examples or demonstrations\n\nStep 2: Technical Details\n- Extract technical specifications\n- Note implementation requirements\n- List important references\n\nStep 3: Generate Key Points\nFocus on:\n1. Critical guidelines\n2. Technical requirements\n3. Important examples\n4. Best practices\n5. Key references\n\nFile: {file_path}\n\nDocumentation:\n{content}\n\nKey Documentation Points:\"\"\"\n    },\n\n    \"Build\": {\n        \"summary\": \"\"\"As a build system expert, analyze this build configuration following these steps:\n\nStep 1: Build Process Analysis\n- Understand build steps and dependencies\n- Identify build targets and artifacts\n- Note any special build requirements\n\nStep 2: Configuration Review\n- Review build settings and parameters\n- Identify critical dependencies\n- Note any platform-specific configurations\n\nStep 3: Generate Summary\nAddress:\n1. Build process overview\n2. Key targets and artifacts\n3. Critical dependencies\n4. Platform requirements\n5. Build optimization settings\n\nFile: {file_path}\n\nBuild Configuration:\n{content}\n\nSummary:\"\"\",\n\n        \"key_points\": \"\"\"As a build system specialist, extract key build points following these steps:\n\nStep 1: Build Configuration Analysis\n- Identify critical build steps\n- List important dependencies\n- Note build optimization settings\n\nStep 2: Platform Requirements\n- Document platform-specific settings\n- Note compatibility requirements\n- List required tools and versions\n\nStep 3: Generate Key Points\nFocus on:\n1. Critical build steps\n2. Essential dependencies\n3. Platform requirements\n4. Build optimizations\n5. Tool requirements\n\nFile: {file_path}\n\nBuild Configuration:\n{content}\n\nKey Build Points:\"\"\"\n    }\n}\n\n# Logging setup\nlogging.basicConfig(\n    filename=LOG_FILE,\n    filemode='a',\n    format='%(asctime)s - %(levelname)s - %(message)s',\n    level=logging.INFO\n)\n\ndef get_available_gpus() -> List[int]:\n    \"\"\"Get available GPU devices with proper error handling.\"\"\"\n    try:\n        if not torch.cuda.is_available():\n            logging.warning(\"CUDA not available, falling back to CPU\")\n            return []\n            \n        count = torch.cuda.device_count()\n        if count == 0:\n            logging.warning(\"No CUDA devices available, falling back to CPU\")\n            return []\n            \n        # Verify each GPU is actually accessible\n        valid_gpus = []\n        for gpu_id in range(count):\n            try:\n                with torch.cuda.device(gpu_id):\n                    # Test GPU accessibility with a small tensor operation\n                    torch.zeros(1).cuda()\n                valid_gpus.append(gpu_id)\n            except RuntimeError as e:\n                logging.warning(f\"GPU {gpu_id} not accessible: {e}\")\n                \n        if not valid_gpus:\n            logging.warning(\"No accessible GPUs found, falling back to CPU\")\n            \n        return valid_gpus\n        \n    except Exception as e:\n        logging.warning(f\"Error detecting GPUs: {e}\")\n        return []\n\nclass ModelWorker:\n    def __init__(self, model_dir: str, hf_token: str, gpu_id: int):\n        self.device = f'cuda:{gpu_id}' if gpu_id >= 0 else 'cpu'\n        try:\n            self.tokenizer = AutoTokenizer.from_pretrained(\n                model_dir,\n                token=hf_token,\n                trust_remote_code=True\n            )\n            \n            if gpu_id >= 0:\n                with torch.cuda.device(gpu_id):\n                    self._init_model(model_dir, hf_token)\n            else:\n                self._init_model(model_dir, hf_token)\n        except Exception as e:\n            logging.error(f\"Error initializing model worker: {e}\")\n            raise\n\n    def _init_model(self, model_dir: str, hf_token: str):\n        try:\n            self.model = AutoModelForCausalLM.from_pretrained(\n                model_dir,\n                trust_remote_code=True,\n                torch_dtype=torch.bfloat16 if self.device != 'cpu' else torch.float32,\n                token=hf_token\n            ).to(self.device)\n            self.model.eval()\n        except Exception as e:\n            logging.error(f\"Error initializing model: {e}\")\n            raise\n\n    def generate_text(self, prompt: str, chunk_size: int = CHUNK_SIZE) -> str:\n        try:\n            if len(prompt) <= chunk_size:\n                if self.device.startswith('cuda'):\n                    with torch.cuda.device(int(self.device.split(':')[1])):\n                        return self._generate(prompt)\n                return self._generate(prompt)\n                \n            return self._generate_chunked(prompt, chunk_size)\n        except Exception as e:\n            logging.error(f\"Error generating text: {e}\")\n            return f\"Error generating text: {str(e)}\"\n\n    def _generate_chunked(self, prompt: str, chunk_size: int) -> str:\n        chunks = []\n        memory = \"\"\n        remaining_text = prompt\n        \n        while remaining_text:\n            # Extract current chunk\n            current_chunk = remaining_text[:chunk_size]\n            remaining_text = remaining_text[chunk_size:]\n            \n            # Generate with memory context\n            context = f\"{memory}\\n\\nContinuing previous analysis...\\n{current_chunk}\"\n            chunk_output = self._generate(context)\n            \n            # Update memory with condensed previous output\n            memory = self._condense_memory(chunk_output)\n            chunks.append(chunk_output)\n        \n        # Combine and refine final output\n        combined = \" \".join(chunks)\n        return self._generate(f\"Please provide a coherent final version of this analysis:\\n{combined}\")\n            \n    def _condense_memory(self, text: str, max_length: int = 1000) -> str:\n        \"\"\"Condense previous output to maintain key context.\"\"\"\n        if len(text) <= max_length:\n            return text\n            \n        # Extract key points (first and last parts)\n        start = text[:max_length//2]\n        end = text[-max_length//2:]\n        return f\"{start}...\\n{end}\"\n\n    def _generate(self, prompt: str) -> str:\n        inputs = self.tokenizer(prompt, return_tensors=\"pt\").to(self.device)\n        \n        with torch.no_grad():\n            output_ids = self.model.generate(\n                inputs.input_ids,\n                max_length=MAX_TOKENS,\n                num_return_sequences=1,\n                temperature=0.7,\n                do_sample=True,\n                pad_token_id=self.tokenizer.pad_token_id\n            )\n        \n        return self.tokenizer.decode(output_ids[0], skip_special_tokens=True)\n\n    def process_file(self, file_data: Dict) -> Dict:\n        try:\n            file_path = file_data['file_path']\n            content = file_data['content']\n            category = file_data['category']\n            \n            template = PROMPT_TEMPLATES.get(category, PROMPT_TEMPLATES[\"Documentation\"])\n            \n            summary = self.generate_text(\n                template[\"summary\"].format(\n                    file_path=file_path,\n                    content=content\n                )\n            )\n            \n            key_points = self.generate_text(\n                template[\"key_points\"].format(\n                    file_path=file_path,\n                    content=content\n                )\n            )\n            \n            return {\n                'category': category,\n                'file_path': file_path,\n                'summary': summary,\n                'key_points': key_points\n            }\n        except Exception as e:\n            logging.error(f\"Error processing file {file_path}: {e}\")\n            return {\n                'category': category,\n                'file_path': file_path,\n                'summary': f\"Error during processing: {str(e)}\",\n                'key_points': \"Processing failed\"\n            }\n\n@ray.remote(num_gpus=1)\nclass DistributedModelWorker(ModelWorker):\n    pass\n\ndef init_ray_cluster():\n    if not ray.is_initialized():\n        try:\n            ray.init()\n        except Exception as e:\n            logging.error(f\"Error initializing Ray cluster: {e}\")\n            raise\n\ndef analyze_codebase(repo_df: pd.DataFrame) -> pd.DataFrame:\n    init_ray_cluster()\n    available_gpus = get_available_gpus()\n    \n    # CPU-only mode\n    if not available_gpus:\n        logging.info(\"Running in CPU-only mode\")\n        worker = ModelWorker(MODEL_DIR, HF_TOKEN, -1)\n        results = []\n        for file_data in tqdm(repo_df.to_dict('records')):\n            try:\n                results.append(worker.process_file(file_data))\n            except Exception as e:\n                logging.error(f\"Error processing file {file_data['file_path']}: {e}\")\n                results.append({\n                    'category': file_data['category'],\n                    'file_path': file_data['file_path'],\n                    'summary': f\"Error during processing: {str(e)}\",\n                    'key_points': \"Processing failed\"\n                })\n        return pd.DataFrame(results)\n    \n    # GPU processing with robust error handling\n    try:\n        workers = [\n            DistributedModelWorker.remote(MODEL_DIR, HF_TOKEN, gpu_id)\n            for gpu_id in available_gpus\n        ]\n        \n        files = repo_df.to_dict('records')\n        num_workers = len(workers)\n        files_per_worker = np.array_split(files, num_workers)\n        \n        futures = []\n        for worker_idx, worker_files in enumerate(files_per_worker):\n            worker = workers[worker_idx]\n            futures.extend([\n                worker.process_file.remote(file_data)\n                for file_data in worker_files\n            ])\n        \n        results = ray.get(futures)\n        return pd.DataFrame(results)\n        \n    except Exception as e:\n        logging.error(f\"Error in GPU processing, falling back to CPU: {e}\")\n        return analyze_codebase(repo_df)  # Recursive call will trigger CPU path\n\ndef generate_documentation(results_df: pd.DataFrame) -> str:\n    doc = \"# Comprehensive Codebase Documentation\\n\\n\"\n    \n    for category in sorted(results_df['category'].unique()):\n        doc += f\"## {category}\\n\\n\"\n        category_files = results_df[results_df['category'] == category]\n        \n        for _, row in category_files.iterrows():\n            doc += f\"### {row['file_path']}\\n\\n\"\n            doc += \"#### Summary\\n\\n\"\n            doc += f\"{row['summary']}\\n\\n\"\n            doc += \"#### Key Technical Points\\n\\n\"\n            doc += f\"{row['key_points']}\\n\\n\"\n            doc += \"---\\n\\n\"\n    \n    return doc\n\ndef main(repo_df: pd.DataFrame):\n    try:\n        # Preprocessing the DataFrame\n        repo_df = repo_df[repo_df['is_binary'] == False]\n        repo_df = repo_df.rename(columns={'Category': 'category'})\n        \n        if repo_df.empty:\n            logging.info(\"No valid files to process.\")\n            return \"No valid files to process.\"\n        \n        # Analyze the codebase\n        results_df = analyze_codebase(repo_df)\n        \n        # Generate documentation\n        documentation = generate_documentation(results_df)\n        \n        # Write to output file\n        with open(OUTPUT_FILE, 'w', encoding='utf-8') as f:\n            f.write(documentation)\n            \n        logging.info(f\"Documentation generated successfully: {OUTPUT_FILE}\")\n        return documentation\n        \n    except Exception as e:\n        logging.error(f\"Error in documentation generation: {e}\")\n        raise\n    finally:\n        if ray.is_initialized():\n            ray.shutdown()\n\nif __name__ == \"__main__\":\n    # Example usage\n    doc = main(pre_patch_df.iloc[:10, :])\n    print(doc)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for source code files in the dataframe\nsource_code_extensions = ['.py', '.js', '.jsx', '.ts', '.tsx', '.java', '.c', '.cpp', '.h', '.hpp', '.rb', '.go', '.sh']\nsource_code_df = pre_patch_df[pre_patch_df['file_path'].str.endswith(tuple(source_code_extensions))]\nprint(f\"Number of source code files: {len(source_code_df)}\")\nprint(source_code_df[['file_path', 'Category']])\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"repo_dataframe['file_path'] = repo_dataframe['file_path'].apply(os.path.normpath)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Knowledge Graph","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport networkx as nx\nimport os\nimport re\nimport logging\nfrom functools import lru_cache\nfrom typing import List\nimport json\nimport ast\nimport matplotlib.pyplot as plt\nimport networkx as nx\nfrom matplotlib.patches import Patch\nimport numpy as np\n\n# Configure logging to DEBUG for detailed output\nlogging.basicConfig(level=logging.DEBUG, format='%(levelname)s:%(message)s')\n\nclass KnowledgeGraph:\n    def __init__(self, dataframe: pd.DataFrame):\n        self.df = dataframe.copy()\n        # Normalize all file paths to ensure consistency\n        self.df['file_path'] = self.df['file_path'].apply(os.path.normpath)\n        self.graph = nx.DiGraph()\n        self.required_columns = {'file_path', 'content', 'is_binary', 'Category'}\n        self.validate_dataframe()\n        self.add_nodes()\n        self.cache_file_content()\n    \n    def validate_dataframe(self):\n        if not self.required_columns.issubset(self.df.columns):\n            missing = self.required_columns - set(self.df.columns)\n            logging.error(f\"Dataframe is missing required columns: {missing}\")\n            raise ValueError(f\"Dataframe must contain columns: {self.required_columns}\")\n        logging.info(\"Dataframe loaded successfully with all required columns.\")\n    \n    def add_nodes(self):\n        for _, row in self.df.iterrows():\n            try:\n                self.graph.add_node(row['file_path'], category=row['Category'])\n                logging.debug(f\"Added node: {row['file_path']} with category: {row['Category']}\")\n            except Exception as e:\n                logging.error(f\"Error adding node for file {row['file_path']}: {e}\")\n    \n    @lru_cache(maxsize=None)\n    def get_file_content(self, file_path: str) -> str:\n        try:\n            content = self.df.loc[self.df['file_path'] == file_path, 'content'].values[0]\n            return content\n        except IndexError:\n            logging.warning(f\"File content not found for {file_path}.\")\n            return \"\"\n    \n    def extract_dependencies(self, file_path: str, content: str, category: str) -> List[str]:\n        dependencies = []\n        try:\n            if category == 'Source Code Files':\n                if file_path.endswith('.py'):\n                    dependencies.extend(self.extract_python_dependencies(file_path, content))\n                elif file_path.endswith('.js') or file_path.endswith('.jsx'):\n                    dependencies.extend(self.extract_javascript_dependencies(file_path, content))\n                elif file_path.endswith('.ts') or file_path.endswith('.tsx'):\n                    dependencies.extend(self.extract_typescript_dependencies(file_path, content))\n                elif file_path.endswith('.java'):\n                    dependencies.extend(self.extract_java_dependencies(file_path, content))\n                elif file_path.endswith(('.cpp', '.c', '.hpp', '.h')):\n                    dependencies.extend(self.extract_cpp_dependencies(file_path, content))\n                elif file_path.endswith('.rb'):\n                    dependencies.extend(self.extract_ruby_dependencies(file_path, content))\n                elif file_path.endswith('.go'):\n                    dependencies.extend(self.extract_go_dependencies(file_path, content))\n                # Add more languages as needed\n            elif category == 'Testing Files':\n                if file_path.endswith('.py'):\n                    dependencies.extend(self.extract_python_dependencies(file_path, content))\n                elif file_path.endswith('.js') or file_path.endswith('.jsx'):\n                    dependencies.extend(self.extract_javascript_dependencies(file_path, content))\n                # Add more languages as needed\n            elif category == 'Scripts and Utilities':\n                dependencies.extend(self.extract_shell_dependencies(file_path, content))\n            elif category == 'Documentation Files':\n                dependencies.extend(self.extract_markdown_assets(file_path, content))\n            elif category == 'Configuration Files':\n                dependencies.extend(self.extract_config_dependencies(file_path, content))\n            elif category == 'Workflow and CI Files':\n                dependencies.extend(self.extract_workflow_dependencies(file_path, content))\n            # Add more categories and their extraction functions as needed\n        except Exception as e:\n            logging.error(f\"Error extracting dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    # Dependency extraction methods for various languages and file types\n    \n    def extract_python_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            tree = ast.parse(content)\n            for node in ast.walk(tree):\n                if isinstance(node, ast.Import):\n                    for alias in node.names:\n                        module = alias.name.split('.')[0]\n                        dep_file = os.path.normpath(os.path.join(os.path.dirname(file_path), f\"{module}.py\"))\n                        if dep_file in self.df['file_path'].values:\n                            dependencies.append(dep_file)\n                            logging.debug(f\"Python dependency found: {dep_file} imported in {file_path}\")\n                elif isinstance(node, ast.ImportFrom):\n                    module = node.module.split('.')[0] if node.module else ''\n                    if module:\n                        dep_file = os.path.normpath(os.path.join(os.path.dirname(file_path), f\"{module}.py\"))\n                        if dep_file in self.df['file_path'].values:\n                            dependencies.append(dep_file)\n                            logging.debug(f\"Python dependency found: {dep_file} imported in {file_path}\")\n        except SyntaxError as se:\n            logging.warning(f\"Syntax error while parsing {file_path}: {se}\")\n        except Exception as e:\n            logging.error(f\"Unexpected error while parsing {file_path}: {e}\")\n        return dependencies\n    \n    def extract_javascript_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # ES6 import statements: import something from './module.js'\n            pattern = r'import\\s+.*\\s+from\\s+[\\'\"](.+?\\.js)[\\'\"]'\n            matches = re.findall(pattern, content)\n            for match in matches:\n                dep_path = os.path.normpath(os.path.join(os.path.dirname(file_path), match))\n                if dep_path in self.df['file_path'].values:\n                    dependencies.append(dep_path)\n                    logging.debug(f\"JavaScript dependency found: {dep_path} imported in {file_path}\")\n            # CommonJS require statements: const module = require('./module.js')\n            pattern_cjs = r'require\\([\\'\"](.+?\\.js)[\\'\"]\\)'\n            matches_cjs = re.findall(pattern_cjs, content)\n            for match in matches_cjs:\n                dep_path = os.path.normpath(os.path.join(os.path.dirname(file_path), match))\n                if dep_path in self.df['file_path'].values:\n                    dependencies.append(dep_path)\n                    logging.debug(f\"JavaScript dependency found: {dep_path} required in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting JavaScript dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_typescript_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # ES6 import statements: import something from './module.ts'\n            pattern = r'import\\s+.*\\s+from\\s+[\\'\"](.+?\\.ts)[\\'\"]'\n            matches = re.findall(pattern, content)\n            for match in matches:\n                dep_path = os.path.normpath(os.path.join(os.path.dirname(file_path), match))\n                if dep_path in self.df['file_path'].values:\n                    dependencies.append(dep_path)\n                    logging.debug(f\"TypeScript dependency found: {dep_path} imported in {file_path}\")\n            # CommonJS require statements: const module = require('./module.ts')\n            pattern_cjs = r'require\\([\\'\"](.+?\\.ts)[\\'\"]\\)'\n            matches_cjs = re.findall(pattern_cjs, content)\n            for match in matches_cjs:\n                dep_path = os.path.normpath(os.path.join(os.path.dirname(file_path), match))\n                if dep_path in self.df['file_path'].values:\n                    dependencies.append(dep_path)\n                    logging.debug(f\"TypeScript dependency found: {dep_path} required in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting TypeScript dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_java_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # Java import statements: import com.example.Module;\n            pattern = r'import\\s+([a-zA-Z0-9_.]+);'\n            matches = re.findall(pattern, content)\n            for match in matches:\n                module = match.split('.')[-1]\n                dep_file = os.path.normpath(os.path.join(os.path.dirname(file_path), f\"{module}.java\"))\n                if dep_file in self.df['file_path'].values:\n                    dependencies.append(dep_file)\n                    logging.debug(f\"Java dependency found: {dep_file} imported in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting Java dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_cpp_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # C/C++ include statements: #include \"module.h\"\n            pattern = r'#include\\s+[<\"](.+?\\.h)[>\"]'\n            matches = re.findall(pattern, content)\n            for match in matches:\n                dep_path = os.path.normpath(os.path.join(os.path.dirname(file_path), match))\n                if dep_path in self.df['file_path'].values:\n                    dependencies.append(dep_path)\n                    logging.debug(f\"C/C++ dependency found: {dep_path} included in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting C/C++ dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_ruby_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # Ruby require statements: require 'module'\n            pattern = r'require\\s+[\\'\"](.+?)[\\'\"]'\n            matches = re.findall(pattern, content)\n            for match in matches:\n                dep_file = os.path.normpath(os.path.join(os.path.dirname(file_path), f\"{match}.rb\"))\n                if dep_file in self.df['file_path'].values:\n                    dependencies.append(dep_file)\n                    logging.debug(f\"Ruby dependency found: {dep_file} required in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting Ruby dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_go_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # Go import statements: import \"github.com/user/module\"\n            pattern = r'import\\s+\"([^\"]+)\"'\n            matches = re.findall(pattern, content)\n            for match in matches:\n                module = match.split('/')[-1]\n                dep_file = os.path.normpath(os.path.join(os.path.dirname(file_path), f\"{module}.go\"))\n                if dep_file in self.df['file_path'].values:\n                    dependencies.append(dep_file)\n                    logging.debug(f\"Go dependency found: {dep_file} imported in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting Go dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_shell_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # Look for 'docker build' commands referencing Dockerfile\n            if 'docker build' in content:\n                dockerfile = os.path.normpath(os.path.join(os.path.dirname(file_path), 'Dockerfile'))\n                if dockerfile in self.df['file_path'].values:\n                    dependencies.append(dockerfile)\n                    logging.debug(f\"Shell dependency found: {dockerfile} referenced in {file_path}\")\n            # Sourcing other scripts: source scripts/helper.sh\n            sourced_scripts = re.findall(r'source\\s+(.+?\\.sh)', content)\n            for script in sourced_scripts:\n                script_path = os.path.normpath(os.path.join(os.path.dirname(file_path), script))\n                if script_path in self.df['file_path'].values:\n                    dependencies.append(script_path)\n                    logging.debug(f\"Shell dependency found: {script_path} sourced in {file_path}\")\n            # Executing other scripts: ./scripts/setup.sh\n            executed_scripts = re.findall(r'\\./(.+?\\.sh)', content)\n            for script in executed_scripts:\n                script_path = os.path.normpath(os.path.join(os.path.dirname(file_path), script))\n                if script_path in self.df['file_path'].values:\n                    dependencies.append(script_path)\n                    logging.debug(f\"Shell dependency found: {script_path} executed in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting shell dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_markdown_assets(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # Regex to find image links: ![Alt Text](assets/image.png)\n            pattern = r'!\\[.*?\\]\\((assets/[^)]+\\.(png|jpg|jpeg|gif|svg))\\)'\n            matches = re.findall(pattern, content, re.IGNORECASE)\n            for match in matches:\n                asset_path = os.path.normpath(match[0])\n                if asset_path in self.df['file_path'].values:\n                    dependencies.append(asset_path)\n                    logging.debug(f\"Markdown asset found: {asset_path} referenced in {file_path}\")\n            # Regex to find other asset references, e.g., scripts or styles\n            pattern_assets = r'\\((assets/[^)]+)\\)'\n            matches_assets = re.findall(pattern_assets, content, re.IGNORECASE)\n            for asset in matches_assets:\n                asset_path = os.path.normpath(asset)\n                if asset_path in self.df['file_path'].values:\n                    dependencies.append(asset_path)\n                    logging.debug(f\"Markdown asset found: {asset_path} referenced in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting Markdown assets from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_config_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # Parse JSON configuration files\n            if file_path.endswith('.json'):\n                try:\n                    config = json.loads(content)\n                    # Example: Look for file references in specific keys\n                    # Modify based on actual config structure\n                    # For demonstration, assume 'scripts' key contains script paths\n                    scripts = config.get('scripts', {})\n                    for script_path in scripts.values():\n                        script_path = os.path.normpath(script_path)\n                        if script_path in self.df['file_path'].values:\n                            dependencies.append(script_path)\n                            logging.debug(f\"Config dependency found: {script_path} referenced in {file_path}\")\n                except json.JSONDecodeError:\n                    logging.warning(f\"JSON decode error in {file_path}\")\n            # Parse YAML configuration files\n            elif file_path.endswith(('.yml', '.yaml')):\n                try:\n                    import yaml\n                    config = yaml.safe_load(content)\n                    # Example: Look for file references in specific keys\n                    scripts = config.get('scripts', {})\n                    for script_path in scripts.values():\n                        script_path = os.path.normpath(script_path)\n                        if script_path in self.df['file_path'].values:\n                            dependencies.append(script_path)\n                            logging.debug(f\"Config dependency found: {script_path} referenced in {file_path}\")\n                except ImportError:\n                    logging.error(\"PyYAML is not installed. Install it using 'pip install pyyaml'\")\n                except yaml.YAMLError:\n                    logging.warning(f\"YAML parse error in {file_path}\")\n            # Add more configuration file types as needed\n        except Exception as e:\n            logging.error(f\"Error extracting configuration dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def extract_workflow_dependencies(self, file_path: str, content: str) -> List[str]:\n        dependencies = []\n        try:\n            # Look for 'run: scripts/deploy.sh' or similar\n            run_scripts = re.findall(r'run:\\s*(?:bash\\s+)?(.+?\\.sh)', content)\n            for script in run_scripts:\n                script_path = os.path.normpath(os.path.join(os.path.dirname(file_path), script))\n                if script_path in self.df['file_path'].values:\n                    dependencies.append(script_path)\n                    logging.debug(f\"Workflow dependency found: {script_path} run in {file_path}\")\n            \n            # Look for test scripts (e.g., pytest)\n            test_scripts = re.findall(r'run:\\s*pytest\\s+(.+)', content)\n            for test_script in test_scripts:\n                # Assuming tests are in 'tests/' directory or similar\n                test_file = os.path.normpath(os.path.join(os.path.dirname(file_path), test_script.strip()))\n                if test_file in self.df['file_path'].values:\n                    dependencies.append(test_file)\n                    logging.debug(f\"Workflow dependency found: {test_file} tested in {file_path}\")\n        except Exception as e:\n            logging.error(f\"Error extracting workflow dependencies from {file_path}: {e}\")\n        return dependencies\n    \n    def build_edges_sequential(self):\n        \"\"\"\n        Builds edges in the graph based on dependencies sequentially.\n        This method replaces the multiprocessing approach for easier debugging.\n        \"\"\"\n        logging.info(\"Starting sequential edge building.\")\n        for idx, row in self.df.iterrows():\n            file_path = row['file_path']\n            content = row['content']\n            category = row['Category']\n            if row['is_binary']:\n                logging.debug(f\"Skipping binary file: {file_path}\")\n                continue  # Skip binary files\n            dependencies = self.extract_dependencies(file_path, content, category)\n            logging.debug(f\"Dependencies for {file_path}: {dependencies}\")\n            for dep in dependencies:\n                if dep in self.df['file_path'].values:\n                    self.graph.add_edge(file_path, dep, relationship='DEPENDS_ON')\n                    logging.debug(f\"Added edge: {file_path} DEPENDS_ON {dep}\")\n                else:\n                    logging.warning(f\"Dependency {dep} for file {file_path} not found in dataframe.\")\n        logging.info(\"Completed sequential edge building.\")\n    \n    def cache_file_content(self):\n        \"\"\"\n        Caches file content to optimize repeated access.\n        Currently implemented using lru_cache decorator on get_file_content.\n        \"\"\"\n        # This method can be expanded if needed\n        pass\n    \n    def get_graph_dataframes(self):\n        \"\"\"\n        Converts the NetworkX graph into pandas DataFrames for nodes and edges.\n        \n        Returns:\n            nodes_df (pd.DataFrame): DataFrame containing node information.\n            edges_df (pd.DataFrame): DataFrame containing edge information.\n        \"\"\"\n        # Extract nodes with attributes\n        nodes_data = []\n        for node, attrs in self.graph.nodes(data=True):\n            node_entry = {'file_path': node}\n            node_entry.update(attrs)\n            nodes_data.append(node_entry)\n        nodes_df = pd.DataFrame(nodes_data)\n        \n        # Extract edges with attributes\n        edges_data = []\n        for source, target, attrs in self.graph.edges(data=True):\n            edge_entry = {\n                'source': source,\n                'target': target,\n                'relationship': attrs.get('relationship', '')\n            }\n            edges_data.append(edge_entry)\n        edges_df = pd.DataFrame(edges_data)\n        \n        return nodes_df, edges_df\n\n\n# Initialize KnowledgeGraph with sample data\nkg = KnowledgeGraph(repo_dataframe)\n\n# Build edges sequentially\nkg.build_edges_sequential()\n\n# Convert graph to DataFrames\nnodes_df, edges_df = kg.get_graph_dataframes()\n\n# Display the DataFrames\nprint(\"Nodes DataFrame:\")\nprint(nodes_df)\nprint(\"\\nEdges DataFrame:\")\nprint(edges_df)\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n\ndef create_knowledge_graph(nodes_df, edges_df, figsize=(20, 16)):\n    # Create a NetworkX graph\n    G = nx.DiGraph()\n    \n    # Add nodes with category as an attribute\n    for _, row in nodes_df.iterrows():\n        G.add_node(row['file_path'], category=row['category'])\n    \n    # Add edges with relationship as an attribute\n    for _, row in edges_df.iterrows():\n        G.add_edge(row['source'], row['target'], relationship=row['relationship'])\n    \n    # Define a visually distinct color map for different categories\n    categories = sorted(nodes_df['category'].unique())\n    color_map = plt.cm.Set3(np.linspace(0, 1, len(categories)))\n    category_colors = {category: color_map[i] for i, category in enumerate(categories)}\n    \n    # Assign colors to nodes based on their category\n    node_colors = [category_colors[G.nodes[node]['category']] for node in G.nodes()]\n    \n    # Create figure and axis\n    fig, ax = plt.subplots(figsize=figsize)\n    \n    # Use a force-directed layout with optimized parameters for spacing\n    pos = nx.spring_layout(\n        G,\n        k=1.5/np.sqrt(len(G.nodes())),  # Optimal distance between nodes\n        iterations=50,  # More iterations for better convergence\n        seed=42  # For reproducibility\n    )\n    \n    # Draw nodes with enhanced visibility\n    nodes = nx.draw_networkx_nodes(\n        G, pos,\n        node_color=node_colors,\n        node_size=2000,  # Larger nodes\n        alpha=0.7,\n        edgecolors='white',  # White border for better contrast\n        linewidths=2\n    )\n    \n    # Draw edges with improved styling\n    edges = nx.draw_networkx_edges(\n        G, pos,\n        edge_color='gray',\n        arrowsize=20,\n        arrowstyle='->',\n        width=2,\n        alpha=0.6,\n        connectionstyle='arc3,rad=0.2'  # Curved edges for better visibility\n    )\n    \n    # Add labels with improved readability\n    labels = nx.draw_networkx_labels(\n        G, pos,\n        font_size=10,\n        font_weight='bold',\n        font_family='sans-serif',\n        bbox=dict(facecolor='white', edgecolor='none', alpha=0.7, pad=4.0)\n    )\n    \n    # Create a custom legend\n    legend_elements = [Patch(facecolor=color, label=cat, alpha=0.7)\n                      for cat, color in category_colors.items()]\n    ax.legend(\n        handles=legend_elements,\n        title='Categories',\n        title_fontsize=12,\n        fontsize=10,\n        loc='center left',\n        bbox_to_anchor=(1, 0.5),\n        frameon=True,\n        facecolor='white',\n        edgecolor='gray'\n    )\n    \n    # Add title and styling\n    plt.title(\n        \"Knowledge Graph Visualization\",\n        pad=20,\n        fontsize=16,\n        fontweight='bold'\n    )\n    \n    # Remove axes and add padding\n    plt.axis('off')\n    plt.tight_layout(pad=2.0)\n    \n    return fig, ax\n\n# Example usage:\nfig, ax = create_knowledge_graph(nodes_df, edges_df, figsize=(34, 30))\nplt.show()\n","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"nodes_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"edges_df","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}