diff --git a/.github/workflows/lint_python.yml b/.github/workflows/lint_python.yml deleted file mode 100644 index 507beee3..00000000 --- a/.github/workflows/lint_python.yml +++ /dev/null @@ -1,18 +0,0 @@ -name: lint_python -on: - pull_request: - push: - # branches: [master] -jobs: - lint_python: - runs-on: ubuntu-latest - steps: - - uses: actions/checkout@master - - uses: actions/setup-python@master - - run: pip install black codespell flake8 isort pytest - - run: black . --diff || true - - run: codespell --quiet-level=2 || true # --ignore-words-list="" --skip="" - - run: flake8 . --count --select=E9,F63,F7,F82 --show-source --statistics - - run: isort --recursive . || true - - run: pip install -r requirements.txt || true - - run: pytest . || true diff --git a/.gitignore b/.gitignore index c7126040..40202465 100644 --- a/.gitignore +++ b/.gitignore @@ -1 +1,2 @@ questions_to_do.txt +.DS_Store diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md index be0d596c..7fb881f7 100644 --- a/CONTRIBUTING.md +++ b/CONTRIBUTING.md @@ -2,7 +2,7 @@ This is a project that is for the community and its true essence is only possible when it is community driven. -Please feel free to contribute to this project and be a part of this. Anything from raising issues to adding new features or even a typo in documentations, all are welcome. Please report issues here [https://github.com/prabhupant/python-ds/issues](https://github.com/prabhupant/python-ds/issues). It is also recommended to go through the Contributions Best Practices below to help organize contributions :smiley: +Please feel free to contribute to this project and be a part of this. Anything from raising issues to adding new features or even a typo in documentations, all are welcome. Please report issues here [https://github.com/prabhupant/python-ds/issues](https://github.com/prabhupant/python-ds/issues). It is also recommended to go through the Contributions Best Practices below to help organize contributions ## Contributions Best Practices @@ -18,7 +18,7 @@ Please feel free to contribute to this project and be a part of this. Anything f ### Issues -Feel free to open up any issue in the repository, whether it is about code improvements or bug fixes or documentation. You can also use any label you want to associate with the issue. Please provide a clear description of the issue while filing it :smiley: +Feel free to open up any issue in the repository, whether it is about code improvements or bug fixes or documentation. You can also use any label you want to associate with the issue. Please provide a clear description of the issue while filing it. ### Code Styling Guide @@ -26,9 +26,9 @@ Python follows a Pep8 styling. Styling the code according to it makes it univers ### Questions -While there are infinite number of data structure and algorithm questions, it is impossible to collect all of them here. So please add only those questions that are either unique or tricky or have some mind blowing approach to solve them. Also, try not to file a PR for a question which is already present in the repo. Interesting questions and implementations are always welcomed :wink: +While there are infinite number of data structure and algorithm questions, it is impossible to collect all of them here. So please add only those questions that are either unique or tricky or have some mind blowing approach to solve them. Also, try not to file a PR for a question which is already present in the repo. Interesting questions and implementations are always welcomed. -For filing a PR for a new question, please open a issue first for the same and then reference it in the PR. For example, let's say you want to add a new question called "find max element in array". So first head over to the issues tab and create a new issue called "New question: find max element in array". Then if you want to work on it, please mention it in the description. After you are done writing the code and ready to file a PR, refer to the issue number in the PR description. Let's say the issue number was #23. So in the PR description, it should be "Fixes #23". That's it! :smiley: +For filing a PR for a new question, please open a issue first for the same and then reference it in the PR. For example, let's say you want to add a new question called "find max element in array". So first head over to the issues tab and create a new issue called "New question: find max element in array". Then if you want to work on it, please mention it in the description. After you are done writing the code and ready to file a PR, refer to the issue number in the PR description. Let's say the issue number was #23. So in the PR description, it should be "Fixes #23". That's it! ### Bookmarks @@ -36,13 +36,13 @@ The Bookmarks directory is for storing awesome links to articles, videos, slides * Don't add a link to a MOOC or course (like DS Algo MIT lectures) -* Ask yourself "Is the link interesting enough that someone will really love it?" or "Is it something that an interviewer might ask in the interview to test you?" or "Is it some common or amazing thing people don't have enough knowledge about?". If yes to any of them, then sure, quickly file the PR :wink: +* Ask yourself "Is the link interesting enough that someone will really love it?" or "Is it something that an interviewer might ask in the interview to test you?" or "Is it some common or amazing thing people don't have enough knowledge about?". If yes to any of them, then sure, quickly file the PR. * If an article has more than one part, be sure to refer to the first only in the PR. ## How To Contribute? -You can add anything to the repo as long as it is related to programming and increasing knowledge :smiley: +You can add anything to the repo as long as it is related to programming and increasing knowledge. 0. Firstly, fork the repo to your GitHub account. Just to get familiar with some terms, this will be the `origin`. To state `origin` and `upstream` explicitly, @@ -80,13 +80,13 @@ $ git push origin max-number-in-array 6. Write a good brief description and refer to the issue (if any) and submit the PR. -7. Now wait for the approval :smiley: +7. Now wait for the approval. ## Join The Development Before you join the development, please fork and clone the repository to your local machine and explore it. Don't worry nothing will happen, atmost some code might not work :wink: -Feel free to contribute and be a part of this endeavour :beers: +Feel free to contribute and be a part of this endeavour. diff --git a/README.md b/README.md index 14069301..7c05260f 100644 --- a/README.md +++ b/README.md @@ -1,14 +1,16 @@ -![logo](logo/logo.png) - # Python Data Structures and Algorithms -This repository contains data structures and algorithms concepts and questions useful for interviews in Python. +No non-sense solutions to common Data Structure and Algorithm interview questions in Python. Follows a consistent approach throughout problems. + +## Objective + +There are a plenty of resources when it comes to interview preparations on the internet. What prompted me to create this project was the dissimilarity across different approaches and the infused complexity of the code. -## :dart: Objective +Feel free to contribute but please follow the Contributing Guidelines as I want to maintain the uniformity of the implementation of data structures and algorithms. Last time around, people bombarded with me with Pull Requests, Issues and Emails insisting me to merge their changes -The open source community has helped me a lot during my interview preparations and studies while I was in my undergrad. I always wanted to give something back to the community. In my endeavour to contribute something back, I will be uploading data structures and algorithms questions in Python in this repo. Feel free to contribute and get in touch! :smiley: +The open source community has helped me a lot during my interview preparations and studies while I was in my undergrad. I always wanted to give something back to the community. In my endeavour to contribute something back, I will be uploading data structures and algorithms questions in Python in this repo. Feel free to contribute and get in touch! -## :file_folder: Structure of the repository +## Structure of the repository As of now, the repository contains 3 main directories: [**Bookmarks**](bookmarks), [**Data Structures**](data_structures) and [**Algorithms**](algorithms). @@ -27,7 +29,7 @@ Contains all data structure questions categorised into sub-directories like stac ### Algorithms -This directory contains various types of algorithm questions like Dynamic Programming, Sorting, Greedy, etc. The current structure of this directory is like - +This directory contains various types of algorithm questions like Dynamic Programming, Sorting, Greedy, etc. The current structure of this directory is as follows: 1. [Dynamic Programming](algorithms/dynamic_programming) 2. [Graphs](algorithms/graph) @@ -35,6 +37,7 @@ This directory contains various types of algorithm questions like Dynamic Progra 4. [Math](algorithms/math) 5. [Misc](algorithms/miscellaneous) 6. [Sorting](algorithms/sorting) +7. [Bit Manipulation](algorithms/bit_manipulation) ### Bookmarks @@ -49,21 +52,21 @@ You can find useful links in this repository in the different markdown files. Be | Videos | [Click Here](bookmarks/videos.md) | | Misc. | [Click Here](bookmarks/misc.md) | -## :clipboard: Things need to be done +## Things need to be done As you can see, the repo is still in its infancy. Here are some key things in the to-do. 1. Queue questions 2. Algorithms -3. More questions in data structures, especially for graph, circular linked list, tries, heaps and hash. +3. More questions in data structures, especially for graph, circular linked list, trees, heaps and hash. -## :raised_hand: Contributing +## Contributing -Contributions are always welcomed. :smiley: -Feel free to raise new issues, file new PRs and star and fork this repo! :wink: +Contributions are always welcomed. +Feel free to raise new issues, file new PRs. Consider giving it a star and fork this repo! To follow the guidelines, refer to [Contributing.md](CONTRIBUTING.md) -## :page_facing_up: License +## License -[MIT @ Prabhu Pant](LICENSE) +[MIT License](LICENSE) diff --git a/algorithms/__init__.py b/algorithms/__init__.py deleted file mode 100644 index 8b137891..00000000 --- a/algorithms/__init__.py +++ /dev/null @@ -1 +0,0 @@ - diff --git a/algorithms/bit_manipulation/range_sum_set_bits.py b/algorithms/bit_manipulation/range_sum_set_bits.py new file mode 100644 index 00000000..6250acf9 --- /dev/null +++ b/algorithms/bit_manipulation/range_sum_set_bits.py @@ -0,0 +1,19 @@ +# Question: Find the sum of number of set bits in all the numbers in the range [1, n]. + +def countBits(n): + + """ Consider a number x and half of the number (x//2). + The binary representation of x has all the digits as + the binary representation of x//2 followed by an additional + digit at the last position. Therefore, we can find the number + of set bits in x by finding the number of set bits in x//2 + and determining whether the last digit in x is 0 or 1. """ + + res = [0] * (n+1) + for i in range(1, n+1): + res[i] = res[i//2] + (i & 1) + return sum(res) + + +# Extension: Find the sum of number of set bits in all the numbers in the range [m, n]. +# Answer: In countBits(m, n), return sum(res) - sum(res[:m]) \ No newline at end of file diff --git a/algorithms/bit_manipulation/sum_of_two_integers.py b/algorithms/bit_manipulation/sum_of_two_integers.py new file mode 100644 index 00000000..c8203c5a --- /dev/null +++ b/algorithms/bit_manipulation/sum_of_two_integers.py @@ -0,0 +1,22 @@ +# Question: Calculate the sum of two integers a and b but without the +# use of the operators + and -. + +#Solution +def getSum(a, b): + """ + :type a: int + :type b: int + :rtype: int + """ + + mask = 0xffffffff + diff = 0 + carry = 0 + while b & mask: + diff = a ^ b + carry = trunc(a & b) + carry = carry << 1 + a = diff + b = carry + if b > 0: return (a & mask) + else: return a \ No newline at end of file diff --git a/algorithms/dynamic_programming/coin_change.py b/algorithms/dynamic_programming/coin_change.py index 96b57648..7b9cbe26 100644 --- a/algorithms/dynamic_programming/coin_change.py +++ b/algorithms/dynamic_programming/coin_change.py @@ -1,16 +1,33 @@ -# Concept is almost same as 01 Knapsack Problem -def min_coin(coins, total): +def min_coins(coins, total): cols = total + 1 - rows = len(coins) - t = [[[0] if col == 0 else float('inf') for col in range(cols)] for i in range(rows)] + min_coins = [float('inf')] * (total + 1) + coins_used = [-1] * (total + 1) - for i in range(rows): - for j in range(1, cols): - if j < coins[i]: - t[i][j] = t[i-1][j] - else: - t[i][j] = min(t[i-1][j], 1 + t[i][j-coins[i]]) + min_coins[0] = 0 # to form 0, we need 0 coins - return t[rows-1][cols-1] + for i in range(0, len(coins)): + for j in range(1, len(min_coins)): + if coins[i] > j: # if the coin value is more than j (curr total), ignore it + continue + + if (1 + min_coins[j - coins[i]]) < min_coins[j]: + min_coins[j] = 1 + min_coins[j - coins[i]] + coins_used[j] = i + + # finding which coins were used + picked_coins = [] + while total > 0: + index_of_coin_used = coins_used[total] + coin = coins[index_of_coin_used] + picked_coins.append(coin) + total -= coin + + print('Min coins needed - ', min_coins[-1]) + print('Coins used - ', picked_coins) + +total = 11 +coins = [9, 6, 5, 1] + +min_coins(coins, total) diff --git a/algorithms/dynamic_programming/hamilton_cycle.py b/algorithms/dynamic_programming/hamilton_cycle.py new file mode 100644 index 00000000..21e2b4ac --- /dev/null +++ b/algorithms/dynamic_programming/hamilton_cycle.py @@ -0,0 +1,27 @@ +import functools + +def hamilton_cycle(graph, n): + height = 1 << n + + dp = [[False for _ in range(n)] for _ in range(height)] + for i in range(n): + dp[1 << i][i] = True + + for i in range(height): + ones, zeros = [], [] + for pos in range(n): + if (1 << pos) & i: + ones.append(pos) + else: + zeros.append(pos) + + for o in ones: + if not dp[i][o]: + continue + + for z in zeros: + if graph[o][z]: + new_val = i + (1 << z) + dp[new_val][z] = True + + return functools.reduce(lambda a, b: a or b, dp[height - 1]) \ No newline at end of file diff --git a/algorithms/dynamic_programming/longest_consecutive_subsequence.py b/algorithms/dynamic_programming/longest_consecutive_subsequence.py new file mode 100644 index 00000000..742ef4b7 --- /dev/null +++ b/algorithms/dynamic_programming/longest_consecutive_subsequence.py @@ -0,0 +1,45 @@ +""" +Given an array of integers, find the length of the longest sub-sequence +such that elements in the subsequence are consecutive integers, the +consecutive numbers can be in any order. + +The idea is to store all the elements in a set first. Then as we are iterating +over the array, we check two things - +1. a number x can be a starting number in a sequence if x-1 is not present in the +set. If this is the case, create a loop and check how many elements from x to x+j are +in the set +2. if x -1 is there in the set, do nothing as this number is not a starting element +and must have been considered in a different sequence +""" + +def find_seq(arr, n): + s = set() + + for num in arr: + s.add(num) + + ans = 0 + elements = [] + + for i in range(n): + temp = [] + + if arr[i] - 1 not in s: + j = arr[i] + + while j in s: + temp.append(j) + j += 1 + + if j - arr[i] > ans: + ans = j - arr[i] + elements = temp.copy() + + return ans, elements + + +arr = [36, 41, 56, 35, 44, 33, 34, 92, 43, 32, 42] + +ans, elements = find_seq(arr, len(arr)) +print('Length - ', ans) +print('Elements - ', elements) diff --git a/algorithms/dynamic_programming/longest_increasing_consecutive_subsequence.py b/algorithms/dynamic_programming/longest_increasing_consecutive_subsequence.py new file mode 100644 index 00000000..c272820c --- /dev/null +++ b/algorithms/dynamic_programming/longest_increasing_consecutive_subsequence.py @@ -0,0 +1,26 @@ +""" +Find the longest increasing consecutive subsequence in an array + +Idea - + +create a dictionary 'seq' and start iterating over the array + +1. if arr[i] - 1 exists in the array, length = length + seq[arr[i] - 1] +2. else, seq[i] = 1 +""" + +def find_seq(arr): + seq = {} + count = 0 + + for num in arr: + if num - 1 in seq: + seq[num] = seq[num - 1] + 1 + count = max(count, seq[num]) + else: + seq[num] = 1 + + return count + +arr = [6, 7, 8, 3, 4, 5, 9, 10] +print(find_seq(arr)) diff --git a/algorithms/dynamic_programming/longest_increasing_subsequence.py b/algorithms/dynamic_programming/longest_increasing_subsequence.py new file mode 100644 index 00000000..8a9fa822 --- /dev/null +++ b/algorithms/dynamic_programming/longest_increasing_subsequence.py @@ -0,0 +1,13 @@ +def LIS(arr): + n = len(arr) + if n == 0: return 0 + res = 1 + dp = [0] * n + dp[0] = 1 + for i in range(1, n): + dp[i] = 1; + for j in range(0 , i): + if arr[i] > arr[j]: + dp[i] = max(dp[i] , dp[j] + 1) + res = max(res , dp[i]) + return res \ No newline at end of file diff --git a/algorithms/dynamic_programming/longest_subarray_sum_divisible_by_k.py b/algorithms/dynamic_programming/longest_subarray_sum_divisible_by_k.py new file mode 100644 index 00000000..4d691f59 --- /dev/null +++ b/algorithms/dynamic_programming/longest_subarray_sum_divisible_by_k.py @@ -0,0 +1,55 @@ +""" +Find the longest subarray in an array whose sum is ` +divisible by k + +source - https://www.geeksforgeeks.org/longest-subarray-sum-divisible-k/ + +The idea is that we create a new array mod_arr where we mod_arr[i] = +sum(arr[0]...arr[i]) % k. So basically this array tells us that upto this +point in the input array, if we take sum of numbers till index i, that sum will +be divisible by k + +We will be creating a hash table for this to store the mod results + +Now, lets say x = sum(arr[0]...arr[i]) % k = mod_arr[i]. If + +1. if we find x == 0, increment length by 1 +2. if x not in hash, create it and store (x, index of x) +3. if x in hash: + this tells us that upto this point, where the remainder of sum of numbers + till this point divided by k is x, that remainder we already saw before as it + exists in the hash. So if we ignore the first dont consider the first occurence of x + and remove that from the sum, then this sum will be divisible by k (because subtracting remainder + from a number makes it divisible). + Now find the max length of such case as + if length = max(length, (i - index(x)) +""" + +def find_length(arr, k): + hash_table = {} + mod_arr = [] + s = 0 + length = 0 + start, end = 0, 0 + + for i in range(0, len(arr)): + s += arr[i] + mod_arr.append(s % k) + + for i in range(0, len(mod_arr)): + if mod_arr[i] == 0: + length += 1 + else: + if mod_arr[i] not in hash_table: + hash_table[mod_arr[i]] = i + else: + if length < (i - mod_arr[i]): + length = i - mod_arr[i] + start = mod_arr[i] + end = i - 1 # i-1 because the current number is not to considered as it makes the sum not divisible by k + + return length, arr[start:end+1] + + +arr = [ 2, 7, 6, 1, 4, 5 ] +print(find_length(arr, 3)) diff --git a/algorithms/dynamic_programming/longest_subarray_with_no_pairsum_divisible_by_k.py b/algorithms/dynamic_programming/longest_subarray_with_no_pairsum_divisible_by_k.py new file mode 100644 index 00000000..844b5611 --- /dev/null +++ b/algorithms/dynamic_programming/longest_subarray_with_no_pairsum_divisible_by_k.py @@ -0,0 +1,54 @@ +""" +Find the longest subarray in the input array such that the pairwise sum of +the elements of this subarray is not divisible by K + +The idea is - +How can we tell that two numbers x and y will make a pairsum that will be +divisible by K just by looking at their remainders? There can be two conditions + +1. It will be only possible if the sum of the remainders when x and y are +divided by K is equal to K. As the sum of the remainders cannot exceed K +so if it reaches K then it means that the sum of those numbers will also be +divisible by K +0 < (X%K) + (Y%K) <= K + +2. If arr[i] % k == 0 and there is also an element j such that arr[j] % k == 0 +and 0 exists in the hash (i.e hash[j] = True) +""" + +def find_subarray(arr, k): + """ + True means divisible by k + """ + start, end = 0, 0 + max_start, max_end = 0, 0 + + n = len(arr) + mod_arr = [0] * n + + mod_arr[arr[0] % k] = mod_arr[arr[0] % k] + 1 + + for i in range(1, n): + mod = arr[i] % k + + while (mod_arr[k - mod] != 0) or (mod == 0 and mod_arr[mod] != 0): + mod_arr[arr[start] % k] = mod_arr[arr[start] % k] - 1 + start += 1 + + mod_arr[mod] = mod_arr[mod] + 1 + end += 1 + + if (end - start) > (max_end - max_start): + max_end = end + max_start = start + + print(f'Max size is {max_end - max_start}') + + for i in (max_start, max_end + 1): + print(arr[i], end=" ") + + +arr = [3, 7, 1, 9, 2] +k = 3 +find_subarray(arr, k) + diff --git a/algorithms/dynamic_programming/partition_sum.py b/algorithms/dynamic_programming/partition_sum.py new file mode 100644 index 00000000..c07b79f6 --- /dev/null +++ b/algorithms/dynamic_programming/partition_sum.py @@ -0,0 +1,46 @@ +# A Dynamic Programming based +# Python3 program to partition problem + +# Returns true if arr[] can be partitioned +# in two subsets of equal sum, otherwise false +def find_partiion(arr, n) : + sum = 0 + + # Calculate sum of all elements + for i in range(n) : + sum += arr[i] + if (sum % 2 != 0) : + return 0 + part = [0] * ((sum // 2) + 1) + + # Initialize the part array as 0 + for i in range((sum // 2) + 1) : + part[i] = 0 + + # Fill the partition table in bottom up manner + for i in range(n) : + + # the element to be included + # in the sum cannot be + # greater than the sum + for j in range(sum // 2, arr[i] - 1, -1) : + + # check if sum - arr[i] + # could be formed + # from a subset + # using elements + # before index i + if (part[j - arr[i]] == 1 or j == arr[i]) : + part[j] = 1 + + return part[sum // 2] + +# Drive code +arr = [ 1, 3, 3, 2, 3, 2 ] +n = len(arr) + +# Function call +if (find_partiion(arr, n) == 1) : + print("Can be divided into two subsets of equal sum") +else : + print("Can not be divided into two subsets of equal sum") diff --git a/algorithms/dynamic_programming/prefix_function.py b/algorithms/dynamic_programming/prefix_function.py new file mode 100644 index 00000000..c2cd4385 --- /dev/null +++ b/algorithms/dynamic_programming/prefix_function.py @@ -0,0 +1,19 @@ + +def prefix_function(s: str) -> [int]: + """ + The prefix function for string s is defined as an array pi of length n, + where pi[i] is the length of the longest proper prefix of the substring + s[0...i] which is also a suffix of this substring. A proper prefix of a + string is a prefix that is not equal to the string itself. + By definition, pi[0] = 0. + """ + n = len(s) + pi = [0] * n + for i in range(1, n): + j = pi[i - 1] + while (j > 0) and (s[i] != s[j]): + j = pi[j - 1] + if s[i] == s[j]: + j += 1 + pi[i] = j + return pi diff --git a/algorithms/dynamic_programming/prefix_sums.py b/algorithms/dynamic_programming/prefix_sums.py new file mode 100644 index 00000000..2a0b70db --- /dev/null +++ b/algorithms/dynamic_programming/prefix_sums.py @@ -0,0 +1,11 @@ +def prefix_sums(ls: [int]) -> [int]: + """ + Returns list of prefix sums for given list of integers. + """ + n = len(ls) + total = 0 + sums = [0] * n + for i in range(n): + total += ls[i] + sums[i] = total + return sums diff --git a/algorithms/graph/bfs.py b/algorithms/graph/bfs.py deleted file mode 100644 index 6c64f75f..00000000 --- a/algorithms/graph/bfs.py +++ /dev/null @@ -1,49 +0,0 @@ -# Python3 Program to print BFS traversal -# from a given source vertex. BFS(int s) -# traverses vertices reachable from s. -from collections import defaultdict - -# This class represents a directed graph -# using adjacency list representation -class Graph: - - # Constructor - def __init__(self): - - # default dictionary to store graph - self.graph = defaultdict(list) - - # function to add an edge to graph - def addEdge(self,u,v): - self.graph[u].append(v) - - # Function to print a BFS of graph - def BFS(self, s): - - # Mark all the vertices as not visited - visited = [False] * (len(self.graph)) - - # Create a queue for BFS - queue = [] - - # Mark the source node as - # visited and enqueue it - queue.append(s) - visited[s] = True - - while queue: - - # Dequeue a vertex from - # queue and print it - s = queue.pop(0) - print (s, end = " ") - - # Get all adjacent vertices of the - # dequeued vertex s. If a adjacent - # has not been visited, then mark it - # visited and enqueue it - for i in self.graph[s]: - if visited[i] == False: - queue.append(i) - visited[i] = True - \ No newline at end of file diff --git a/algorithms/graph/dfs.py b/algorithms/graph/dfs.py deleted file mode 100644 index 4a452536..00000000 --- a/algorithms/graph/dfs.py +++ /dev/null @@ -1,46 +0,0 @@ -# Python program to print DFS traversal for complete graph -from __future__ import print_function -from collections import defaultdict - -# This class represents a directed graph using adjacency -# list representation -class Graph: - - # Constructor - def __init__(self): - - # default dictionary to store graph - self.graph = defaultdict(list) - - # function to add an edge to graph - def addEdge(self,u,v): - self.graph[u].append(v) - - # A function used by DFS - def DFSUtil(self, v, visited): - - # Mark the current node as visited and print it - visited[v]= True - print(v, end=" ") - - # Recur for all the vertices adjacent to - # this vertex - for i in self.graph[v]: - if visited[i] == False: - self.DFSUtil(i, visited) - - - # The function to do DFS traversal. It uses - # recursive DFSUtil() - def DFS(self): - V = len(self.graph) #total vertices - - # Mark all the vertices as not visited - visited =[False]*(V) - - # Call the recursive helper function to print - # DFS traversal starting from all vertices one - # by one - for i in range(V): - if visited[i] == False: - self.DFSUtil(i, visited) diff --git a/algorithms/graph/find_all_paths.py b/algorithms/graph/find_all_paths.py deleted file mode 100644 index 9152bf2a..00000000 --- a/algorithms/graph/find_all_paths.py +++ /dev/null @@ -1,32 +0,0 @@ -''' -find all the possible paths in a directed cyclic graph from -a start point to a end point. -''' - - - - -def find_all_paths(graph, start, end, path=[]): - path = path + [start] - if start == end: - return [path] - if start not in graph.keys(): - return [] - paths = [] - for node in graph[start]: - if node not in path: #to prevent cyclic rotations - newpaths = find_all_paths(graph, node, end, path) - #print(newpaths) - for newpath in newpaths: - paths.append(newpath) - return paths - - -graph={1:[2,4], - 2:[3], - 4:[5], - 3:[5] - } - -for i in graph.keys(): - print(i,'to 5',find_all_paths(graph,i,5)) diff --git a/algorithms/graph/index.md b/algorithms/graph/index.md deleted file mode 100644 index e5197adc..00000000 --- a/algorithms/graph/index.md +++ /dev/null @@ -1,4 +0,0 @@ -# Index of graph - -* dfs.py -* bfs.py diff --git a/algorithms/graph/topological_sort.py b/algorithms/graph/topological_sort.py deleted file mode 100644 index ea214d0d..00000000 --- a/algorithms/graph/topological_sort.py +++ /dev/null @@ -1,33 +0,0 @@ -from collections import defaultdict - - -def topological_sort(graph: dict) -> list: - """Provides the topologically sorted nodes of a graph in a list. Takes input as a dictionary, - where the key is a node and the value is a list of the nodes that the key is a source node for.""" - - # Keeps track of the "degree" of a node; once this reaches 0, we push it onto the output. - leading_in = defaultdict(lambda: 0) - - for key, values in graph.items(): - if key not in leading_in.keys(): - leading_in[key] = 0 - for node in values: - leading_in[node] += 1 - - queue = [] - output = [] - - for node, degree in leading_in.items(): - if degree == 0: - queue.append(node) - output.append(node) - - while queue: - node = queue.pop(0) - for destination in graph.get(node, []): - leading_in[destination] -= 1 - if leading_in[destination] == 0: - queue.append(destination) - output.append(destination) - - return output diff --git a/algorithms/greedy/__init__.py b/algorithms/greedy/__init__.py deleted file mode 100644 index 8b137891..00000000 --- a/algorithms/greedy/__init__.py +++ /dev/null @@ -1 +0,0 @@ - diff --git a/algorithms/greedy/activity_selection.py b/algorithms/greedy/activity_selection.py index 8dbd822f..374ad563 100644 --- a/algorithms/greedy/activity_selection.py +++ b/algorithms/greedy/activity_selection.py @@ -4,15 +4,30 @@ #s[]--> An array that contains start time of all activities #f[] --> An array that contains finish time of all activities -def print_max_activities(s, f): - n = len(f) - # the first activity is always selected +def find_activities(arr): + n = len(arr) + selected = [] + + arr.sort(key = lambda x: x[1]) + i = 0 - print(i, end=' ') - # for the rest - for j in range(n): - if s[j] >= f[i]: - print(j, end=' ') + # since it is a greedy algorithm, the first acitivity is always + # selected because it is the most optimal choice at that point + selected.append(arr[i]) + + for j in range(1, n): + start_time_next_activity = arr[j][0] + end_time_prev_activity = arr[i][1] + + if start_time_next_activity >= end_time_prev_activity: + selected.append(arr[j]) i = j + + return selected + + +arr = [[5, 9], [1, 2], [3, 4], [0, 6],[5, 7], [8, 9]] +print(find_activities(arr)) + diff --git a/algorithms/greedy/cost_of_tiles.py b/algorithms/greedy/cost_of_tiles.py new file mode 100644 index 00000000..05fde8da --- /dev/null +++ b/algorithms/greedy/cost_of_tiles.py @@ -0,0 +1,45 @@ +""" +Find the min cost of tiles to cover a floor. +Floor is represented by 2D array where - +* = tile already placed +. = no tile + +tiles available are 1*1 and 1*2 and their costs +are A and B + +Source - https://www.geeksforgeeks.org/minimize-cost-to-cover-floor-using-tiles-of-dimensions-11-and-12/ +""" + +def cost(arr, A, B): + n = len(arr) + m = len(arr[0]) + + ans = 0 + + for i in range(n): + j = 0 + + while j < m: + if arr[i][j] == '*': # tile is already there + j += 1 + continue + + if j == m - 1: # if j is pointing to last tile, you can use only 1*1 tile + ans += A + else: + if arr[i][j+1] == '.': + ans += min(2 * A, B) + j += 1 + else: + ans += A + + j += 1 + + print('Cost of tiling is - ', ans) + +arr = [ [ '.', '.', '*' ], + [ '.', '*', '*' ] ] + +A, B = 2, 10 + +cost(arr, A, B) diff --git a/algorithms/greedy/index.md b/algorithms/greedy/index.md deleted file mode 100644 index d9b79484..00000000 --- a/algorithms/greedy/index.md +++ /dev/null @@ -1 +0,0 @@ -# Index of Greedy \ No newline at end of file diff --git a/algorithms/greedy/min_platforms.py b/algorithms/greedy/min_platforms.py new file mode 100644 index 00000000..fd372f0a --- /dev/null +++ b/algorithms/greedy/min_platforms.py @@ -0,0 +1,37 @@ +""" +Given the arrival and departure times of buses at a station +find the min number of platforms that must be there +""" + + +def find_platforms(arrival, departure): + n = len(arrival) + + arrival.sort() + departure.sort() + + i = 1 + j = 0 + + ans = 1 # atleast one platform is required + plat = 1 + + while i < n and j < n: + if arrival[i] <= departure[j]: + plat += 1 + i += 1 + + elif arrival[i] > departure[j]: + plat -= 1 + j += 1 + + ans = max(ans, plat) + + + return ans + + +arr = [900, 940, 950, 1100, 1500, 1800] +dep = [910, 1200, 1120, 1130, 1900, 2000] + +print(find_platforms(arr, dep)) diff --git a/algorithms/math/divisors.py b/algorithms/math/divisors.py new file mode 100644 index 00000000..ea923837 --- /dev/null +++ b/algorithms/math/divisors.py @@ -0,0 +1,63 @@ +import math + +class divisors: + + def findAllDivisors(self, n): + result = list() + + for i in range(1, n+1): + if (n%i == 0): + result.append(i) + + return result + + def divisorsCount(self, n): + divs = self.findAllDivisors(n) + return len(divs) + + def oddFactors(self, n): + result = list() + + for i in range(1, n+1): + if (n%i == 0 and i%2==1): + result.append(i) + + return result + + def oddFactorsSum(self, n): + divs = self.evenFactors(n) + return sum(divs) + + def evenFactors(self, n): + result = list() + + for i in range(1, n+1): + if (n%i == 0 and i%2==0): + result.append(i) + + return result + + def evenFactorsSum(self, n): + divs = self.evenFactors(n) + return sum(divs) + + def primeFactors(self, n): + result = list() + + while n % 2 == 0: + result.append(2) + n = n / 2 + + for i in range(3,int(math.sqrt(n))+1,2): + while n % i== 0: + result.append(i) + n = n / i + + if n > 2: + result.append(n) + + return result + + def primeFactorsSum(self, n): + divs = self.primeFactors(n) + return sum(divs) \ No newline at end of file diff --git a/algorithms/math/factorial_iterative.py b/algorithms/math/factorial_iterative.py new file mode 100644 index 00000000..d722ec92 --- /dev/null +++ b/algorithms/math/factorial_iterative.py @@ -0,0 +1,18 @@ +#Calculate factorial of a given number using iterative method. + +def factorial(number): + + answer = 1 + + if number == 0: + return 1 + else: + for num in range(1, number+1): + answer = answer * num + return answer + +if __name__ == '__main__': + + enter_number = int(input("Enter a number whose factorial is required : ")) + result = factorial(enter_number) + print(result) \ No newline at end of file diff --git a/algorithms/math/factorial_recursive.py b/algorithms/math/factorial_recursive.py new file mode 100644 index 00000000..ed5f3b38 --- /dev/null +++ b/algorithms/math/factorial_recursive.py @@ -0,0 +1,16 @@ +#Calculate factorial of a given number using recursive method. + +def factorial(number): + + if number == 0 or number == 1: + return 1 + + answer = number * factorial(number - 1) + + return answer + +if __name__ == '__main__': + + enter_number = int(input("Enter a number whose factorial is required : ")) + result = factorial(enter_number) + print(result) \ No newline at end of file diff --git a/algorithms/math/greatest_common_divisor.py b/algorithms/math/greatest_common_divisor.py index d00efe53..53e70f1d 100644 --- a/algorithms/math/greatest_common_divisor.py +++ b/algorithms/math/greatest_common_divisor.py @@ -26,4 +26,10 @@ def gcd(x, y): if y == 0: return x - return gcd(y, x % y) \ No newline at end of file + return gcd(y, x % y) + +# Iterative optimized +def gcd(x, y): + while y != 0: + x, y = y, x % y + return x diff --git a/algorithms/math/index.md b/algorithms/math/index.md index fc7b2649..d6248d88 100644 --- a/algorithms/math/index.md +++ b/algorithms/math/index.md @@ -6,3 +6,5 @@ * [Sieve of Erastothenes](sieve_of_eratosthenes.py) * [Perfect Square](perfect_square.py) * [Number Convertion](number_convertion.py) +* [Iterative Factorial](factorial_iterative.py) +* [Recursive Factorial](factorial_recursive.py) diff --git a/algorithms/math/power_of_two.py b/algorithms/math/power_of_two.py new file mode 100644 index 00000000..6fcfdb22 --- /dev/null +++ b/algorithms/math/power_of_two.py @@ -0,0 +1,34 @@ +""" + This simple code is to check if a given number is a poer of two or not. + Method: + if a number is a power of two, then its binary representation is (2^k). + ex: 4 --> 2^2 --> 100 --> k = 2 + 8 --> 2^3 --> 1000 --> k = 3 + 16--> 2^4 --> 10000 --> k = 4 + + assume that n is a power of two, then (n-1)&(n) will be zero; + ex: + n = 8 --> (1000) , n-1 = 7 ---> (0111) + performing bit (and) operation between both, then n&(n-1) = 0000 + + n = 12 --> (1100) , n-1 = 11 ---> (1011) + performing bit (and) operation between both, then n&(n-1) = 1000 + + Conclusion: + The result of the above bit "and" operation will be zero, ONLY if the given number is a pwoer of two. + + NOTE: + - Since Python considers 0 as "false", then we are gonna return the inversion of the result; i.e. return not(n&(n-1). + - BUT If n = 0, the result will be zero indicating that 0 is a power of 2, wich is not true. + So to fix that, we are going to perform an extra logical "and" operation with the oreginal number. +""" + + +def pow_of_two(n): + return(n and (not(n&(n-1)))) + +for i in range(20): + if pow_of_two(i): + print(f"{i} is a power of 2.") + else: + print(f"{i} is NOT a power of 2.") diff --git a/algorithms/math/recursive_fibonacci.py b/algorithms/math/recursive_fibonacci.py index 4094e6b9..eb319cd9 100644 --- a/algorithms/math/recursive_fibonacci.py +++ b/algorithms/math/recursive_fibonacci.py @@ -1,5 +1,6 @@ """ - Recursivly compute the Fibonacci sequence + Recursivly compute the Fibonacci sequence using two different methods + rec_fib(n) requires O(Fibo(n)) operations, whereas binary_rec_fib(n) requires less than O(n) """ def rec_fib(n): @@ -10,10 +11,23 @@ def rec_fib(n): else: return rec_fib(n-1)+rec_fib(n-2) +def binary_rec_fib(n): + if n == 2 or n == 1: + return 1 + elif n == 0: + return 0 + else: + # This recursive step takes advantage of the following two properties of the fibonacci numbers: + # Fibo(2n) = Fibo(n+1)^2 + Fibo(n)^2 + # Fibo(2n+1) = Fibo(n+1)^2 - Fibo(n-1)^2 + sgn = n % 2 + return binary_rec_fib((n-sgn)/2 + 1)**2 - ((-1)**sgn) * binary_rec_fib((n+sgn)/2 - 1)**2 + def main(): + times = [] n : int = int(input("n := ")) for i in range(0, n): - print(rec_fib(i)) + print(binary_rec_fib(i)) if __name__ == "__main__": main() diff --git a/algorithms/miscellaneous/markov.py b/algorithms/miscellaneous/markov.py new file mode 100755 index 00000000..bffe7675 --- /dev/null +++ b/algorithms/miscellaneous/markov.py @@ -0,0 +1,48 @@ +import codecs +import random + +MAX_LETTERS = 2000 + +def readFile(f, mp, k): + seed = '' + mostFreqSeed = seed + mostFreq = 1 + for line in f: + for ch in line: + seed, mostFreq, mostFreqSeed = processSeed(seed, mostFreqSeed, mostFreq, ch, k, mp) + return mostFreqSeed + +def processSeed(seed, mostFreqSeed, mostFreq, ch, k, mp): + seed += ch + if len(seed) == k+1: + oldSeed = seed[:-1] + mp.setdefault(oldSeed, []).append(ch) + if mostFreq < len(mp[oldSeed]): + mostFreq, mostFreqSeed = len(mp[oldSeed]), oldSeed + seed = seed[1:] + return seed, mostFreq, mostFreqSeed + +def generateText(mp, mostFreqSeed): + text, curSeed = mostFreqSeed, mostFreqSeed + while (len(text) < MAX_LETTERS): + ch = random.choice(mp[curSeed]) + text, curSeed = text+ch, curSeed+ch + curSeed = curSeed[1:] + return text + +def main(): + fileName = input("Enter the file name: ") + ".txt" + f = codecs.open(fileName, encoding='utf-8') + k = int(input("Enter the Markov order [1-10]: ")) + assert (k >= 1 and k <= 10) + mp = {} + mostFreqSeed = readFile(f, mp, k) + f.close() + text = generateText(mp, mostFreqSeed) + print(text) + result = codecs.open("result.txt", "w", encoding='utf-8') + result.write(text) + result.close() + +if __name__ == "__main__": + main() \ No newline at end of file diff --git a/algorithms/sorting/bubble_sort.py b/algorithms/sorting/bubble_sort.py index 85668cac..1c4a24cc 100644 --- a/algorithms/sorting/bubble_sort.py +++ b/algorithms/sorting/bubble_sort.py @@ -1,12 +1,30 @@ -''' +""" Bubble Sort worst time complexity occurs when array is reverse sorted - O(n^2) Best time scenario is when array is already sorted - O(n) -''' +""" def bubble_sort(array): n = len(array) - for i in range(n): for j in range(0, n-i-1): if array[j] > array[j+1]: - array[j], array[j+1] = array[j+1], array[j] \ No newline at end of file + array[j], array[j+1] = array[j+1], array[j] + return array + + +def bubble_sort_optimized(array): + """ + Optimizes on bubble sort by taking care of already swapped cases + Reference - https://github.com/prabhupant/python-ds/pull/346 + """ + has_swapped = True + + num_of_iterations = 0 + + while has_swapped: + has_swapped = False + for i in range(len(array) - num_of_iterations - 1): + if array[i] > array[i + 1]: + array[i], array[i + 1] = array[i + 1], array[i] + has_swapped = True + num_of_iterations += 1 diff --git a/algorithms/sorting/counting_sort.py b/algorithms/sorting/counting_sort.py new file mode 100644 index 00000000..d9f359c8 --- /dev/null +++ b/algorithms/sorting/counting_sort.py @@ -0,0 +1,35 @@ +""" +High level description: +Counting sort is a sorting technique based on keys between a specific range, +with efective performance on the predetermined range size of the values. +It works by counting the number of objects having distinct key values (kind of hashing). +Then doing some arithmetic to calculate the position of each object in the output sequence. + +Time complexity: +O(n+k) where n is the number of elements in input array and k is the range of input. + +Auxiliary Space: O(n+k) +""" + +def counting_sort(arr): + # Find min and max values + min_value = min(arr) + max_value = max(arr) + + # Count number appearances in the array + counting_arr = [0]*(max_value-min_value+1) + for num in arr: + counting_arr[num-min_value] += 1 + + # Rearrange sequence in the array + index = 0 + for i, count in enumerate(counting_arr): + for _ in range(count): + arr[index] = min_value + i + index += 1 + +test_array = [3, 3, 2, 6, 4, 7, 9, 7, 8] + +counting_sort(test_array) + +print(test_array) diff --git a/algorithms/sorting/gnome_sort.py b/algorithms/sorting/gnome_sort.py new file mode 100644 index 00000000..444b127d --- /dev/null +++ b/algorithms/sorting/gnome_sort.py @@ -0,0 +1,38 @@ +''' +Gnome sort is not best sorting algorithms but sure it takes its pride. +It has time O(n^2) +''' + + +def gnome_sort(arr): + """ + Examples: + >>> gnome_sort([0, 5, 2, 3, 2]) + [0, 2, 2, 3, 5] + + >>> gnome_sort([]) + [] + >>> gnome_sort([-2, -45, -5]) + [-45, -5, -2] + """ + + # first case + size = len(arr) + + if size <= 1: + return arr + ind = 0 + # while loop + while ind < size: + if ind == 0: + ind += 1 + elif arr[ind] >= arr[ind - 1]: + ind += 1 + else: + # swap + temp = arr[ind - 1] + arr[ind - 1] = arr[ind] + arr[ind] = temp + ind -= 1 + + return arr diff --git a/bookmarks/articles.md b/bookmarks/articles.md index a2b8e1ca..ccdd64e2 100644 --- a/bookmarks/articles.md +++ b/bookmarks/articles.md @@ -55,3 +55,7 @@ This is a list of articles that may be useful for algorithms and data structures - https://www.ultravioletsoftware.com/single-post/2017/03/23/An-introduction-into-the-WSGI-ecosystem - https://jwt.io/introduction/ + +- https://khashtamov.com/en/how-to-become-a-data-engineer/ + +- https://blog.mirrorfly.com/xmpp-vs-websockets-instant-messaging-protocol-comparison/ diff --git a/bookmarks/misc.md b/bookmarks/misc.md index 4bb89d2f..2afa94e6 100644 --- a/bookmarks/misc.md +++ b/bookmarks/misc.md @@ -15,3 +15,7 @@ This is a list of misc links that may be useful when learning or researching dat - https://machinelearningmastery.com - https://stackoverflow.com/questions/10631326/difference-between-select-into-and-insert-into-from-old-table + +- Why SSL are not issued for IP address - https://stackoverflow.com/a/33419662/6111200 + +- PATCH vs PUT https://stackoverflow.com/questions/28459418/use-of-put-vs-patch-methods-in-rest-api-real-life-scenarios/39338329#39338329 diff --git a/bookmarks/topics.md b/bookmarks/topics.md index bf8c7be2..d1d54010 100644 --- a/bookmarks/topics.md +++ b/bookmarks/topics.md @@ -14,4 +14,6 @@ This is a list of links to topics that may be helpful in learning or researching - https://stackoverflow.com/questions/40200413/sessions-vs-token-based-authentication -- https://stackoverflow.com/questions/15678406/when-to-use-myisam-and-innodb \ No newline at end of file +- https://stackoverflow.com/questions/15678406/when-to-use-myisam-and-innodb + +- https://www.bigocheatsheet.com/ diff --git a/bookmarks/videos.md b/bookmarks/videos.md deleted file mode 100644 index 0ac8813e..00000000 --- a/bookmarks/videos.md +++ /dev/null @@ -1,9 +0,0 @@ -# Videos - -This is a list of video links that may be useful for algorithms and data structures learning. - -## Links: - -- https://www.youtube.com/watch?v=mSzUb7f47qk - -- https://www.youtube.com/user/mycodeschool \ No newline at end of file diff --git a/data_structures/__init__.py b/data_structures/__init__.py deleted file mode 100644 index 8b137891..00000000 --- a/data_structures/__init__.py +++ /dev/null @@ -1 +0,0 @@ - diff --git a/data_structures/array/binary_search_infinite_array.py b/data_structures/array/binary_search_infinite_array.py index 6eff9207..1db83e07 100644 --- a/data_structures/array/binary_search_infinite_array.py +++ b/data_structures/array/binary_search_infinite_array.py @@ -23,7 +23,7 @@ def search(arr, val): high = 1 while temp < val: - low = 0 + low = high high = 2 * high temp = arr[high] diff --git a/data_structures/array/duplicate.py b/data_structures/array/duplicate.py index ad4a3fa5..4838b1aa 100644 --- a/data_structures/array/duplicate.py +++ b/data_structures/array/duplicate.py @@ -1,6 +1,12 @@ # Find duplicate in an array of integers given that the integers are in random order and # not necessarily each integer i is 0 <= i <= N where N = length of array +# Solution - use tortoise and hare algorithm. The tortoise pointer moves slower while the hare pointer +# moves faster + +# Note: this array will always contain a duplicate number due to the pigeonhole principle. You are trying to fit +# N different numbers in an array of size N - 1 so one number will be repeated + def duplicate(arr): tortoise = arr[0] hare = arr[0] @@ -11,10 +17,14 @@ def duplicate(arr): if tortoise == hare: break - ptr1 = nums[0] - ptr2 = tortoise - while ptr1 != ptr2: - ptr1 = nums[ptr1] - ptr2 = nums[ptr2] + tortoise = arr[0] + + while tortoise != hare: + tortoise = arr[tortoise] + hare = arr[hare] + + return hare + - return ptr1 +arr = [3,5,1,2,4,5] +print(duplicate(arr)) \ No newline at end of file diff --git a/data_structures/array/duplicates.py b/data_structures/array/duplicates.py index 04940c0e..a146efba 100644 --- a/data_structures/array/duplicates.py +++ b/data_structures/array/duplicates.py @@ -4,5 +4,5 @@ def duplicate(arr): if arr[abs(x) - 1] < 0: res.append(abs(x)) else: - nums[abs(x) - 1] *= -1 + arr[abs(x) - 1] *= -1 return res diff --git a/data_structures/array/equilibrium_index.py b/data_structures/array/equilibrium_index.py new file mode 100644 index 00000000..2b1d0601 --- /dev/null +++ b/data_structures/array/equilibrium_index.py @@ -0,0 +1,23 @@ +# Find the equilibrium index of an array. An equilibrium index is such that +# the sum of elements to the left of it is equal to sum of elements to the right +# of it + +def find_equi(arr): + total_sum = sum(arr) + + left_sum = 0 + + for i, num in enumerate(arr): + + total_sum -= num + + if left_sum == total_sum: + return i + + left_sum += num + + return -1 + + +arr = [-7, 1, 5, 2, -4, 3, 0] +print(find_equi(arr)) diff --git a/data_structures/array/even_more_than_odd.py b/data_structures/array/even_more_than_odd.py new file mode 100644 index 00000000..62735d9c --- /dev/null +++ b/data_structures/array/even_more_than_odd.py @@ -0,0 +1,16 @@ +# Rearrange an array such that numbers at even indexes are greater than numbers +# at odd indexes + +def rearrange(arr): + for i in range(1, len(arr)): + if i % 2 == 0: + if arr[i] > arr[i-1]: + arr[i-1], arr[i] = arr[i], arr[i-1] + else: + if arr[i] < arr[i-1]: + arr[i-1], arr[i] = arr[i], arr[i-1] + print(arr) + + +arr = [ 1, 3, 2, 2, 5 ] +rearrange(arr) \ No newline at end of file diff --git a/data_structures/array/find_given_sum_in_array.py b/data_structures/array/find_given_sum_in_array.py index 0aacc5b4..e3adcfcb 100644 --- a/data_structures/array/find_given_sum_in_array.py +++ b/data_structures/array/find_given_sum_in_array.py @@ -24,4 +24,4 @@ def find_sum(arr, s): arr = [15, 2, 4, 8, 9, 5, 10, 23] -print(find_sum(arr, 6)) +print(find_sum(arr, 6)) \ No newline at end of file diff --git a/data_structures/array/first_repeating_char.py b/data_structures/array/first_repeating_char.py index 2ba272eb..89477411 100644 --- a/data_structures/array/first_repeating_char.py +++ b/data_structures/array/first_repeating_char.py @@ -1,4 +1,4 @@ -# Find the first character in a string without using extra space +# Find the first repeated character in a string without using extra space # With extra space its simple. Just check for the element in a hash map # If present, then it is the recurrent char diff --git a/data_structures/array/kadane_algorithm.py b/data_structures/array/kadane_algorithm.py index f2fee2d3..19824af5 100644 --- a/data_structures/array/kadane_algorithm.py +++ b/data_structures/array/kadane_algorithm.py @@ -1,3 +1,12 @@ +""" +Kadane's algorithm is used to find the maximum contiguous sum in an array. +The logic is simple. Take the first element in the sum and then find current max num. +Curr max = max(arr[i], curr_max + arr[i]) - we add this number if it increases the sum, +otherwise we take the number if it is more than the sum + +Then keep track of max of this value +""" + def max_sum(arr): max_so_far = arr[0] curr_max = arr[0] diff --git a/data_structures/array/majority_element.py b/data_structures/array/majority_element.py index 511902dd..457686ad 100644 --- a/data_structures/array/majority_element.py +++ b/data_structures/array/majority_element.py @@ -1,3 +1,7 @@ +# The only prerequisite condition of this algorithm is that the array +# definitely contains a majority element, otherwise it will just return +# the last element + def majority(arr): maj_index = 0 count = 1 @@ -11,3 +15,8 @@ def majority(arr): count = 1 return arr[maj_index] + + +arr = [3, 3, 1,5,6,8,3,0,7] + +print(majority(arr)) \ No newline at end of file diff --git a/data_structures/array/max_product_three_elements.py b/data_structures/array/max_product_three_elements.py index 0d31fd53..5883172e 100644 --- a/data_structures/array/max_product_three_elements.py +++ b/data_structures/array/max_product_three_elements.py @@ -1,3 +1,6 @@ +# Q - Find the max product of three elements of an array +# A - Find the max 3 numbers and 2 min numbers. Then find the max of (min1*min2*max1, max1*max2*max3) + import sys def product(arr): diff --git a/data_structures/array/min_swaps.py b/data_structures/array/min_swaps.py new file mode 100644 index 00000000..3fdd13a7 --- /dev/null +++ b/data_structures/array/min_swaps.py @@ -0,0 +1,39 @@ +# Minimum swaps required to bring all elements less than or equal to k together + +def min_swaps(arr, k): + # First find out how many elements are there which are less than or + # equal to k + count = 0 + for i in arr: + if i <= k: + count += 1 + + # This count defines a window - inside this window all our elements should + # be placed + # Find the count of bad elements - elements which are more than k and that will be + # our starting answer as we will have to swap them out + bad = 0 + for i in range(0, count): + if arr[i] > k: + bad += 1 + + ans = bad + j = count + + for i in range(0, len(arr)): + if j == len(arr): + break + + if arr[i] > k: + bad -= 1 # because we have moved the bad element out of the window + + if arr[j] > k: + bad += 1 + + ans = min(bad, ans) + j += 1 + + print('answer - ', ans) + +arr = [2,7,9,5,8,7,4] +min_swaps(arr, 5) \ No newline at end of file diff --git a/data_structures/array/moves_zeros_to_end.py b/data_structures/array/moves_zeros_to_end.py index 036c5cd8..c5589961 100644 --- a/data_structures/array/moves_zeros_to_end.py +++ b/data_structures/array/moves_zeros_to_end.py @@ -1,3 +1,6 @@ +# Move all zeros in an array to the end + + def move(arr): count = 0 for a in arr: diff --git a/data_structures/array/number_of_1_in_sorted_array.py b/data_structures/array/number_of_1_in_sorted_array.py index 79689872..bfcfda50 100644 --- a/data_structures/array/number_of_1_in_sorted_array.py +++ b/data_structures/array/number_of_1_in_sorted_array.py @@ -1,4 +1,8 @@ -# The array is sorted in decreasing order +""" +Count the number of 1s in a sorted array +Instead of linearly searching the array to find the first occurence, +do a binary search to find the first 0 +""" def count(arr): start = 0 diff --git a/data_structures/array/number_of_elements_that_can_searched_using_binary_search.py b/data_structures/array/number_of_elements_that_can_searched_using_binary_search.py new file mode 100644 index 00000000..9137db3c --- /dev/null +++ b/data_structures/array/number_of_elements_that_can_searched_using_binary_search.py @@ -0,0 +1,52 @@ +""" +In an input of unsorted integer array, find the number of elements +that can be searched using binary search + +The idea is the an element is binary searchable if the elements to the +left of it are smaller than it and the elements to the right of it +are bigger than it + +So maintain two arrays - left_max and right_min such that in i'th index - + +* left_max[i] contains the max element between 0 and i-1 (left to right movement) +* right_min[i] contains the min element between i+1 and n-1 (right to left movement) + +Now for every element in the array, if its index its i, then it is binary searchable +if left_max[i] < arr[i] < right_min[i] +""" +import sys + +def get_searchable_numbers(arr, n): + left_max = [None] * n + right_min = [None] * n + + left_max[0] = float('-inf') + right_min[n-1] = float('inf') + + for i in range(1, n): + left_max[i] = max(left_max[i-1], arr[i-1]) + + for i in range(len(arr) - 2, -1, -1): + right_min[i] = min(right_min[i+1], arr[i+1]) + + res = [] + count = 0 + + for i in range(0, n): + num = arr[i] + left = left_max[i] + right = right_min[i] + + if left < num < right: + res.append(num) + count += 1 + + return count, res + + +if __name__ == '__main__': + #arr = [5,1,4,3,6,8,10,7,9] + arr = [4,1,3,9,8,10,11] + count, res = get_searchable_numbers(arr, len(arr)) + + print(count, res) diff --git a/data_structures/array/peak_element.py b/data_structures/array/peak_element.py index e69de29b..ded661b0 100644 --- a/data_structures/array/peak_element.py +++ b/data_structures/array/peak_element.py @@ -0,0 +1,21 @@ +# A peak element is an element such that both of its neighbours are smaller than it +# In case of corner elements, consider only one neighbour + + +def peak(arr, low, high): + n = len(arr) + + while low <= high: + mid = (high - low) // 2 + + if (mid == 0 or arr[mid-1] <= arr[mid]) and (mid == n-1 or arr[mid+1] <= arr[mid]): + return(arr[mid]) + + elif mid > 0 and arr[mid-1] > arr[mid]: + high = mid - 1 + + else: + low = mid + 1 + +arr = [1, 3, 20, 4, 1, 0] +print(peak(arr, 0, len(arr) - 1)) \ No newline at end of file diff --git a/data_structures/array/permutations_of_word.py b/data_structures/array/permutations_of_word.py index c0085b54..e5eb8998 100644 --- a/data_structures/array/permutations_of_word.py +++ b/data_structures/array/permutations_of_word.py @@ -6,7 +6,7 @@ def permutation(lst): l = [] for i in range(len(lst)): m = lst[i] - rem_lst = lst[:i] + lst[i+i:] + rem_lst = lst[:i] + lst[i+1:] for p in permutation(rem_lst): l.append([m] + p) return l diff --git a/data_structures/array/rearrange_positive_negative.py b/data_structures/array/rearrange_positive_negative.py new file mode 100644 index 00000000..ab1b971c --- /dev/null +++ b/data_structures/array/rearrange_positive_negative.py @@ -0,0 +1,23 @@ +# Rearragne positive and negative numbers in an array such that they appear +# alternately. If there are more numbers of any one kind, put them at the end + +def rearrange(arr): + i = -1 + + for j in range(len(arr)): + if arr[j] < 0: + i += 1 # maintaining index of the last negative number + arr[j], arr[i] = arr[i], arr[j] + + pos = i + 1 # index of first positive number + neg = 0 # index of first negative number + + while pos < len(arr) and neg < pos and arr[neg] < 0: + arr[pos], arr[neg] = arr[neg], arr[pos] + pos += 1 + neg += 2 + + print(arr) + +arr = [-1, 2, -3, 4, 5, 6, -7, 8, 9] +rearrange(arr) \ No newline at end of file diff --git a/data_structures/array/rotation.py b/data_structures/array/rotation.py index 36b9db0b..c89ab715 100644 --- a/data_structures/array/rotation.py +++ b/data_structures/array/rotation.py @@ -14,14 +14,14 @@ def rotate(arr, d): while 1: k = j + d if k >= n: - k = k -n + k -= n if k == i: break arr[j] = arr[k] j = k arr[j] = temp - + arr = [1,2,3,4,5] -rotate(arr, 1) +rotate(arr, 3) print(arr) diff --git a/data_structures/array/square_of_sorted_array.py b/data_structures/array/square_of_sorted_array.py index 88ef6c87..68f196b9 100644 --- a/data_structures/array/square_of_sorted_array.py +++ b/data_structures/array/square_of_sorted_array.py @@ -1,3 +1,8 @@ +""" +Find the square of all the numbers of a sorted array such that after finding the square of the sorted array, the +resultant array containing the squared numbers remains sorted +""" + def square(arr): n = len(arr) j = 0 @@ -26,8 +31,3 @@ def square(arr): j += 1 return ans - - - - - diff --git a/data_structures/array/transpose_matrix.py b/data_structures/array/transpose_matrix.py new file mode 100644 index 00000000..c7abb7f4 --- /dev/null +++ b/data_structures/array/transpose_matrix.py @@ -0,0 +1,14 @@ +class Solution: + def transpose(self, A: List[List[int]]) -> List[List[int]]: + l=[] + i=0 + while(i!=len(A[0])): + x=[] + j=0 + while(j 1: + if left_to_right: + print_level(root.left, level-1, left_to_right) + print_level(root.right, level-1, left_to_right) + else: + print_level(root.right, level-1, left_to_right) + print_level(root.left, level-1, left_to_right) + + +root = Node(1) +root.left = Node(2) +root.right = Node(3) +root.left.left = Node(7) +root.left.right = Node(6) +root.right.left = Node(5) +root.right.right = Node(4) + +print_spiral(root) \ No newline at end of file diff --git a/data_structures/binary_trees/top_view.py b/data_structures/binary_trees/top_view.py new file mode 100644 index 00000000..565e9e00 --- /dev/null +++ b/data_structures/binary_trees/top_view.py @@ -0,0 +1,52 @@ +""" +Print the top view of a binary tree + +Almost like vertical traversal +""" + +class Node: + + def __init__(self, val): + self.val = val + self.left = None + self.right = None + self.col = None + + +def top_view(root): + if not root: + return + + queue = [] + col = 0 + d = {} + + queue.append(root) + root.col = col + + while queue: + root = queue.pop(0) + col = root.col + + if col not in d: + d[col] = root.val + + if root.left: + queue.append(root.left) + root.left.col = col - 1 + if root.right: + queue.append(root.right) + root.right.col = col + 1 + + for i in sorted(d): + print(d[i], end=" ") + + +root = Node(1) +root.left = Node(2) +root.right = Node(3) +root.left.right = Node(4) +root.left.right.right = Node(5) +root.left.right.right.right = Node(6) + +top_view(root) \ No newline at end of file diff --git a/data_structures/binary_trees/traversals.py b/data_structures/binary_trees/traversals.py new file mode 100644 index 00000000..5c26eb39 --- /dev/null +++ b/data_structures/binary_trees/traversals.py @@ -0,0 +1,15 @@ +class Node: + + def __init__(self, val): + self.val = val + self.left = None + self.right = None + + +class Tree: + + def __init__(self, root): + self.root = root + + + def inorder() diff --git a/data_structures/binary_trees/vertical_traversal.py b/data_structures/binary_trees/vertical_traversal.py new file mode 100644 index 00000000..344a546f --- /dev/null +++ b/data_structures/binary_trees/vertical_traversal.py @@ -0,0 +1,102 @@ +""" +Print vertical view of a binary tree + + 1 + / \ + 2 3 + / \ / \ + 4 5 6 7 + / \ + 8 9 + + +The output of print this tree vertically will be: +4 +2 +1 5 6 +3 8 +7 +9 + +source - https://www.geeksforgeeks.org/print-binary-tree-vertical-order-set-2/ +""" + +class Node: + + def __init__(self, val): + self.val = val + self.left = None + self.right = None + self.col = None + + +def print_vertical_util(root, col, d): + if not root: + return + + if col in d: + d[col].append(root.val) + else: + d[col] = [root.val] + + print_vertical_util(root.left, col-1, d) + print_vertical_util(root.right, col+1, d) + + +def print_vertical(root): + d = {} + col = 0 + + print_vertical_util(root, col, d) + + for k, v in sorted(d.items()): + for i in v: + print(i, end=' ') + print() + + +def print_vertical_iterative(root): + queue = [] + col = 0 + d = {} + + queue.append(root) + root.col = col + + while queue: + root = queue.pop(0) + col = root.col + + if col not in d: + d[col] = [root.val] + else: + d[col].append(root.val) + + if root.left: + queue.append(root.left) + root.left.col = col - 1 + if root.right: + queue.append(root.right) + root.right.col = col + 1 + + for k, v in sorted(d.items()): + for i in v: + print(i, end=' ') + print() + + + +root = Node(1) +root.left = Node(2) +root.right = Node(3) +root.left.left = Node(4) +root.left.right = Node(5) +root.right.left = Node(6) +root.right.right = Node(7) +root.right.left.right = Node(8) +root.right.right.right = Node(9) + +print_vertical(root) + +print("Iterative solution - ") +print_vertical_iterative(root) \ No newline at end of file diff --git a/data_structures/bst/average_of_levels.py b/data_structures/bst/average_of_levels.py index 5028cf58..273c18d6 100644 --- a/data_structures/bst/average_of_levels.py +++ b/data_structures/bst/average_of_levels.py @@ -1,3 +1,7 @@ +""" +Find the mathematical average of all levels of a BST +""" + import collections class Node(): diff --git a/data_structures/bst/duplicate_keys.py b/data_structures/bst/duplicate_keys.py index 4e0e5cbf..a09dc2c1 100644 --- a/data_structures/bst/duplicate_keys.py +++ b/data_structures/bst/duplicate_keys.py @@ -1,3 +1,11 @@ +""" +Create a BST such that it can have duplicate nodes. +Technically BST cannot have duplicate nodes. The work around is to +have a counter associated with every node. Increment its count whenever +there is a duplicate. If the count goes to zero, delete that node +""" + + class Node(): def __init__(self, val): @@ -37,12 +45,41 @@ def insert(root, val): return root +def min_value_node(root): + curr = root + while curr: + curr = curr.left + return curr.val + + def delete(root, val): if not root: - return - if root.val == val: + return None + + if val < root.val: + root.left = delete(root.left, val) + elif val > root.val: + root.right = delete(root.right, val) + else: if root.count > 1: root.count -= 1 + else: + # check if left node is None + if not root.left: + temp = root.right + root = None + return temp + # chec if right node is None + if not root.right: + temp = root.left + root = None + return temp + + temp = min_value_node(root.right) + root.val = temp.val + root.right = delete(root.right, temp.val) + + return root root = Node(5) @@ -53,3 +90,7 @@ def delete(root, val): insert(root, 10) inorder(root) + +print("After deletion") +root = delete(root, 8) +inorder(root) \ No newline at end of file diff --git a/data_structures/bst/index.md b/data_structures/bst/index.md deleted file mode 100644 index 97a6320f..00000000 --- a/data_structures/bst/index.md +++ /dev/null @@ -1,31 +0,0 @@ -# Index of BST - -* [Deletion](deletion.py) -* [Sorted Array To BST](sorted_array_to_bst.py) -* [Print Left Node](print_left_node.py) -* [Diameter](diameter.py) -* [BFS](bfs.py) -* [Check if BT is BST](check_if_bt_if_bst.py) -* [Trim BST](trim_bst.py) -* [BT to BST](bt_to_bst.py) -* [Binary Search Tree](binary_search_tree.py) -* [Convert BST to Right Node Tree](convert_bst_to_right_node_tree.py) -* [kth largest in BST](kth_largest_in_bst.py) -* [Range Sum](range_sum.py) -* [DFS Iterative](dfs_iterative.py) -* [Print Ancestor](print_ancestor.py) -* [Min Max Value in BST](min_max_value_in_bst.py) -* [Check BT is Subtree of another BT](check_bt_is_subtree_of_another_bt.py) -* [Ceil](ceil.py) -* [Closest Element](closest_element.py) -* [kth smallest in BST](kth_smallest_in_bst.py) -* [Insertion Iterative](insertion_iterative.py) -* [Lowest Common Ancestor](lowest_common_ancestor.py) -* [DFS Recursion](dfs_recursion.py) -* [Search](search.py) -* [Insertion Recursive](insertion_recursive.py) -* [Duplicate Keys](duplicate_keys.py) -* [Merge Sum](merge_sum.py) -* [Linked List to BST](linked_list_to_bst.py) -* [Reverse Inorder Traversal](reverse_inorder_traversal.py) -* [Average of Levels](average_of_levels.py) diff --git a/data_structures/bst/kth_largest_in_bst.py b/data_structures/bst/kth_largest_in_bst.py index 1a013baf..e96eb84c 100644 --- a/data_structures/bst/kth_largest_in_bst.py +++ b/data_structures/bst/kth_largest_in_bst.py @@ -1,3 +1,11 @@ +""" +Print the kth largest element in a BST + +The inorder traversal gives elements of BST in ascending order. Do reverse inorder +(Right-Node-Left) and print the kth element +""" + + class Node(): def __init__(self, val): diff --git a/data_structures/bst/kth_smallest_in_bst.py b/data_structures/bst/kth_smallest_in_bst.py index aefa1345..26131ee1 100644 --- a/data_structures/bst/kth_smallest_in_bst.py +++ b/data_structures/bst/kth_smallest_in_bst.py @@ -1,3 +1,11 @@ +""" +Print the kth smallest number in BST + +The inorder traversal of BST gives elements in ascending order. So do inorder, +keep count and return the kth value +""" + + class Node(): def __init__(self, val): diff --git a/data_structures/bst/print_ancestor.py b/data_structures/bst/print_ancestor.py index b443c559..b065efa0 100644 --- a/data_structures/bst/print_ancestor.py +++ b/data_structures/bst/print_ancestor.py @@ -14,6 +14,3 @@ def print_ancestor_recursive(root, key): if print_ancestor_recursive(root.left, key) or print_ancestor_recursive(root.right, key): return root.data return False - - - diff --git a/data_structures/bst/second_largest_in_bst.py b/data_structures/bst/second_largest_in_bst.py index 598b36bb..334a3dd6 100644 --- a/data_structures/bst/second_largest_in_bst.py +++ b/data_structures/bst/second_largest_in_bst.py @@ -1,3 +1,14 @@ +""" +There can be two condition for finding the second largest element in a BST + +1. If right subtree does not exist, find the largest on the left side +Otherwise, +2. If right exists but right's left and right's right do not, that means +you are currently at the second largest element + +Move to right +""" + class Node: def __init__(self, val): diff --git a/data_structures/bst/trim_bst.py b/data_structures/bst/trim_bst.py index 31a17422..1ba86545 100644 --- a/data_structures/bst/trim_bst.py +++ b/data_structures/bst/trim_bst.py @@ -1,3 +1,7 @@ +""" +Trim a BST so that all elements lie within a given high and low range +""" + class Node(): def __init__(self, val): @@ -12,7 +16,7 @@ def trim(root, L, R): if root.val > R: return trim(root.left, L, R) if root.val < L: - return trim(root. right, L, R) + return trim(root.right, L, R) root.left = trim(root.left, L, R) root.right = trim(root.right, L, R) return root diff --git a/data_structures/circular_linked_list/check_circular_linked_list.py b/data_structures/circular_linked_list/check_circular_linked_list.py index 6a72e6e7..4da76507 100644 --- a/data_structures/circular_linked_list/check_circular_linked_list.py +++ b/data_structures/circular_linked_list/check_circular_linked_list.py @@ -1,3 +1,7 @@ +""" +Check if a linked list is a circular linked list +""" + class Node(): def __init__(self, val): diff --git a/data_structures/circular_linked_list/index.md b/data_structures/circular_linked_list/index.md deleted file mode 100644 index a14e9993..00000000 --- a/data_structures/circular_linked_list/index.md +++ /dev/null @@ -1,5 +0,0 @@ -# Index of circular linked list - -* [Check Circular Linked List](check_circular_linked_list.py) -* [Delete](delete.py) -* [Traversal](traversal.py) diff --git a/data_structures/deque/deque.py b/data_structures/deque/deque.py index d578c8e2..85c7dd2c 100644 --- a/data_structures/deque/deque.py +++ b/data_structures/deque/deque.py @@ -47,7 +47,7 @@ def get_last(self): def size(self): return len(self.data) - def isEmpty(self): + def is_empty(self): if len(self.data) == 0: return True return False @@ -59,7 +59,7 @@ def contains(self, elem): return False - def printElems(self): + def print_elements(self): result = "" for i in self.data: diff --git a/data_structures/doubly_linked_list/index.md b/data_structures/doubly_linked_list/index.md deleted file mode 100644 index 0a7d6f6d..00000000 --- a/data_structures/doubly_linked_list/index.md +++ /dev/null @@ -1,3 +0,0 @@ -# Index of doubly_linked_list - -* [Doubly Linked List](doubly_linked_list.py) diff --git a/data_structures/graphs/Adjacency_matrix.py b/data_structures/graphs/adjacency_matrix.py similarity index 71% rename from data_structures/graphs/Adjacency_matrix.py rename to data_structures/graphs/adjacency_matrix.py index 7895a279..f173bc37 100644 --- a/data_structures/graphs/Adjacency_matrix.py +++ b/data_structures/graphs/adjacency_matrix.py @@ -4,10 +4,7 @@ def __init__(self, vertices, directed: bool): self.V = vertices self.e = 0 self.d = directed - self.graph = [] - for i in range(self.V): - lst = [0] * self.V - self.graph.append(lst) + self.graph = [[0 for i in range(vertices)] for j in range(vertices)] def add_edge(self, ver1, ver2): if self.d: @@ -17,7 +14,7 @@ def add_edge(self, ver1, ver2): self.graph[ver2][ver1] = 1 def remove_edge(self, ver1, ver2): - if self.d[ver1][ver2] == 0: + if self.graph[ver1][ver2] == 0: print("No edge between %d and %d" % (ver1, ver2)) return if self.d: @@ -30,6 +27,11 @@ def print_graph(self): for i in self.graph: print(i) - - +if __name__=="__main__": + g1 = Graph(3,0) + g1.add_edge(0,0) + g1.add_edge(1,1) + g1.add_edge(2,2) + g1.remove_edge(2,1) + g1.print_graph() diff --git a/data_structures/graphs/all_paths_between_two_vertices.py b/data_structures/graphs/all_paths_between_two_vertices.py index 958a2b82..3632d36e 100644 --- a/data_structures/graphs/all_paths_between_two_vertices.py +++ b/data_structures/graphs/all_paths_between_two_vertices.py @@ -1,5 +1,5 @@ # Use backtracking -# The only with this approach is that if there is a cycle, then +# The only poroblem with this approach is that if there is a cycle, then # it can show infinitely many paths # Reference - https://www.geeksforgeeks.org/count-possible-paths-two-vertices/ diff --git a/data_structures/graphs/bellman_ford.py b/data_structures/graphs/bellman_ford.py new file mode 100644 index 00000000..49aabe8a --- /dev/null +++ b/data_structures/graphs/bellman_ford.py @@ -0,0 +1,102 @@ +""" +Bellman Ford algorithm is used to find the single source shortest path in a weighted directed graph. +Example - find the shortest path from node A to F + +Advantages of this algorithm over Djikstra's - +1. Bellman Ford also works for negative weight edges +2. Also finds negative weight cycles in a graph + +Disvantage - +1. Slower than Dijsktra's + +** Idea (https://www.youtube.com/watch?v=-mOEd_3gTK0&t) + +0. Shortest path can be found as - +if distance[v] > distance[u] + weight(u, v), then the right side is the shorter path +1. Mark parent and the new weight each time this is run +2. Do this V-1 times. In the first iteration, the algorithm will find the shortest path between the source +and source's neighbours +3. In the second, between source and source's neighbours' neighbours and so on (V - 1) times +4. Do this again. If in this Vth iteration, the weights of path decrease, then there is a negative cycle in the +graph + +Time complexity - O(V * E) +Space complexity - O(V) +""" + +from collections import defaultdict + +class Edge: + + def __init__(self, source, dest, weight): + self.source = source + self.dest = dest + self.weight = weight + + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + self.edges = {} + + + def add_edge(self, u, v, weight): + self.graph[u].append(v) + self.edges[(u, v)] = Edge(u, v, weight) + + + def bellman_ford(self, source, destination): + parent = [-1] * self.vertices + distance = [float("inf")] * self.vertices + + parent[source] = source + distance[source] = source + + for i in range(self.vertices - 1): # Doing V - 1 times to find shortest distance + + for (u, v) in self.edges: + wt = self.edges[(u, v)].weight + + if distance[v] > distance[u] + wt: + distance[v] = distance[u] + wt + parent[v] = u + + # Now for the Vth iteration, check for the negative cycle + + negative_cycle_present = False + + for (u, v) in self.edges: + wt = self.edges[(u, v)].weight + if distance[v] > distance[u] + wt: + negative_cycle_present = True + break + + if negative_cycle_present: + print('Contains negative cycle') + else: + print('No negative cycle') + + # Printing the shortest path from source to destination + + while True: + print(f'{destination}', end=' ') + destination = parent[destination] + + if destination == source: + break + + +g = Graph(5) + +g.add_edge(0, 1, 4) +g.add_edge(0, 2, 5) +g.add_edge(0, 3, 8) +g.add_edge(1, 2, -3) +g.add_edge(2, 4, 4) +g.add_edge(3, 4, 2) +g.add_edge(4, 3, 1) + +g.bellman_ford(0, 3) \ No newline at end of file diff --git a/data_structures/graphs/bfs.py b/data_structures/graphs/bfs.py index 82bda7d3..67722def 100644 --- a/data_structures/graphs/bfs.py +++ b/data_structures/graphs/bfs.py @@ -2,8 +2,9 @@ class Graph: - def __init__(self): + def __init__(self, vertices): self.graph = defaultdict(list) + self.vertices = vertices def add_edge(self, u, v): @@ -11,29 +12,31 @@ def add_edge(self, u, v): def bfs(self, s): - visited = [False] * len(self.graph) + visited = [False] * self.vertices queue = [] queue.append(s) visited[s] = True + bfs = [] + while queue: s = queue.pop(0) print(s, end=' ') - for i in self.graph[s]: if visited[i] == False: queue.append(i) visited[i] = True +g = Graph(6) -g = Graph() -g.add_edge(0, 1) -g.add_edge(0, 2) -g.add_edge(1, 2) -g.add_edge(2, 0) -g.add_edge(2, 3) -g.add_edge(3, 3) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 3) +g.add_edge(0, 3) +g.add_edge(2, 4) +g.add_edge(3, 4) +g.add_edge(3, 5) g.bfs(0) diff --git a/data_structures/graphs/bipartite_graph.py b/data_structures/graphs/bipartite_graph.py index 0c851248..0296de6e 100644 --- a/data_structures/graphs/bipartite_graph.py +++ b/data_structures/graphs/bipartite_graph.py @@ -1,11 +1,19 @@ -# Check for bipartite graph +""" +Check for bipartite graph -# Do a BFS and make source of red color and its neighbour blue -# Keep on doing this. -# If self loop, return false -# If current color == neighbour color, false, else true +Do a BFS and make source of red color and its neighbour blue +Keep on doing this. +If self loop, return false +If current color == neighbour color, false, else true -# Reference - https://www.geeksforgeeks.org/bipartite-graph/ +Reference - https://www.geeksforgeeks.org/bipartite-graph/ + +A bipartite graph is a graph whose vertices can be divided into two disjoint sets U and V +such that every edge connects a vertex in U to one in V. Also, a bipartite graph does not +contain any odd length cycles + +Also, a graph with no edge is a bipartite graph - https://math.stackexchange.com/a/53947 +""" from collections import defaultdict @@ -16,31 +24,61 @@ def __init__(self, vertices): self.graph = defaultdict(list) + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def bfs(self, s): + visited = [False] * self.V + queue = [] + queue.append(s) + visited[s] = True + + while queue: + s = queue.pop(0) + print(s, end=' ') + + for i in self.graph[s]: + if not visited[i]: + visited[i] = True + queue.append(i) + + def is_bipartite(self, s): - # 0 -> color 0 - # 1 -> color 1 + # 0 -> color 0 (blue) + # 1 -> color 1 (red) # -1 -> no color assigned colors = [-1] * self.V - colors[s] = 1 + colors[s] = 1 # Color soure vertex red queue = [] queue.append(s) while queue: - u = queue.pop() + s = queue.pop(0) - if self.graph[u][v] == 1: # Check for self loop + if s in self.graph[s]: # Check for self loop return False # An edge u to v exists and destination is not # colored - for v in range(self.V): - if self.graph[u][v] == 1 and colors[v] == -1: - colors[v] = 1 - colors[u] - queue.append(v) + for i in self.graph[s]: + if colors[i] == -1: + colors[i] = 1 - colors[s] + queue.append(i) - elif self.graph[u][v] == 1 and colors[v] == colors[u]: + elif colors[i] == colors[s]: return False return True + + +g = Graph(5) +g.add_edge(0, 1) +g.add_edge(1, 2) +g.add_edge(2, 3) +g.add_edge(3, 4) +g.add_edge(4, 0) +print(g.is_bipartite(0)) \ No newline at end of file diff --git a/data_structures/graphs/check_if_graph_is_tree.py b/data_structures/graphs/check_if_graph_is_tree.py new file mode 100644 index 00000000..90ed3425 --- /dev/null +++ b/data_structures/graphs/check_if_graph_is_tree.py @@ -0,0 +1,59 @@ +""" +A graph is a tree if - +1. It does not contain cycles +2. The graph is connected + +Do DFS and see if every vertex can be visited from a source vertex and check for cycle +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def is_tree(self, s): + visited = [False] * self.vertices + parent = [-1] * self.vertices + stack = [] + + visited[s] = True + no_of_visited = 1 + stack.append(s) + + while stack: + s = stack.pop() + + for i in self.graph[s]: + if visited[i] == False: + parent[i] = s + visited[i] = True + stack.append(i) + no_of_visited += 1 + elif parent[s] != i: + return "Not a tree" + + if no_of_visited == self.vertices: + return "Graph is Tree" + else: + return "Not a tree" + + +g = Graph(7) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 3) +g.add_edge(2, 4) +g.add_edge(4, 5) +g.add_edge(1, 6) + +print(g.is_tree(0)) \ No newline at end of file diff --git a/data_structures/graphs/connected_components_undirected_graphs.py b/data_structures/graphs/connected_components_undirected_graphs.py new file mode 100644 index 00000000..feb3dce4 --- /dev/null +++ b/data_structures/graphs/connected_components_undirected_graphs.py @@ -0,0 +1,46 @@ +from collections import defaultdict + + +class Graph: + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def dfs(self, v, temp, visited): + visited[v] = True + temp.append(v) + + for i in self.graph[v]: + if visited[i] == False: + temp = self.dfs(i, temp, visited) + + return temp + + + def connected_components(self): + visited = [False] * self.vertices + cc = [] + + for i in range(self.vertices): + if visited[i] == False: + temp = [] + cc.append(self.dfs(i, temp, visited)) + + return cc + + +g = Graph(5) +g.add_edge(0, 1) +g.add_edge(1, 2) +g.add_edge(2, 0) +g.add_edge(0, 3) +g.add_edge(3, 4) + +print(g.connected_components()) \ No newline at end of file diff --git a/data_structures/graphs/count_edges.py b/data_structures/graphs/count_edges.py index f1601542..6b9d1f87 100644 --- a/data_structures/graphs/count_edges.py +++ b/data_structures/graphs/count_edges.py @@ -1,6 +1,12 @@ -# Use handshaking lemma -# deg(v) = 2|E| -# Time - O(V) +""" +Count the number of edges in an undirected graph + +Use Handshaking Lemma (Note: Handshaking Lemma is only for undirected graph) + +For all v in V, deg(v) = 2|E| +""" + +Time - O(V) class Graph: @@ -39,4 +45,4 @@ def count_edges(self): g.add_edge(6, 8 ) g.add_edge(7, 8 ) -print(g.count_edges()) +print(g.count_edges()) \ No newline at end of file diff --git a/data_structures/graphs/count_sink_nodes.py b/data_structures/graphs/count_sink_nodes.py new file mode 100644 index 00000000..7cce157f --- /dev/null +++ b/data_structures/graphs/count_sink_nodes.py @@ -0,0 +1,31 @@ +""" +Count the number of sink nodes + +Sink nodes are the nodes without any outgoing edge + +Solution - +vertices contains the total number of edges and graph is an adjacency list. +Simply subtract the vertices and len(graph) +""" + +from collections import defaultdict + +class Graph: + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + + + def count_sink_nodes(self): + return self.vertices - len(self.graph) + + +g = Graph(3) +g.add_edge(0, 1) +g.add_edge(0, 2) +print(g.count_sink_nodes()) \ No newline at end of file diff --git a/data_structures/graphs/count_trees.py b/data_structures/graphs/count_trees.py new file mode 100644 index 00000000..28033ee9 --- /dev/null +++ b/data_structures/graphs/count_trees.py @@ -0,0 +1,51 @@ +""" +Count the number of trees in a forest + +A forest is a collection of trees. Do a DFS. If any vertex if not reachable +from any other, then it means it is not a part of that subgraph (tree). Hence +it is a different tree, so increment the count +""" + +from collections import defaultdict + +class Graph: + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[v].append(u) + self.graph[u].append(v) + + + def count_trees(self): + visited = [False] * self.vertices + count = 0 + + for s in range(self.vertices): + if not visited[s]: + visited[s] = True + stack = [] + stack.append(s) + count += 1 + + while stack: + print(stack) + s = stack.pop() + + for i in self.graph[s]: + if not visited[i]: + visited[i] = True + stack.append(i) + + return count + + +g = Graph(5) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(3, 4) + +print('Count of trees - ', g.count_trees()) \ No newline at end of file diff --git a/data_structures/graphs/cycle_in_directed_graph_iterative.py b/data_structures/graphs/cycle_in_directed_graph_iterative.py new file mode 100644 index 00000000..cc0b3cd2 --- /dev/null +++ b/data_structures/graphs/cycle_in_directed_graph_iterative.py @@ -0,0 +1,53 @@ +from collections import defaultdict + + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + + + def dfs(self): + visited = [False] * self.vertices + recursion_stack = [False] * self.vertices + stack = [] + + for v in range(self.vertices): + if visited[v] == False: + visited[v] = True + recursion_stack[v] = True + + stack.append(v) + + while stack: + s = stack.pop() + + recursion_stack[s] = True + + for i in self.graph[s]: + if visited[i] == False: + stack.append(i) + visited[i] = True + recursion_stack[i] = True + elif recursion_stack[i] == True: + return "Contains Cycle" + + recursion_stack[v] = False + + return "No cycle" + + +g = Graph(4) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 2) +g.add_edge(2, 0) +g.add_edge(2, 3) +g.add_edge(3, 3) +print(g.dfs()) \ No newline at end of file diff --git a/data_structures/graphs/cycle_in_directed_graph_using_colors.py b/data_structures/graphs/cycle_in_directed_graph_using_colors.py new file mode 100644 index 00000000..44dfb5ac --- /dev/null +++ b/data_structures/graphs/cycle_in_directed_graph_using_colors.py @@ -0,0 +1,60 @@ +""" +Using three colors - white, gray and black +White - vertices that are not processed (inital state of all vertices) +Gray - vertices that are in DFS +Black - fully traversed vertices (i.e its progenies are also done) + +If while traversing any adjacent node is colored Gray, that means cycle exists +""" + +from collections import defaultdict + +class Graph: + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[u].append(v) + + + def detect_cycle(self): + color = ['white'] * self.vertices + visited = [False] * self.vertices + + stack = [] + + for v in range(self.vertices): + if color[v] == 'white': + color[v] = 'gray' + + stack.append(v) + + while stack: + s = stack.pop() + + for i in self.graph[s]: + if color[i] == 'white': + stack.append(i) + color[i] = 'gray' + elif color[i] == 'gray': + return "Cycle detected" + + color[v] = 'black' + + return "Cycle not present" + + +g = Graph(4) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 2) +g.add_edge(2, 0) +g.add_edge(2, 3) +g.add_edge(3, 3) +# g.add_edge(0, 1) +# g.add_edge(0, 2) +# g.add_edge(1, 3) +print(g.detect_cycle()) \ No newline at end of file diff --git a/data_structures/graphs/cycle_in_directed_graph_using_colors_recursive.py b/data_structures/graphs/cycle_in_directed_graph_using_colors_recursive.py new file mode 100644 index 00000000..c4a687ed --- /dev/null +++ b/data_structures/graphs/cycle_in_directed_graph_using_colors_recursive.py @@ -0,0 +1,58 @@ +""" +Using three colors - white, gray and black +White - vertices that are not processed (inital state of all vertices) +Gray - vertices that are in DFS +Black - fully traversed vertices (i.e its progenies are also done) + +If while traversing any adjacent node is colored Gray, that means cycle exists +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + + + def dfs(self, vertex, colors): + colors[vertex] = 'Gray' + + for v in self.graph[vertex]: + + if colors[v] == 'Gray': + return True + + elif colors[v] == 'White' and self.dfs(v, colors) == True: + return True + + colors[vertex] = 'Black' + return False + + + def is_cyclic(self): + colors = ['White'] * self.vertices + + for vertex in self.graph.keys(): + if colors[vertex] == 'White': + if self.dfs(vertex, colors) == True: + return True + + return False + + +g = Graph(4) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 2) +g.add_edge(2, 0) +g.add_edge(2, 3) +g.add_edge(3, 3) + +print(g.is_cyclic()) \ No newline at end of file diff --git a/data_structures/graphs/cyclic_in_undirected_graph.py b/data_structures/graphs/cycle_in_undirected_graph.py similarity index 100% rename from data_structures/graphs/cyclic_in_undirected_graph.py rename to data_structures/graphs/cycle_in_undirected_graph.py diff --git a/data_structures/graphs/cycle_in_undirected_graph_iterative.py b/data_structures/graphs/cycle_in_undirected_graph_iterative.py new file mode 100644 index 00000000..565cedde --- /dev/null +++ b/data_structures/graphs/cycle_in_undirected_graph_iterative.py @@ -0,0 +1,60 @@ +""" +This is almost same as the iterative one for directed graph but we cannot use the +concept of recursion stack here because in directed graphs, there is defined path to +traverse but in undirected its possible that an edge (or a path) can be traversed +infite number of times. + +Instead check for parents (which means vertex from which you reached the current vertex). +If a vertex is visited and you are not coming to this vertex from the current "source" vertex +(source - vertex from which DFS has started), then it means that in the same DFS chain, there is +another path to reach this vertex - hence a cycle +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def dfs(self): + visited = [False] * self.vertices + parent = [-1] * self.vertices + + stack = [] + + for v in range(self.vertices): + if visited[v] == False: + visited[v] = True + + stack.append(v) + + while stack: + s = stack.pop() + + for i in self.graph[s]: + if visited[i] == False: + parent[i] = s + stack.append(i) + visited[i] = True + elif parent[s] != i: + return "Contains Cycle" + + return "No cycle" + + +g = Graph(5) +g.add_edge(1, 0) +g.add_edge(1, 2) +g.add_edge(2, 0) +g.add_edge(0, 3) +g.add_edge(3, 4) +print(g.dfs()) diff --git a/data_structures/graphs/cycle_in_undirected_graph_union_find.py b/data_structures/graphs/cycle_in_undirected_graph_union_find.py new file mode 100644 index 00000000..e8e0af42 --- /dev/null +++ b/data_structures/graphs/cycle_in_undirected_graph_union_find.py @@ -0,0 +1,59 @@ +""" +This method cannot be used in Directed graph because of the direction of the edge. +In a union of set containing element A and B, you cannot specify whether A is going to B +or vice versa +""" + +from collections import defaultdict + + +class Graph: + + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[u].append(v) + + + def union(self, parent, x, y): + parent[x] = y + + + def find_parent(self, parent, i): + if parent[i] == -1: + return i + else: + return self.find_parent(parent, parent[i]) + + + def check_cyclic(self): + parent = [-1] * self.vertices + + for i in self.graph.keys(): + for j in self.graph[i]: + x = self.find_parent(parent, i) + y = self.find_parent(parent, j) + + if x == y: + return "Cyclic" + + self.union(parent, x, y) + + return "Not cyclic" + + +g = Graph(4) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 2) +g.add_edge(2, 0) +g.add_edge(2, 3) +g.add_edge(3, 3) +# g.add_edge(0, 1) +# g.add_edge(0, 2) +# g.add_edge(1, 3) +print(g.check_cyclic()) \ No newline at end of file diff --git a/data_structures/graphs/dag_longest_path.py b/data_structures/graphs/dag_longest_path.py new file mode 100644 index 00000000..796c43da --- /dev/null +++ b/data_structures/graphs/dag_longest_path.py @@ -0,0 +1,71 @@ +""" +The idea is similar to DAG Shortest Path. Only the comparision part changes +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v, w): + self.graph[u].append((v, w)) + + + def topological_sort_util(self, vertex, visited, stack): + visited[vertex] = True + + for v, weight in self.graph[vertex]: + if visited[v] == False: + self.topological_sort_util(v, visited, stack) + + stack.insert(0, vertex) + + + def topological_sort(self): + visited = [False] * self.vertices + stack = [] + + for v in range(self.vertices): + if visited[v] == False: + self.topological_sort_util(v, visited, stack) + + return stack + + + def longest_path(self, s): + stack = self.topological_sort() + + distance = [-float("inf")] * self.vertices + distance[s] = 0 + + while stack: + i = stack.pop(0) + + for vertex, weight in self.graph[i]: + if distance[vertex] < distance[i] + weight: + distance[vertex] = distance[i] + weight + + for i in range(self.vertices): + print(f"{s} -> {i} = {distance[i]}") + + +g = Graph(6) +g.add_edge(0, 1, 5) +g.add_edge(0, 2, 3) +g.add_edge(1, 3, 6) +g.add_edge(1, 2, 2) +g.add_edge(2, 4, 4) +g.add_edge(2, 5, 2) +g.add_edge(2, 3, 7) +g.add_edge(3, 5, 1) +g.add_edge(3, 4, -1) +g.add_edge(4, 5, -2) + +source = 0 + +g.longest_path(source) \ No newline at end of file diff --git a/data_structures/graphs/dag_shortest_path.py b/data_structures/graphs/dag_shortest_path.py new file mode 100644 index 00000000..36bebda4 --- /dev/null +++ b/data_structures/graphs/dag_shortest_path.py @@ -0,0 +1,79 @@ +""" +Find shortest path from source vertex to all vertices + +In general, Bellaman-Ford and Dijkstra can be used to find shortest path. But for +DAG, we can improvise + +Bellman-Ford: O(V*E), Dijsktra: O(E+VlogV) + +For DAG, using Topological sort will solve this problem in O(V+E). We are using Topo +sort because it lays out a graph in its linear representation according to the paths. +Hence we can proceed in this order to find shortest path +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v, w): + self.graph[u].append((v, w)) + + + def topological_sort_util(self, vertex, visited, stack): + visited[vertex] = True + + for v, weight in self.graph[vertex]: + if visited[v] == False: + self.topological_sort_util(v, visited, stack) + + stack.insert(0, vertex) + + + def topological_sort(self): + visited = [False] * self.vertices + stack = [] + + for v in range(self.vertices): + if visited[v] == False: + self.topological_sort_util(v, visited, stack) + + return stack + + + def shortest_path(self, s): + stack = self.topological_sort() + + distance = [float("inf")] * self.vertices + distance[s] = 0 + + while stack: + i = stack.pop(0) + + for vertex, weight in self.graph[i]: + if distance[vertex] > distance[i] + weight: + distance[vertex] = distance[i] + weight + + for i in range(self.vertices): + print(f"{s} -> {i} = {distance[i]}") + + +g = Graph(6) +g.add_edge(0, 1, 5) +g.add_edge(0, 2, 3) +g.add_edge(1, 3, 6) +g.add_edge(1, 2, 2) +g.add_edge(2, 4, 4) +g.add_edge(2, 5, 2) +g.add_edge(2, 3, 7) +g.add_edge(3, 4, -1) +g.add_edge(4, 5, -2) + +source = 0 + +g.shortest_path(source) \ No newline at end of file diff --git a/data_structures/graphs/dfs.py b/data_structures/graphs/dfs.py index b6608f2b..6af4a6cc 100644 --- a/data_structures/graphs/dfs.py +++ b/data_structures/graphs/dfs.py @@ -7,7 +7,8 @@ def __init__(self): def add_edge(self, u, v): - self.graph[u].append[v] + self.graph[u].append(v) + self.graph[v].append(u) def dfs_util(self, v, visited): @@ -22,3 +23,13 @@ def dfs_util(self, v, visited): def dfs(self, v): visited = [False] * (len(self.graph)) self.dfs_util(v, visited) + + +g = Graph() +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(0, 3) +g.add_edge(1, 4) +g.add_edge(2, 5) +g.add_edge(3, 6) +print(g.dfs(0)) diff --git a/data_structures/graphs/dijsktra_algorithm.py b/data_structures/graphs/dijsktra_algorithm.py new file mode 100644 index 00000000..e69de29b diff --git a/data_structures/graphs/index.md b/data_structures/graphs/index.md deleted file mode 100644 index 7baba1cf..00000000 --- a/data_structures/graphs/index.md +++ /dev/null @@ -1,3 +0,0 @@ -# Index of graphs - -* [Adjacency List](adjacency_list.py) diff --git a/data_structures/graphs/iterative_dfs.py b/data_structures/graphs/iterative_dfs.py index 65c092dc..c2eb7169 100644 --- a/data_structures/graphs/iterative_dfs.py +++ b/data_structures/graphs/iterative_dfs.py @@ -5,7 +5,8 @@ class Graph: - def __init__(self): + def __init__(self, vertices): + self.vertices = vertices self.graph = defaultdict(list) @@ -13,30 +14,36 @@ def add_edge(self, u, v): self.graph[u].append(v) - def dfs(self, s): - visited = [False] * len(self.graph) - + def dfs(self): + visited = [False] * self.vertices stack = [] - stack.append(s) - visited[s] = True - - while stack: - s = stack.pop() - print(s, end=' ') - - for i in self.graph[s]: - if visited[i] == False: - stack.append(i) - visited[i] = True - - -g = Graph() -g.add_edge(0, 1) -g.add_edge(0, 2) -g.add_edge(1, 2) -g.add_edge(2, 0) -g.add_edge(2, 3) -g.add_edge(3, 3) - -g.dfs(0) + for s in range(self.vertices): + if visited[s] == False: + visited[s] = True + + stack.append(s) + + while stack: + s = stack.pop() + print(s, end=' ') + + for i in self.graph[s]: + if visited[i] == False: + stack.append(i) + visited[i] = True + + +# g = Graph(4) +# g.add_edge(0, 1) +# g.add_edge(0, 2) +# g.add_edge(1, 2) +# g.add_edge(2, 0) +# g.add_edge(2, 3) +# g.add_edge(3, 3) +g = Graph(5) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(3, 4) + +g.dfs() diff --git a/data_structures/graphs/kosaraju_algorithm.py b/data_structures/graphs/kosaraju_algorithm.py index cfc2a2b0..99277672 100644 --- a/data_structures/graphs/kosaraju_algorithm.py +++ b/data_structures/graphs/kosaraju_algorithm.py @@ -1,15 +1,22 @@ -# Reference - https://www.geeksforgeeks.org/strongly-connected-components/ +""" +Reference - https://www.geeksforgeeks.org/strongly-connected-components/ -# This is used to find all the strongly connected components and does DFS -# 2 times. +Algorithm - https://youtu.be/RpgcYiky7uw + +This is used to find all the strongly connected components and does DFS +2 times. + +Note: A single vertex is also strongly connected +""" from collections import defaultdict class Graph: + def __init__(self, vertices): - self.V = vertices + self.vertices = vertices self.graph = defaultdict(list) @@ -17,49 +24,65 @@ def add_edge(self, u, v): self.graph[u].append(v) - def dfs_util(self, v, visited): + def fill_time(self, v, visited, stack): visited[v] = True - print(v, end=' ') for i in self.graph[v]: if visited[i] == False: - self.dfs_util(i, visited) + self.fill_time(i, visited, stack) + stack = stack.append(v) - def fill_order(self, v, visited, stack): + + def dfs(self, v, visited): visited[v] = True + print(v, end=' ') + for i in self.graph[v]: if visited[i] == False: - self.fill_order(i, visited, stack) - stack.append(v) + self.dfs(i, visited) - - def get_transpose(self): - g = Graph(self.V) + + def create_tranpose(self): + tgraph = Graph(self.vertices) for i in self.graph: for j in self.graph[i]: - g.add_edge(j, i) - - return g + tgraph.add_edge(j, i) + return tgraph + def kosaraju(self): + visited = [False] * self.vertices stack = [] - visited = [False] * self.V - - for i in range(self.V): - if visited[i] == False: - self.fill_order(i, visited, stack) - gr = self.get_transpose() + for v in range(self.vertices): + if visited[v] == False: + self.fill_time(v, visited, stack) - visited = [False] * self.V + tgraph = self.create_tranpose() + visited = [False] * self.vertices while stack: - i = stack.pop() - if visited[i] == False: - gr.dfs_util(i, visited) - print() - - + s = stack.pop() + + if visited[s] == False: + tgraph.dfs(s, visited) + print() + + +g = Graph(11) +g.add_edge(0, 1) +g.add_edge(2, 0) +g.add_edge(1, 2) +g.add_edge(1, 3) +g.add_edge(3, 4) +g.add_edge(4, 5) +g.add_edge(5, 3) +g.add_edge(6, 5) +g.add_edge(6, 7) +g.add_edge(7, 8) +g.add_edge(8, 9) +g.add_edge(9, 10) +g.kosaraju() \ No newline at end of file diff --git a/data_structures/graphs/level_of_nodes.py b/data_structures/graphs/level_of_nodes.py new file mode 100644 index 00000000..c453fa95 --- /dev/null +++ b/data_structures/graphs/level_of_nodes.py @@ -0,0 +1,49 @@ +""" +Level is the distance of a node from a source node. This concept can be used to find +the distance between 2 nodes in an unweighted graph as well. A simple BFS traversal +between these 2 nodes will give the level and level will always be the shortest distance +between nodes. +""" + +from collections import defaultdict + +class Graph: + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def print_levels(self, s): + levels = [None] * self.vertices + levels[s] = 0 + queue = [] + queue.append(s) + + while queue: + s = queue.pop(0) + + for i in self.graph[s]: + if levels[i] == None: + levels[i] = levels[s] + 1 + queue.append(i) + + print('Node \t Level') + for node, level in enumerate(levels): + print(f'{node} \t {level}') + +g = Graph(8) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 3) +g.add_edge(1, 4) +g.add_edge(1, 5) +g.add_edge(2, 5) +g.add_edge(2, 6) +g.add_edge(6, 7) +g.print_levels(0) \ No newline at end of file diff --git a/data_structures/graphs/max_edges_that_can_be_added_to_dag.py b/data_structures/graphs/max_edges_that_can_be_added_to_dag.py new file mode 100644 index 00000000..f4ba0900 --- /dev/null +++ b/data_structures/graphs/max_edges_that_can_be_added_to_dag.py @@ -0,0 +1,83 @@ +""" +Find the max number of edges that can be added to a Directed Acyclic Graph such that +it remains a DAG. + +Solution - +The trick is to add all edges from left to right. Now adding any edge from right to left +will make the graph cyclic because that edge's counterpart will exist from left-to-right. +Hence a cycle will be formed. + +1. Topologically sort all the edges +2. If edge is not there left to right, create the edge +3. Count the number of edges added + +source - https://www.geeksforgeeks.org/maximum-edges-can-added-dag-remains-dag/ + +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + + + def topological_sort_util(self, v, visited, stack): + visited[v] = True + + for i in self.graph[v]: + if visited[i] == False: + self.topological_sort_util(i, visited, stack) + + stack.insert(0, v) + + + def topological_sort(self): + visited = [False] * self.vertices + stack = [] + + for i in range(self.vertices): + if visited[i] == False: + self.topological_sort_util(i, visited, stack) + + return stack + + + def max_edges(self): + topo = self.topological_sort() + visited = [False] * self.vertices + count = 0 + + for i in range(len(topo)): + vertex = topo[i] + # Mark the connected vertices visited + for j in self.graph[vertex]: + visited[j] = True + + # Print the unmarked nodes from topo + for j in range(i+1, len(topo)): + if visited[topo[j]] == False: + print(f"{vertex} -> {topo[j]}") + count += 1 + + visited[topo[j]] = False + + print('Maximum edges that can be added - ', count) + + +g = Graph(6) +g.add_edge(5, 2) +g.add_edge(5, 0) +g.add_edge(4, 0) +g.add_edge(4, 1) +g.add_edge(2, 3) +g.add_edge(3, 1) + +g.max_edges() \ No newline at end of file diff --git a/data_structures/graphs/max_edges_to_make_bipartite.py b/data_structures/graphs/max_edges_to_make_bipartite.py new file mode 100644 index 00000000..9b42fb82 --- /dev/null +++ b/data_structures/graphs/max_edges_to_make_bipartite.py @@ -0,0 +1,60 @@ +""" +Find the max edges that can be added to a tree so that it stays a bipartite graph + +keep count of nodes of each color - say count_color1, count_color2 +These nodes can be connected in count_color1 x count_color2 ways = max no of edges of a bipartite graph +Also, a tree has n-1 edges +So answer = (count_color1 x count_color2) - (n - 1) + +If the answer comes negative, then its impossible to add any edge because of odd cycles +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def bfs(self, s): + count_color0 = 0 + count_color1 = 0 + + colors = [-1] * self.vertices + colors[s] = 1 + count_color1 += 1 + + queue = [] + queue.append(s) + + while queue: + s = queue.pop(0) + + for i in self.graph[s]: + if colors[i] == -1: + colors[i] = 1 - colors[s] + if colors[i] == 0: + count_color0 += 1 + else: + count_color1 += 1 + queue.append(i) + + ans = (count_color0 * count_color1) - (self.vertices - 1) + return ans if ans > 0 else 0 + + +g = Graph(5) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 3) +g.add_edge(2, 4) + +print(g.bfs(0)) \ No newline at end of file diff --git a/data_structures/graphs/min_edges_between_two_vertices.py b/data_structures/graphs/min_edges_between_two_vertices.py new file mode 100644 index 00000000..13456d5f --- /dev/null +++ b/data_structures/graphs/min_edges_between_two_vertices.py @@ -0,0 +1,52 @@ +from collections import defaultdict + + +class Graph: + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def find_min_edges(self, u, v): + visited = [False] * self.vertices + distance = [0] * self.vertices + queue = [] + + queue.append(u) + visited[u] = True + + while queue: + s = queue.pop(0) + + for i in self.graph[s]: + if visited[i] == False: + distance[i] = distance[s] + 1 + queue.append(i) + visited[i] = True + + return distance[v] + + +g = Graph(9) + +g.add_edge(0, 1) +g.add_edge(0, 7) +g.add_edge(1, 7) +g.add_edge(1, 2) +g.add_edge(2, 3) +g.add_edge(2, 5) +g.add_edge(2, 8) +g.add_edge(3, 4) +g.add_edge(3, 5) +g.add_edge(4, 5) +g.add_edge(5, 6) +g.add_edge(6, 7) +g.add_edge(7, 8) + +print(g.find_min_edges(0, 5)) diff --git a/data_structures/graphs/min_nodes_to_reach_all_nodes.py b/data_structures/graphs/min_nodes_to_reach_all_nodes.py new file mode 100644 index 00000000..d60f6f9b --- /dev/null +++ b/data_structures/graphs/min_nodes_to_reach_all_nodes.py @@ -0,0 +1,11 @@ +""" +We are given a DAG. Find the smallest set of vertices from which all +nodes in the graph are reachable. It's guaranteed that a unique solution exists. + +Solution - +The crux of the question is that - +1. If any node has an indegree > 1 (i.e it is reachable from any other node), then it means +in a connected graph it will be possible to reach here if its parent node also has an indegree +2. So the min will be the set of nodes with indegree = 0 +""" + diff --git a/data_structures/graphs/min_number_of_operations.py b/data_structures/graphs/min_number_of_operations.py new file mode 100644 index 00000000..7d783e30 --- /dev/null +++ b/data_structures/graphs/min_number_of_operations.py @@ -0,0 +1,54 @@ +""" +Minimum number of operations required to transform a number x to y +Valid operations - multiplication by 2, subtraction by 1 +Ex - x = 4, y = 7 +1. 4 * 2 = 8 +2. 8 - 1 = 7 +Answer = 2 +""" + + +from collections import defaultdict + + +class Node: + + + def __init__(self, value, level): + self.value = value + self.level = level + + +def min_steps(x, y): + node_x = Node(x, 0) + + visited = [] + queue = [] + queue.append(node_x) + + while queue: + s = queue.pop(0) + + if s.value == y: + return s.level + + visited.append(s.value) + + if s.value * 2 == y or s.value - 1 == y: + return s.level + 1 + + # If not visited already, add its children + + if s.value * 2 not in visited: + new_node = Node(s.value * 2, s.level + 1) + queue.append(new_node) + + if s.value - 1 not in visited: + new_node = Node(s.value - 1, s.level + 1) + queue.append(new_node) + + +x = 2 +y = 5 + +print(min_steps(x, y)) \ No newline at end of file diff --git a/data_structures/graphs/mother_vertex.py b/data_structures/graphs/mother_vertex.py index 40f8ac10..794c88ee 100644 --- a/data_structures/graphs/mother_vertex.py +++ b/data_structures/graphs/mother_vertex.py @@ -1,9 +1,11 @@ -# A mother vertex is a vertex such that all other vertices -# can be reached by a path from this vertex +""" +A mother vertex is a vertex such that all other vertices +can be reached by a path from this vertex -# Reference - https://www.geeksforgeeks.org/find-a-mother-vertex-in-a-graph/ +Reference - https://www.geeksforgeeks.org/find-a-mother-vertex-in-a-graph/ -# Time complexity - 2 * O(V + E) = O(V + E) +Time complexity - 2 * O(V + E) = O(V + E) +""" from collections import defaultdict diff --git a/data_structures/graphs/print_all_paths_between_nodes.py b/data_structures/graphs/print_all_paths_between_nodes.py new file mode 100644 index 00000000..0b370237 --- /dev/null +++ b/data_structures/graphs/print_all_paths_between_nodes.py @@ -0,0 +1,52 @@ +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + + + def print_path(self, s, d, visited, path): + visited[s] = True + path.append(s) + + if s == d: + print(path) + + else: + for i in self.graph[s]: + if visited[i] == False: + self.print_path(i, d, visited, path) + + # If path from this node does not lead to the destination, remove it + # from the path stack and mark it as not visited + path.pop() + visited[s] = False + + + def print_all_paths(self, s, d): + visited = [False] * self.vertices + path = [] + self.print_path(s, d, visited, path) + + +g = Graph(4) + +g.add_edge(0, 3) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(1, 3) +g.add_edge(2, 0) +g.add_edge(2, 1) + +s = 2 +d = 3 + +print(f'Paths from {s} to {d} are - ') +g.print_all_paths(s, d) \ No newline at end of file diff --git a/data_structures/graphs/root_which_gives_min_height.py b/data_structures/graphs/root_which_gives_min_height.py index 77a738a9..7a079196 100644 --- a/data_structures/graphs/root_which_gives_min_height.py +++ b/data_structures/graphs/root_which_gives_min_height.py @@ -1,4 +1,7 @@ -# Reference - https://www.geeksforgeeks.org/roots-tree-gives-minimum-height/ +""" +Find the node in an undirected graph which gives the minimum height +Reference - https://www.geeksforgeeks.org/roots-tree-gives-minimum-height/ +""" from collections import defaultdict from queue import Queue @@ -22,10 +25,10 @@ def root_min_height(self): q = Queue() for i in range(self.V): - if self.degree[i] == 1: + if self.degree[i] == 1: # To identify leaf nodes q.put(i) - - + + # now move inwards from the leaf node while self.V > 2: for i in range(q.qsize()): t = q.get() @@ -38,7 +41,6 @@ def root_min_height(self): if self.degree[j] == 1: q.put(j) - res = list() while q.qsize() > 0: res.append(q.get()) diff --git a/data_structures/graphs/same_path.py b/data_structures/graphs/same_path.py new file mode 100644 index 00000000..82219691 --- /dev/null +++ b/data_structures/graphs/same_path.py @@ -0,0 +1,80 @@ +""" +Check if two nodes are on the same path in a tree. Use DFS and the concept of intime and outtime. +Intime - time when a node is visited for the first time +Outtime - time when a node is visited for the second time after all its children have been visited +For any pair of node if they are on the same path - +intime[u] < intime[v] and outtime[u] > outtime[v] +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + self.intime = None + self.outtime = None + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def dfs(self): + visited = [False] * self.vertices + intime = [0] * self.vertices + outtime = [0] * self.vertices + timer = 0 + num_children_visited = [0] * self.vertices + + for s in range(self.vertices): + if not visited[s]: + visited[s] = True + + stack = [] + stack.append(s) + + while stack: + s = stack.pop() + timer += 1 + intime[s] = timer + + for i in self.graph[s]: + if visited[i] == False: + stack.append(i) + visited[i] = True + num_children_visited[s] += 1 + + if num_children_visited[i] == len(self.graph[i]): + timer += 1 + outtime[i] = timer + + print('intime - ', intime) + print('outtime - ', outtime) + print('num_children_visited - ', num_children_visited) + self.intime = intime + self.outtime = outtime + + + def check_same_path(self, u, v): + if (self.intime[u] < self.intime[v] and self.outtime[u] > self.outtime[v]) or \ + (self.intime[v] < self.intime[u] and self.outtime[v] > self.outtime[u]): + return True + return False + + +g = Graph(7) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(0, 3) +g.add_edge(1, 4) +g.add_edge(2, 5) +g.add_edge(3, 6) + +g.dfs() + +print('Same path - ', g.check_same_path(0, 6)) +print(g.graph[6]) \ No newline at end of file diff --git a/data_structures/graphs/same_path_recursive.py b/data_structures/graphs/same_path_recursive.py new file mode 100644 index 00000000..3b2087aa --- /dev/null +++ b/data_structures/graphs/same_path_recursive.py @@ -0,0 +1,65 @@ +""" +Check if two nodes are on the same path in a undirected graph. Use DFS and the concept of intime and outtime. +Intime - time when a node is visited for the first time +Outtime - time when a node is visited for the second time after all its children have been visited +For any pair of node if they are on the same path - +intime[u] < intime[v] and outtime[u] > outtime[v] +""" + +from collections import defaultdict + +class Graph: + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def dfs(self, vertex, intime, outtime, timer, visited): + visited[vertex] = True + timer += 1 + intime[vertex] = timer + + for v in self.graph[vertex]: + if visited[v] == False: + self.dfs(v, intime, outtime, timer, visited) + + timer += 1 + outtime[vertex] = timer + + + def on_same_path(self, u, v): + intime = [-1] * self.vertices + outtime = [-1] * self.vertices + timer = 0 + visited = [False] * self.vertices + + for vertex in self.graph: + if visited[vertex] == False: + self.dfs(vertex, intime, outtime, timer, visited) + + + if (intime[u] < intime[v] and outtime[u] > outtime[v]) \ + or (intime[v] < intime[u] and outtime[v] > outtime[u]): + return True + return False + + +g = Graph(9) +g.add_edge(0, 1) +g.add_edge(0, 2) +g.add_edge(2, 5) +g.add_edge(1, 3) +g.add_edge(1, 4) +g.add_edge(4, 6) +g.add_edge(4, 7) +g.add_edge(4, 8) + +print(g.on_same_path(0, 4)) +print(g.on_same_path(1, 8)) +print(g.on_same_path(1, 5)) \ No newline at end of file diff --git a/data_structures/graphs/shortest_path_unweighted_graph.py b/data_structures/graphs/shortest_path_unweighted_graph.py new file mode 100644 index 00000000..ca5cd60b --- /dev/null +++ b/data_structures/graphs/shortest_path_unweighted_graph.py @@ -0,0 +1,68 @@ +""" +Find the shortest path between two nodes in an unweighted undirected graph. Remember this +is about finding the shortest path, not the shortest distance. For shortest +distance you can simply calculate the level of nodes from the source vertex +and that will give the answer. For shortest path, use the concept of parents of +Bellman-Ford algorithm. + +Simply do a BFS and keep track of parents of each node. Then recursively print the +parents of destination node until the source node. +""" + +from collections import defaultdict + +class Graph: + + def __init__(self, vertices): + self.vertices = vertices + self.graph = defaultdict(list) + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def bfs(self, s): + parent = [-1] * self.vertices + visited = [False] * self.vertices + visited[s] = True + queue = [] + queue.append(s) + + while queue: + s = queue.pop(0) + + for i in self.graph[s]: + if visited[i] == False: + queue.append(i) + parent[i] = s + visited[i] = True + + return parent + + + def shortest_path(self, source, dest): + parent = self.bfs(source) + + while True: + print(dest, end=' ') + dest = parent[dest] + + if dest == source: + break + + +g = Graph(8) +g.add_edge(0, 1) +g.add_edge(0, 3) +g.add_edge(1, 2) +g.add_edge(3, 4) +g.add_edge(3, 7) +g.add_edge(4, 5) +g.add_edge(4, 6) +g.add_edge(4, 7) +g.add_edge(5, 6) +g.add_edge(6, 7) + +g.shortest_path(0, 7) \ No newline at end of file diff --git a/data_structures/graphs/two_cliques.py b/data_structures/graphs/two_cliques.py new file mode 100644 index 00000000..4eefba9c --- /dev/null +++ b/data_structures/graphs/two_cliques.py @@ -0,0 +1,75 @@ +""" +Find if a graph can be divided into two cliques + +A clique is a subgraph such that all vertices in it are completely +connected with each other + +Solution - +A bipartite graph is a graph which can be divided into two sets U and V such +that every edge from u or v has a destination in the other. +So take the complement of the graph (create edge where there is none, and destroy +edges where present) and if the complement is a bipartite, it means that there are edges +in U and V that connect each other. So complement's complement (i.e the original graph) will +not have these edges and hence it can be divided into two clique +""" + +from collections import defaultdict + +class Graph: + + + def __init__(self, vertices): + self.graph = defaultdict(list) + self.cgraph = defaultdict(list) + self.vertices = vertices + + + def add_edge(self, u, v): + self.graph[u].append(v) + self.graph[v].append(u) + + + def is_bipartite(self): + colors = [-1] * self.vertices + queue = [] + + for v in range(self.vertices): + if colors[v] == -1: + colors[v] = 1 + + queue.append(v) + + while queue: + s = queue.pop(0) + + for i in self.cgraph[s]: + if colors[i] == -1: + colors[i] = 1 - colors[s] + queue.append(i) + + elif colors[i] == colors[s]: + return False + + return True + + + def make_complement(self): + for src, dest in self.graph.items(): + for v in range(self.vertices): + if v not in dest and src != v: + self.cgraph[src].append(v) + self.cgraph[v].append(src) + + + def two_cliques(self): + self.make_complement() + return self.is_bipartite() + + +g = Graph(5) +g.add_edge(0, 3) +g.add_edge(3, 4) +g.add_edge(0, 1) +g.add_edge(1, 2) +g.add_edge(2, 0) +print(g.two_cliques()) \ No newline at end of file diff --git a/data_structures/hash/hash_table.py b/data_structures/hash/hash_table.py new file mode 100644 index 00000000..c1c05ced --- /dev/null +++ b/data_structures/hash/hash_table.py @@ -0,0 +1,29 @@ +""" +Create a hash table from scratch. Use chaining for hash collision +""" + +class HashTable: + + + def __init__(self): + self.hash_table = + + + def check_collision(self): + pass + + + def add_to_linked_list(self): + pass + + + def insert(self): + pass + + + def delete(self): + pass + + + def get(self): + pass diff --git a/data_structures/heap/heap_using_heapq.py b/data_structures/heap/heap_using_heapq.py new file mode 100644 index 00000000..7e28d109 --- /dev/null +++ b/data_structures/heap/heap_using_heapq.py @@ -0,0 +1,26 @@ +""" +Heap in python using heapq library function + +Note: by default, heapq creates a min-heap. To make it a +max-heap, add items after multiplying them by -1 +""" + +from heapq import heappop, heappush, heapify + +heap = [] +heapify(heap) + +heappush(heap, 10) +heappush(heap, 11) +heappush(heap, 2) +heappush(heap, 4) +heappush(heap, 14) +heappush(heap, 1) + +print('first element - ', heap[0]) +print('popping min element - ', heappop(heap)) +print('first element - ', heap[0]) + +# Heap prints as an array and can be access using indexes +print(heap) +print(heap[2]) diff --git a/data_structures/heap/kth_largest_element_in_stream.py b/data_structures/heap/kth_largest_element_in_stream.py new file mode 100644 index 00000000..dfc278e4 --- /dev/null +++ b/data_structures/heap/kth_largest_element_in_stream.py @@ -0,0 +1,61 @@ +""" +Use a priority queue - min heap + +as soon as the stream reaches a length of K, start finding the number + +Since we are using a priority queue (min heap), the minimum number will be +at the first index. O(1) time to extract it + +We have to make sure that the length of the stream does not go above K because +we to find the kth largest element which in terms of this heap means the smallest +element. For example lets say we have K = 4. So as soon as the stream reaches a +length of 4, we start to find the 4th largest number. Now as we are maintaining the +length of the array at 4, 4th largest number will mean the smallest number. Using this +we are designing the program. + +If the number entered in the stream is less than the current min, we dont take it as it +wont affect the result. +""" + +from heapq import heapify, heappop, heappush + +class Stream: + + def __init__(self, k): + self.heap = [] + self.stream = [] + self.k = k + self.curr_min = None + heapify(self.heap) + + + def insert(self, x): + self.stream.append(x) + + if len(self.heap) < self.k: # when the heap is empty or size is less than K + heappush(self.heap, x) + self.curr_min = self.heap[0] + else: + if x > self.curr_min: + heappop(self.heap) # remove the curr min element + heappush(self.heap, x) # insert x + self.curr_min = self.heap[0] + + + def find_kth_max(self): + if len(self.heap) == self.k: + print(f'{self.k}th max number - {self.heap[0]}') + +k = 3 +x = Stream(k) + +num = input() + +while num != 'q': + if num == 's': + print(f'Stream - {x.stream} | K - {k}') + else: + num = int(num) + x.insert(num) + x.find_kth_max() + num = input() diff --git a/data_structures/heap/max_heap.py b/data_structures/heap/max_heap.py new file mode 100644 index 00000000..29fcef70 --- /dev/null +++ b/data_structures/heap/max_heap.py @@ -0,0 +1,143 @@ +""" +Thing to remember - +* index of parent = i / 2 +* index of left child = 2i + 1 +* index of right child = 2i + 2 +""" + +class MaxHeap: + + def __init__(self, maxsize): + self.maxsize = maxsize + self.size = 0 # current number of elements in the heap + self.heap = [0] * self.maxsize + self.front = 0 + + + def parent(self, pos): + return (pos) // 2 + + + def left_child(self, pos): + return 2*pos + 1 + + + def right_child(self, pos): + return 2*pos + 2 + + + def mid_index(self): + return self.size // 2 + + + def last_index(self): + return self.size - 1 + + + def is_leaf(self, pos): + """ + Every node that is after the middle index of the heap + is a leaf node because their children cannot exist as the + index of children are twice their index as those indexes + do not exist in the heap + """ + if self.mid_index() <= pos <= self.last_index(): + return True + return False + + + def is_empty(self): + if self.size == 0: + return True + return False + + + def insert(self, value): + if self.is_empty(): # if the heap is empty + self.heap[self.front] = value + self.size += 1 + return + + if self.size >= self.maxsize: # if max size has been reached + return + + self.size += 1 + self.heap[self.last_index()] = value + + curr = self.last_index() + + # While inserting the element in the heap we have to + # make sure that the inserted element is always smaller + # than its parent. So basically here we are adjusting the + # position of the parent + while self.heap[curr] > self.heap[self.parent(curr)]: + self.swap(curr, self.parent(curr)) + curr = self.parent(curr) + + + def max_heapify(self, pos): + """ + This function will run whenever a node is non-leaf + node and smaller than its childen + """ + if not self.is_leaf(pos): + left = self.heap[self.left_child(pos)] + right = self.heap[self.right_child(pos)] + curr = self.heap[pos] + + if curr < left or curr < right: + + # This check is only to prevent out-of-index error + if left > right: + self.swap(pos, self.left_child(pos)) + self.max_heapify(self.left_child(pos)) + else: + self.swap(pos, self.right_child(pos)) + self.max_heapify(self.right_child(pos)) + + + def swap(self, x, y): + self.heap[x], self.heap[y] = self.heap[y], self.heap[x] + + + def pop_max(self): + max_element = self.heap[self.front] # max element is always at the front + self.heap[self.front] = self.heap[self.last_index()] # placing last element at the front + self.heap[self.last_index()] = 0 + self.size -= 1 # decrease size as one element has been popped + self.max_heapify(self.front) # heapify the heap again + return max_element + + + def print(self): + """ + Priting in inorder + """ + for i in range(0, self.mid_index() + 1): + parent = self.heap[i] + left = self.heap[self.left_child(i)] + right = self.heap[self.right_child(i)] + + print(f"Parent: {self.heap[i]}") + + if left: + print(f"Left child: {left}") + if right: + print(f"Right child: {right}") + + +if __name__ == '__main__': + max_heap = MaxHeap(15) + max_heap.insert(5) + max_heap.insert(3) + max_heap.insert(17) + max_heap.insert(10) + max_heap.insert(84) + max_heap.insert(19) + max_heap.insert(6) + max_heap.insert(22) + max_heap.insert(9) + + max_heap.print() + print('Max element is - ', max_heap.pop_max()) + max_heap.print() diff --git a/data_structures/heap/median_of_infinite_stream.py b/data_structures/heap/median_of_infinite_stream.py new file mode 100644 index 00000000..9ce834a8 --- /dev/null +++ b/data_structures/heap/median_of_infinite_stream.py @@ -0,0 +1,95 @@ +""" +Awesome explanation - https://youtu.be/1LkOrc-Le-Y + +Median is the middle element in a sorted array. The stream of input +integers can be in any order and we will have to store the integers in +such a way that the stream is maintained in an increasing order. + +So the main idea is that take an array and divide it into two parts of +equal length and the median will be as follows - + +* if the total number of integers in the stream is even, then both part will +have the same length. Hence the median will be the average of last element of the first +part (i.e max element of the first part) and the first element of the second part (i.e +min element of the second part) + +* if total number of integers in the stream is odd, add the extra element in the first part. +In this case, the median will be the last element of the first part + +Now we just have to maintain the order of both the parts of the array. Since we want the max +element from the first part, we can use a max heap there. And we can use min heap for the second +part as we need the min element from the second part + +Time complexity: + if the size of stream is N, we will have to iterate N times. LogN because insertion in heap + takes this much time. O(1) for getting the max or min element + + N * LogN +""" + +from heapq import heappush, heappop, heapify + +class MedianStream: + + def __init__(self): + self.stream = [] + self.min_heap = [] + self.max_heap = [] + heapify(self.max_heap) + heapify(self.min_heap) + self.curr_median = None + + + def add_number(self, num): + """ + min heap length <= maxheap length <= min heap length + 1 + """ + self.stream.append(num) + + if len(self.max_heap) == len(self.min_heap) == 0: + self.curr_median = num + + if len(self.max_heap) > len(self.min_heap): + if num < self.curr_median: + max_popped = -1 * heappop(self.max_heap) + heappush(self.min_heap, max_popped) + heappush(self.max_heap, -1 * num) + self.find_median('avg') + else: + heappush(self.min_heap, num) + self.find_median('avg') + else: + if num > self.curr_median: + # num will go to the min heap and the min element + # of min heap will go the max heap + min_popped = heappop(self.min_heap) + heappush(self.max_heap, -1 * min_popped) + heappush(self.min_heap, num) + self.find_median('max') + + else: + # num will go to the max heap + heappush(self.max_heap, -1 * num) + self.find_median('max') + + + def find_median(self, how): + if how == 'max': + self.curr_median = -1 * self.max_heap[0] + elif how == 'avg': + self.curr_median = (self.min_heap[0] + (-1 * self.max_heap[0])) / 2 + + +x = MedianStream() + +num = input() + +while num != 'q': + if num == 's': + print('Stream of integers - ', x.stream) + else: + num = int(num) + x.add_number(num) + print('Median - ', x.curr_median) + + num = input() diff --git a/data_structures/heap/min_heap.py b/data_structures/heap/min_heap.py new file mode 100644 index 00000000..069524d4 --- /dev/null +++ b/data_structures/heap/min_heap.py @@ -0,0 +1,122 @@ +""" +See max_heap.py for more detailed comments +""" + +class MinHeap: + + + def __init__(self, maxsize): + self.maxsize = maxsize + self.size = 0 + self.first = 0 + self.heap = [0] * self.maxsize + + + def is_empty(self): + return self.size == 0 + + + def is_leaf(self, pos): + if self.mid_index() <= pos <= self.last_index(): + return True + return False + + + def parent(self, pos): + return pos // 2 + + + def left_child(self, pos): + return 2*pos + 1 + + + def right_child(self, pos): + return 2*pos + 2 + + + def last_index(self): + return self.size - 1 + + + def mid_index(self): + return self.size // 2 + + + def swap(self, x, y): + self.heap[x], self.heap[y] = self.heap[y], self.heap[x] + + + def pop_min(self): + min_element = self.heap[self.first] + self.heap[self.first] = self.heap[self.last_index()] + self.heap[self.last_index()] = 0 + self.size -= 1 + self.min_heapify(self.first) + return min_element + + + def min_heapify(self, pos): + if not self.is_leaf(pos): + left = self.heap[self.left_child(pos)] + right = self.heap[self.right_child(pos)] + curr = self.heap[pos] + + if curr > left or curr > right: + + if left > right: + self.swap(pos, self.left_child(pos)) + self.min_heapify(self.left_child(pos)) + else: + self.swap(pos, self.right_child(pos)) + self.min_heapify(self.right_child(pos)) + + + def insert(self, element): + if self.is_empty(): + self.heap[self.first] = element + self.size += 1 + return + + if self.size >= self.maxsize: + return + + self.size += 1 + self.heap[self.last_index()] = element + + curr = self.last_index() + + while self.heap[curr] < self.heap[self.parent(curr)]: + self.swap(curr, self.parent(curr)) + curr = self.parent(curr) + + + def print(self): + for i in range(0, self.mid_index() + 1): + parent = self.heap[i] + left = self.heap[self.left_child(i)] + right = self.heap[self.right_child(i)] + + print(f"Parent: {self.heap[i]}") + + if left: + print(f"Left child: {left}") + if right: + print(f"Right child: {right}") + + +if __name__ == '__main__': + min_heap = MinHeap(15) + min_heap.insert(5) + min_heap.insert(3) + min_heap.insert(17) + min_heap.insert(10) + min_heap.insert(84) + min_heap.insert(19) + min_heap.insert(6) + min_heap.insert(22) + min_heap.insert(9) + + min_heap.print() + print(min_heap.heap) + print('Min element is - ', min_heap.pop_min()) + min_heap.print() diff --git a/data_structures/heap/sum_elements_range.py b/data_structures/heap/sum_elements_range.py new file mode 100644 index 00000000..4a3b6645 --- /dev/null +++ b/data_structures/heap/sum_elements_range.py @@ -0,0 +1,24 @@ +""" +Find the sum of elements between k1th and k2th smallest elements +""" + +from heapq import heappush, heapify, heappop + +heap = [20, 8, 22, 4, 12, 10, 14] +k1 = 3 +k2 = 6 + +heapify(heap) + +# extracting min k1 times + +for i in range(k1): + heappop(heap) + +# now do extract min k2 - (k1 + 1) times +s = 0 + +for i in range(k2 - k1 - 1): + s += heappop(heap) + +print(s) diff --git a/data_structures/linked_list/delete_last_occurrence.py b/data_structures/linked_list/delete_last_occurrence.py new file mode 100644 index 00000000..80951ab0 --- /dev/null +++ b/data_structures/linked_list/delete_last_occurrence.py @@ -0,0 +1,36 @@ +""" +Delete last occurence of a number in linked list. +""" + +class Node(): + + def __init__(self, val): + self.val = val + self.next = None + +def delete_last_occurrence(head, val): + if not head: + return None + + curr = head + prev = None + final_prev = None + final_occ = None + + while curr != None: + if curr.val == val: + final_prev = prev + final_occ = curr + + prev = curr + curr = curr.next + + + if final_occ: + # special case that checks for a 1 node list that equals the val + if final_prev: + final_prev.next = final_occ.next + else: + head = None + + return head \ No newline at end of file diff --git a/data_structures/linked_list/index.md b/data_structures/linked_list/index.md index 7c598aed..47a0074e 100644 --- a/data_structures/linked_list/index.md +++ b/data_structures/linked_list/index.md @@ -9,3 +9,4 @@ * [Odd-Even Arrangement](odd_even_arrangement.py) * [Merge Linked List](merge_linked_list.py) * [Remove Duplicates](remove_duplicates.py) +* [Delete Last Occurrence](delete_last_occurrence.py) diff --git a/data_structures/palindromic_tree/palindromic_tree.py b/data_structures/palindromic_tree/palindromic_tree.py new file mode 100644 index 00000000..47992793 --- /dev/null +++ b/data_structures/palindromic_tree/palindromic_tree.py @@ -0,0 +1,68 @@ +import typing + + +class Node: + def __init__(self): + self.next: typing.Dict[str, Node] = {} + self.frequency = 0 + self.length = 0 + self.suffix = None + + +class PalindromicTree: + def __init__(self): + self.null_root = Node() + self.imaginary_root = Node() + + self.null_root.length = 0 + self.null_root.suffix = self.imaginary_root + self.imaginary_root.length = -1 + self.imaginary_root.suffix = self.imaginary_root + + self.counter = 0 + self.all_palindromes: list[Node] = [] + self.previous = self.imaginary_root + + def add_letter(self, string, index): + while index - 1 - self.previous.length < 0 or string[index - 1 - self.previous.length] != string[index]: + self.previous = self.previous.suffix + + if self.previous.next.get(string[index]) is not None: + node = self.previous.next.get(string[index]) + node.frequency += 1 + self.previous = node + return + + new_node = Node() + + self.counter += 1 + new_node.frequency = 1 + new_node.length = self.previous.length + 2 + self.previous.next[string[index]] = new_node + + if new_node.length == 1: + new_node.suffix = self.null_root + self.previous = new_node + else: + self.previous = self.previous.suffix + while index - 1 - self.previous.length < 0 or string[index - 1 - self.previous.length] != string[index]: + self.previous = self.previous.suffix + new_node.suffix = self.previous.next[string[index]] + self.previous = new_node + self.all_palindromes.append(new_node) + + def how_many_palindromes(self): + all_nr = 0 + for node in reversed(self.all_palindromes): + node.suffix.frequency += node.frequency + all_nr += node.frequency + print(f"There are {self.counter} unique palindromes and {all_nr} in total") + + +tree = PalindromicTree() + +s = "abaxxaba" +for i in range(len(s)): + tree.add_letter(s, i) +tree.how_many_palindromes() + diff --git a/data_structures/queue/index.md b/data_structures/queue/index.md deleted file mode 100644 index 8506ea63..00000000 --- a/data_structures/queue/index.md +++ /dev/null @@ -1,3 +0,0 @@ -# Index of Queue - -[Queue](queue.py) \ No newline at end of file diff --git a/data_structures/segment_tree/seg_tree_max.py b/data_structures/segment_tree/seg_tree_max.py new file mode 100644 index 00000000..eb40f116 --- /dev/null +++ b/data_structures/segment_tree/seg_tree_max.py @@ -0,0 +1,44 @@ +import sys + +class MaxSegTree(): + + def __init__(self, n, arr = None): + self._t = [-sys.maxsize - 1] * (4 * n) + self._n = n + if arr: self._build(arr, 1 , 0 , n - 1) + + def _build(self, a, v, tl, tr): + if tl == tr: + self._t[v] = a[tl] + else: + tm = (tl + tr) // 2 + self._build(a, v*2, tl, tm) + self._build(a, v*2+1, tm+1, tr) + self._t[v] = max(self._t[v*2] ,self._t[v*2+1]) + + def _max_util(self, v, tl, tr, l, r): + if l > r: return -sys.maxsize - 1 + + if l == tl and r == tr : return self._t[v] + + tm = (tl + tr) // 2 + return max(self._max_util(v*2, tl, tm, l, min(r, tm)) ,self._max_util(v*2+1, tm+1, tr, max(l, tm+1), r)) + + def _update_util(self, v, tl, tr, pos, new_val): + if tl == tr: + self._t[v] = new_val + else: + tm = (tl + tr) // 2 + if pos <= tm: self._update_util(v*2, tl, tm, pos, new_val) + else: self._update_util(v*2+1, tm+1, tr, pos, new_val) + self._t[v] = max(self._t[v*2] , self._t[v*2+1]) + + def max_element(self, l, r): + return self._max_util(1 , 0 , self._n - 1 , l , r) + + def update(self, pos, new_val): + self._update_util(1 , 0 , self._n - 1 , pos , new_val) + + def add(self, pos, change): + value = self.max_element(pos , pos) + self._update_util(1 , 0 , self._n - 1 , pos , value + change) \ No newline at end of file diff --git a/data_structures/segment_tree/seg_tree_sum.py b/data_structures/segment_tree/seg_tree_sum.py new file mode 100644 index 00000000..ee7481f9 --- /dev/null +++ b/data_structures/segment_tree/seg_tree_sum.py @@ -0,0 +1,42 @@ +class SumSegTree(): + + def __init__(self, n, arr = None): + self._t = [0] * (4 * n) + self._n = n + if arr: self._build(arr, 1 , 0 , n - 1) + + def _build(self, a, v, tl, tr): + if tl == tr: + self._t[v] = a[tl] + else: + tm = (tl + tr) // 2 + self._build(a, v*2, tl, tm) + self._build(a, v*2+1, tm+1, tr) + self._t[v] = self._t[v*2] + self._t[v*2+1] + + def _sum_util(self, v, tl, tr, l, r): + if l > r: return 0 + + if l == tl and r == tr : return self._t[v] + + tm = (tl + tr) // 2 + return self._sum_util(v*2, tl, tm, l, min(r, tm)) + self._sum_util(v*2+1, tm+1, tr, max(l, tm+1), r) + + def _update_util(self, v, tl, tr, pos, new_val): + if tl == tr: + self._t[v] = new_val + else: + tm = (tl + tr) // 2 + if pos <= tm: self._update_util(v*2, tl, tm, pos, new_val) + else: self._update_util(v*2+1, tm+1, tr, pos, new_val) + self._t[v] = self._t[v*2] + self._t[v*2+1] + + def sum(self, l, r): + return self._sum_util(1 , 0 , self._n - 1 , l , r) + + def update(self, pos, new_val): + self._update_util(1 , 0 , self._n - 1 , pos , new_val) + + def add(self, pos, change): + value = self._sum(pos , pos) + self._update_util(1 , 0 , self._n - 1 , pos , value + change) \ No newline at end of file diff --git a/data_structures/stack/balanced_expression.py b/data_structures/stack/balanced_expression.py index 784666c0..60e3c506 100644 --- a/data_structures/stack/balanced_expression.py +++ b/data_structures/stack/balanced_expression.py @@ -1,24 +1,30 @@ -# simple program to check if an expression is balanced using stack stack = [] def checkBalanced(expr): for i in expr: if i == "{" or i == "[" or i == "(": stack.append(i) elif i == "}" or i == "]" or i == ")": - temp = stack.pop() - if i == "}" and temp != "{": + if not stack: return False - elif i == "]" and temp != "[": + top = stack.pop() + if i == "}" and top != "{": return False - elif i == ")" and temp != "(": + elif i == "]" and top != "[": return False + elif i == ")" and top != "(": + return False + else: + print("Invalid Expression") + return False - return True + if not len(stack): + return True + else: + return False # main function expr = input() -result = checkBalanced(expr) -if result: - print("Expression is balanced") +if not checkBalanced(expr): + print("Not Balanced") else: - print("Expression is not balanced") + print('Balanced') diff --git a/data_structures/stack/largest_rectangle_area_in_histogram.py b/data_structures/stack/largest_rectangle_area_in_histogram.py index bee42b5b..6aa55947 100644 --- a/data_structures/stack/largest_rectangle_area_in_histogram.py +++ b/data_structures/stack/largest_rectangle_area_in_histogram.py @@ -1,8 +1,7 @@ ''' Largest rectangle area in a histogram:: -Find the largest rectangular area possible in a given histogram where the largest rectangle can be made of a number of contiguous bars. For simplicity, assume that all bars have same width and the width is 1 unit. - - +Find the largest rectangular area possible in a given histogram where the largest rectangle can be made of a number of contiguous bars. +For simplicity, assume that all bars have same width and the width is 1 unit. ''' def max_area_histogram(histogram): @@ -12,36 +11,26 @@ def max_area_histogram(histogram): max_area = 0 # Initialize max area index = 0 - while index < len(histogram): - + + while index < len(histogram): if (not stack) or (histogram[stack[-1]] <= histogram[index]): stack.append(index) index += 1 - - else: - top_of_stack = stack.pop() - - area = (histogram[top_of_stack] * - ((index - stack[-1] - 1) - if stack else index)) - + area = (histogram[top_of_stack] * ((index - stack[-1] - 1) if stack else index)) max_area = max(max_area, area) while stack: - top_of_stack = stack.pop() - - area = (histogram[top_of_stack] * - ((index - stack[-1] - 1) - if stack else index)) + area = (histogram[top_of_stack] * ((index - stack[-1] - 1) if stack else index)) max_area = max(max_area, area) - return max_area + + hist = [4, 7, 1, 8, 4, 9, 5] print("Maximum area is", - max_area_histogram(hist)) +max_area_histogram(hist)) diff --git a/data_structures/stack/stack_using_list.py b/data_structures/stack/stack_using_list.py new file mode 100644 index 00000000..172892dd --- /dev/null +++ b/data_structures/stack/stack_using_list.py @@ -0,0 +1,25 @@ +class Stack(): + """ + Stack follows Last-In-First-Out methodology + """ + + def __init__(self): + self.entries = [] + + def size(self): + return len(self.entries) + + def push(self, val): + self.entries.append(val) + + def pop(self): + if self.size() > 0: + self.entries.pop(self.size() - 1) + + + + + + + + \ No newline at end of file diff --git a/data_structures/strings/KMP_Pattern_Search.py b/data_structures/strings/KMP_Pattern_Search.py new file mode 100644 index 00000000..37bfcd2c --- /dev/null +++ b/data_structures/strings/KMP_Pattern_Search.py @@ -0,0 +1,57 @@ +#Python program for KMP Algorithm +def KMP_pattern_search(pattern, text): + P = len(pattern) + Q = len(text) + + # create long_prefix_suffix[] that will hold the longest prefix suffix + # values for pattern + long_prefix_suffix = [0]*P + + # index for pattern[] + j=0 + + # preprocess the pattern (caluclate long_prefix_suffix[] array) + long_Prefix_Suffix_Array(pattern, P, long_prefix_suffix) + + # index for text[] + i=0 + + while i < Q: + if pattern[j] == text[i]: + i += 1 + j += 1 + + if j == P: + print("Pattern found at index " + str(i-j)) + j=long_prefix_suffix[j-1] + + # mismatch after j matches + elif i < Q and pattern[j] != text[i]: + if j != 0: + j=long_prefix_suffix[j-1] + else: + i += 1 + +def long_Prefix_Suffix_Array(pattern, P, long_prefix_suffix): + # length of the previous longest prefix suffix + l=0 + long_prefix_suffix[0]=0 + i=1 + + # the loop calculates long_prefix_suffix[i] for i = 1 to P-1 + while i < P: + if pattern[i] == pattern[l]: + l += 1 + long_prefix_suffix[i] = l + i += 1 + else: + if l != 0: + l=long_prefix_suffix[l-1] + else: + long_prefix_suffix[i] = 0 + i += 1 + +text = "ABABDABACDABABCABAB" +pattern = "ABABCABAB" +KMP_pattern_search(pattern, text) + diff --git a/data_structures/strings/unique_char_check.py b/data_structures/strings/unique_char_check.py new file mode 100644 index 00000000..3af3083c --- /dev/null +++ b/data_structures/strings/unique_char_check.py @@ -0,0 +1,24 @@ +""" +Question +You are given a string S, check if all characters are unique. + +SAMPLE INPUT 1 +abcd +SAMPLE OUTPUT 1 +True + +SAMPLE INPUT 2 +aabc +SAMPLE OUTPUT 2 +False +""" +from collections import Counter +def unique_char_check(S): + character_count = Counter(S) + + if len(character_count) == len(S): + return True + return False + +S = input() +print(unique_char_check(S)) \ No newline at end of file diff --git a/index_updater.py b/index_updater.py deleted file mode 100755 index 7f4778ae..00000000 --- a/index_updater.py +++ /dev/null @@ -1,28 +0,0 @@ -#!/usr/bin/env python3 - -import sys -import os - -# Terminal parameters. -if len(sys.argv[1:]) == 0: - print('Please, give a path') - sys.exit(1) -else: - mypath = sys.argv[1] - -# Going trought folders and subfolders. -for root, dirs, files in os.walk(mypath): - curr_files = list() - curr_folder = os.path.basename(root) - - for name in files: - if name.endswith('.py') and not name == '__init__.py' and not dirs: - curr_files.append(name) - - # If we have files, write "index.md" - if curr_files: - with open(os.path.join(root, 'index.md'), mode='w') as md_file: - md_file.write('# Index of {}\n\n'.format(curr_folder)) - for line in curr_files: - file_name = line.split('.')[0].replace('_',' ').title() - md_file.write('* [' + file_name + '](' +line + ')\n') diff --git a/logo/logo.png b/logo/logo.png deleted file mode 100644 index 28749cb7..00000000 Binary files a/logo/logo.png and /dev/null differ